Merge tag 'v3.13-rc4' into core/locking

Merge Linux 3.13-rc4, to refresh this rather old tree with the latest fixes. Signed-off-by: Ingo Molnar <mingo@kernel.org>
author: Ingo Molnar <mingo@kernel.org> 2013-12-17 09:27:08 -0500
committer: Ingo Molnar <mingo@kernel.org> 2013-12-17 09:27:08 -0500
commit: bb799d3b980eb803ca2da4a4eefbd9308f8d988a (patch)
tree: 69fbe0cd6d47b23a50f5e1d87bf7489532fae149 /fs
parent: 919fc6e34831d1c2b58bfb5ae261dc3facc9b269 (diff)
parent: 319e2e3f63c348a9b66db4667efa73178e18b17d (diff)
100 files changed, 2016 insertions, 909 deletions
diff --git a/fs/9p/vfs_dentry.c b/fs/9p/vfs_dentry.c
index f039b104a98e..b03dd23feda8 100644
--- a/fs/9p/vfs_dentry.c
+++ b/fs/9p/vfs_dentry.c
@@ -43,23 +43,6 @@
 #include "fid.h"
 /**
- * v9fs_dentry_delete - called when dentry refcount equals 0
- * @dentry:  dentry in question
- *
- * By returning 1 here we should remove cacheing of unused
- * dentry components.
- *
- */
-static int v9fs_dentry_delete(const struct dentry *dentry)
-{
-        p9_debug(P9_DEBUG_VFS, " dentry: %s (%p)\n",
-                 dentry->d_name.name, dentry);
-        return 1;
-}
-/**
 * v9fs_cached_dentry_delete - called when dentry refcount equals 0
 * @dentry:  dentry in question
 *
@@ -134,6 +117,6 @@ const struct dentry_operations v9fs_cached_dentry_operations = {
 };
 const struct dentry_operations v9fs_dentry_operations = {
-        .d_delete = v9fs_dentry_delete,
+        .d_delete = always_delete_dentry,
        .d_release = v9fs_dentry_release,
 };
diff --git a/fs/affs/Changes b/fs/affs/Changes
index a29409c1ffe0..b41c2c9792ff 100644
--- a/fs/affs/Changes
+++ b/fs/affs/Changes
@@ -91,7 +91,7 @@ more 2.4 fixes: [Roman Zippel]
 Version 3.11
 ------------
- Converted to use 2.3.x page cache [Dave Jones <dave@powertweak.com>]
+- Converted to use 2.3.x page cache [Dave Jones]
 - Corruption in truncate() bugfix [Ken Tyler <kent@werple.net.au>]
 Version 3.10
diff --git a/fs/aio.c b/fs/aio.c
index 823efcbb6ccd..6efb7f6cb22e 100644
--- a/fs/aio.c
+++ b/fs/aio.c
@@ -80,6 +80,8 @@ struct kioctx {
        struct percpu_ref       users;
        atomic_t                dead;
+        struct percpu_ref       reqs;
        unsigned long           user_id;
        struct __percpu kioctx_cpu *cpu;
@@ -107,7 +109,6 @@ struct kioctx {
        struct page             **ring_pages;
        long                    nr_pages;
-        struct rcu_head         rcu_head;
        struct work_struct      free_work;
        struct {
@@ -250,8 +251,10 @@ static void aio_free_ring(struct kioctx *ctx)
        put_aio_ring_file(ctx);
-        if (ctx->ring_pages && ctx->ring_pages != ctx->internal_pages)
+        if (ctx->ring_pages && ctx->ring_pages != ctx->internal_pages) {
                kfree(ctx->ring_pages);
+                ctx->ring_pages = NULL;
+        }
 }
 static int aio_ring_mmap(struct file *file, struct vm_area_struct *vma)
@@ -364,8 +367,10 @@ static int aio_setup_ring(struct kioctx *ctx)
        if (nr_pages > AIO_RING_PAGES) {
                ctx->ring_pages = kcalloc(nr_pages, sizeof(struct page *),
                                          GFP_KERNEL);
-                if (!ctx->ring_pages)
+                if (!ctx->ring_pages) {
+                        put_aio_ring_file(ctx);
                        return -ENOMEM;
+                }
        }
        ctx->mmap_size = nr_pages * PAGE_SIZE;
@@ -463,26 +468,34 @@ static int kiocb_cancel(struct kioctx *ctx, struct kiocb *kiocb)
        return cancel(kiocb);
 }
-static void free_ioctx_rcu(struct rcu_head *head)
+static void free_ioctx(struct work_struct *work)
 {
-        struct kioctx *ctx = container_of(head, struct kioctx, rcu_head);
+        struct kioctx *ctx = container_of(work, struct kioctx, free_work);
+        pr_debug("freeing %p\n", ctx);
+        aio_free_ring(ctx);
        free_percpu(ctx->cpu);
        kmem_cache_free(kioctx_cachep, ctx);
 }
+static void free_ioctx_reqs(struct percpu_ref *ref)
+{
+        struct kioctx *ctx = container_of(ref, struct kioctx, reqs);
+        INIT_WORK(&ctx->free_work, free_ioctx);
+        schedule_work(&ctx->free_work);
+}
 /*
 * When this function runs, the kioctx has been removed from the "hash table"
 * and ctx->users has dropped to 0, so we know no more kiocbs can be submitted -
 * now it's safe to cancel any that need to be.
 */
-static void free_ioctx(struct work_struct *work)
+static void free_ioctx_users(struct percpu_ref *ref)
 {
-        struct kioctx *ctx = container_of(work, struct kioctx, free_work);
+        struct kioctx *ctx = container_of(ref, struct kioctx, users);
-        struct aio_ring *ring;
        struct kiocb *req;
-        unsigned cpu, avail;
-        DEFINE_WAIT(wait);
        spin_lock_irq(&ctx->ctx_lock);
@@ -496,54 +509,8 @@ static void free_ioctx(struct work_struct *work)
        spin_unlock_irq(&ctx->ctx_lock);
-        for_each_possible_cpu(cpu) {
+        percpu_ref_kill(&ctx->reqs);
-                struct kioctx_cpu *kcpu = per_cpu_ptr(ctx->cpu, cpu);
+        percpu_ref_put(&ctx->reqs);
-                atomic_add(kcpu->reqs_available, &ctx->reqs_available);
-                kcpu->reqs_available = 0;
-        }
-        while (1) {
-                prepare_to_wait(&ctx->wait, &wait, TASK_UNINTERRUPTIBLE);
-                ring = kmap_atomic(ctx->ring_pages[0]);
-                avail = (ring->head <= ring->tail)
-                         ? ring->tail - ring->head
-                         : ctx->nr_events - ring->head + ring->tail;
-                atomic_add(avail, &ctx->reqs_available);
-                ring->head = ring->tail;
-                kunmap_atomic(ring);
-                if (atomic_read(&ctx->reqs_available) >= ctx->nr_events - 1)
-                        break;
-                schedule();
-        }
-        finish_wait(&ctx->wait, &wait);
-        WARN_ON(atomic_read(&ctx->reqs_available) > ctx->nr_events - 1);
-        aio_free_ring(ctx);
-        pr_debug("freeing %p\n", ctx);
-        /*
-         * Here the call_rcu() is between the wait_event() for reqs_active to
-         * hit 0, and freeing the ioctx.
-         *
-         * aio_complete() decrements reqs_active, but it has to touch the ioctx
-         * after to issue a wakeup so we use rcu.
-         */
-        call_rcu(&ctx->rcu_head, free_ioctx_rcu);
-}
-static void free_ioctx_ref(struct percpu_ref *ref)
-{
-        struct kioctx *ctx = container_of(ref, struct kioctx, users);
-        INIT_WORK(&ctx->free_work, free_ioctx);
-        schedule_work(&ctx->free_work);
 }
 static int ioctx_add_table(struct kioctx *ctx, struct mm_struct *mm)
@@ -602,6 +569,16 @@ static int ioctx_add_table(struct kioctx *ctx, struct mm_struct *mm)
        }
 }
+static void aio_nr_sub(unsigned nr)
+{
+        spin_lock(&aio_nr_lock);
+        if (WARN_ON(aio_nr - nr > aio_nr))
+                aio_nr = 0;
+        else
+                aio_nr -= nr;
+        spin_unlock(&aio_nr_lock);
+}
 /* ioctx_alloc
 *      Allocates and initializes an ioctx.  Returns an ERR_PTR if it failed.
 */
@@ -639,8 +616,11 @@ static struct kioctx *ioctx_alloc(unsigned nr_events)
        ctx->max_reqs = nr_events;
-        if (percpu_ref_init(&ctx->users, free_ioctx_ref))
+        if (percpu_ref_init(&ctx->users, free_ioctx_users))
-                goto out_freectx;
+                goto err;
+        if (percpu_ref_init(&ctx->reqs, free_ioctx_reqs))
+                goto err;
        spin_lock_init(&ctx->ctx_lock);
        spin_lock_init(&ctx->completion_lock);
@@ -651,10 +631,10 @@ static struct kioctx *ioctx_alloc(unsigned nr_events)
        ctx->cpu = alloc_percpu(struct kioctx_cpu);
        if (!ctx->cpu)
-                goto out_freeref;
+                goto err;
        if (aio_setup_ring(ctx) < 0)
-                goto out_freepcpu;
+                goto err;
        atomic_set(&ctx->reqs_available, ctx->nr_events - 1);
        ctx->req_batch = (ctx->nr_events - 1) / (num_possible_cpus() * 4);
@@ -666,7 +646,8 @@ static struct kioctx *ioctx_alloc(unsigned nr_events)
        if (aio_nr + nr_events > (aio_max_nr * 2UL) ||
            aio_nr + nr_events < aio_nr) {
                spin_unlock(&aio_nr_lock);
-                goto out_cleanup;
+                err = -EAGAIN;
+                goto err_ctx;
        }
        aio_nr += ctx->max_reqs;
        spin_unlock(&aio_nr_lock);
@@ -675,23 +656,20 @@ static struct kioctx *ioctx_alloc(unsigned nr_events)
        err = ioctx_add_table(ctx, mm);
        if (err)
-                goto out_cleanup_put;
+                goto err_cleanup;
        pr_debug("allocated ioctx %p[%ld]: mm=%p mask=0x%x\n",
                 ctx, ctx->user_id, mm, ctx->nr_events);
        return ctx;
-out_cleanup_put:
+err_cleanup:
-        percpu_ref_put(&ctx->users);
+        aio_nr_sub(ctx->max_reqs);
-out_cleanup:
+err_ctx:
-        err = -EAGAIN;
        aio_free_ring(ctx);
-out_freepcpu:
+err:
        free_percpu(ctx->cpu);
-out_freeref:
+        free_percpu(ctx->reqs.pcpu_count);
        free_percpu(ctx->users.pcpu_count);
-out_freectx:
-        put_aio_ring_file(ctx);
        kmem_cache_free(kioctx_cachep, ctx);
        pr_debug("error allocating ioctx %d\n", err);
        return ERR_PTR(err);
@@ -726,10 +704,7 @@ static void kill_ioctx(struct mm_struct *mm, struct kioctx *ctx)
                 * -EAGAIN with no ioctxs actually in use (as far as userspace
                 *  could tell).
                 */
-                spin_lock(&aio_nr_lock);
+                aio_nr_sub(ctx->max_reqs);
-                BUG_ON(aio_nr - ctx->max_reqs > aio_nr);
-                aio_nr -= ctx->max_reqs;
-                spin_unlock(&aio_nr_lock);
                if (ctx->mmap_size)
                        vm_munmap(ctx->mmap_base, ctx->mmap_size);
@@ -861,6 +836,8 @@ static inline struct kiocb *aio_get_req(struct kioctx *ctx)
        if (unlikely(!req))
                goto out_put;
+        percpu_ref_get(&ctx->reqs);
        req->ki_ctx = ctx;
        return req;
 out_put:
@@ -930,12 +907,6 @@ void aio_complete(struct kiocb *iocb, long res, long res2)
                return;
        }
-        /*
-         * Take rcu_read_lock() in case the kioctx is being destroyed, as we
-         * need to issue a wakeup after incrementing reqs_available.
-         */
-        rcu_read_lock();
        if (iocb->ki_list.next) {
                unsigned long flags;
@@ -1010,7 +981,7 @@ void aio_complete(struct kiocb *iocb, long res, long res2)
        if (waitqueue_active(&ctx->wait))
                wake_up(&ctx->wait);
-        rcu_read_unlock();
+        percpu_ref_put(&ctx->reqs);
 }
 EXPORT_SYMBOL(aio_complete);
@@ -1421,6 +1392,7 @@ static int io_submit_one(struct kioctx *ctx, struct iocb __user *user_iocb,
        return 0;
 out_put_req:
        put_reqs_available(ctx, 1);
+        percpu_ref_put(&ctx->reqs);
        kiocb_free(req);
        return ret;
 }
diff --git a/fs/bio.c b/fs/bio.c
index 2bdb4e25ee77..33d79a4eb92d 100644
--- a/fs/bio.c
+++ b/fs/bio.c
@@ -601,7 +601,7 @@ EXPORT_SYMBOL(bio_get_nr_vecs);
 static int __bio_add_page(struct request_queue *q, struct bio *bio, struct page
                          *page, unsigned int len, unsigned int offset,
-                          unsigned short max_sectors)
+                          unsigned int max_sectors)
 {
        int retried_segments = 0;
        struct bio_vec *bvec;
diff --git a/fs/btrfs/Kconfig b/fs/btrfs/Kconfig
index f9d5094e1029..aa976eced2d2 100644
--- a/fs/btrfs/Kconfig
+++ b/fs/btrfs/Kconfig
@@ -9,12 +9,17 @@ config BTRFS_FS
        select XOR_BLOCKS
        help
-          Btrfs is a new filesystem with extents, writable snapshotting,
+          Btrfs is a general purpose copy-on-write filesystem with extents,
-          support for multiple devices and many more features.
+          writable snapshotting, support for multiple devices and many more
+          features focused on fault tolerance, repair and easy administration.
-          Btrfs is highly experimental, and THE DISK FORMAT IS NOT YET
+          The filesystem disk format is no longer unstable, and it's not
-          FINALIZED.  You should say N here unless you are interested in
+          expected to change unless there are strong reasons to do so. If there
-          testing Btrfs with non-critical data.
+          is a format change, file systems with a unchanged format will
+          continue to be mountable and usable by newer kernels.
+          For more information, please see the web pages at
+          http://btrfs.wiki.kernel.org.
          To compile this file system support as a module, choose M here. The
          module will be called btrfs.
diff --git a/fs/btrfs/async-thread.c b/fs/btrfs/async-thread.c
index 8aec751fa464..c1e0b0caf9cc 100644
--- a/fs/btrfs/async-thread.c
+++ b/fs/btrfs/async-thread.c
@@ -495,6 +495,7 @@ static int __btrfs_start_workers(struct btrfs_workers *workers)
        spin_lock_irq(&workers->lock);
        if (workers->stopping) {
                spin_unlock_irq(&workers->lock);
+                ret = -EINVAL;
                goto fail_kthread;
        }
        list_add_tail(&worker->worker_list, &workers->idle_list);
diff --git a/fs/btrfs/check-integrity.c b/fs/btrfs/check-integrity.c
index e0aab4456974..131d82800b3a 100644
--- a/fs/btrfs/check-integrity.c
+++ b/fs/btrfs/check-integrity.c
@@ -77,6 +77,15 @@
 * the integrity of (super)-block write requests, do not
 * enable the config option BTRFS_FS_CHECK_INTEGRITY to
 * include and compile the integrity check tool.
+ *
+ * Expect millions of lines of information in the kernel log with an
+ * enabled check_int_print_mask. Therefore set LOG_BUF_SHIFT in the
+ * kernel config to at least 26 (which is 64MB). Usually the value is
+ * limited to 21 (which is 2MB) in init/Kconfig. The file needs to be
+ * changed like this before LOG_BUF_SHIFT can be set to a high value:
+ * config LOG_BUF_SHIFT
+ *       int "Kernel log buffer size (16 => 64KB, 17 => 128KB)"
+ *       range 12 30
 */
 #include <linux/sched.h>
@@ -124,6 +133,7 @@
 #define BTRFSIC_PRINT_MASK_INITIAL_DATABASE                     0x00000400
 #define BTRFSIC_PRINT_MASK_NUM_COPIES                           0x00000800
 #define BTRFSIC_PRINT_MASK_TREE_WITH_ALL_MIRRORS                0x00001000
+#define BTRFSIC_PRINT_MASK_SUBMIT_BIO_BH_VERBOSE                0x00002000
 struct btrfsic_dev_state;
 struct btrfsic_state;
@@ -323,7 +333,6 @@ static void btrfsic_release_block_ctx(struct btrfsic_block_data_ctx *block_ctx);
 static int btrfsic_read_block(struct btrfsic_state *state,
                              struct btrfsic_block_data_ctx *block_ctx);
 static void btrfsic_dump_database(struct btrfsic_state *state);
-static void btrfsic_complete_bio_end_io(struct bio *bio, int err);
 static int btrfsic_test_for_metadata(struct btrfsic_state *state,
                                     char **datav, unsigned int num_pages);
 static void btrfsic_process_written_block(struct btrfsic_dev_state *dev_state,
@@ -1677,7 +1686,6 @@ static int btrfsic_read_block(struct btrfsic_state *state,
        for (i = 0; i < num_pages;) {
                struct bio *bio;
                unsigned int j;
-                DECLARE_COMPLETION_ONSTACK(complete);
                bio = btrfs_io_bio_alloc(GFP_NOFS, num_pages - i);
                if (!bio) {
@@ -1688,8 +1696,6 @@ static int btrfsic_read_block(struct btrfsic_state *state,
                }
                bio->bi_bdev = block_ctx->dev->bdev;
                bio->bi_sector = dev_bytenr >> 9;
-                bio->bi_end_io = btrfsic_complete_bio_end_io;
-                bio->bi_private = &complete;
                for (j = i; j < num_pages; j++) {
                        ret = bio_add_page(bio, block_ctx->pagev[j],
@@ -1702,12 +1708,7 @@ static int btrfsic_read_block(struct btrfsic_state *state,
                               "btrfsic: error, failed to add a single page!\n");
                        return -1;
                }
-                submit_bio(READ, bio);
+                if (submit_bio_wait(READ, bio)) {
-                /* this will also unplug the queue */
-                wait_for_completion(&complete);
-                if (!test_bit(BIO_UPTODATE, &bio->bi_flags)) {
                        printk(KERN_INFO
                               "btrfsic: read error at logical %llu dev %s!\n",
                               block_ctx->start, block_ctx->dev->name);
@@ -1730,11 +1731,6 @@ static int btrfsic_read_block(struct btrfsic_state *state,
        return block_ctx->len;
 }
-static void btrfsic_complete_bio_end_io(struct bio *bio, int err)
-{
-        complete((struct completion *)bio->bi_private);
-}
 static void btrfsic_dump_database(struct btrfsic_state *state)
 {
        struct list_head *elem_all;
@@ -2998,14 +2994,12 @@ int btrfsic_submit_bh(int rw, struct buffer_head *bh)
        return submit_bh(rw, bh);
 }
-void btrfsic_submit_bio(int rw, struct bio *bio)
+static void __btrfsic_submit_bio(int rw, struct bio *bio)
 {
        struct btrfsic_dev_state *dev_state;
-        if (!btrfsic_is_initialized) {
+        if (!btrfsic_is_initialized)
-                submit_bio(rw, bio);
                return;
-        }
        mutex_lock(&btrfsic_mutex);
        /* since btrfsic_submit_bio() is also called before
@@ -3015,6 +3009,7 @@ void btrfsic_submit_bio(int rw, struct bio *bio)
            (rw & WRITE) && NULL != bio->bi_io_vec) {
                unsigned int i;
                u64 dev_bytenr;
+                u64 cur_bytenr;
                int bio_is_patched;
                char **mapped_datav;
@@ -3033,6 +3028,7 @@ void btrfsic_submit_bio(int rw, struct bio *bio)
                                       GFP_NOFS);
                if (!mapped_datav)
                        goto leave;
+                cur_bytenr = dev_bytenr;
                for (i = 0; i < bio->bi_vcnt; i++) {
                        BUG_ON(bio->bi_io_vec[i].bv_len != PAGE_CACHE_SIZE);
                        mapped_datav[i] = kmap(bio->bi_io_vec[i].bv_page);
@@ -3044,16 +3040,13 @@ void btrfsic_submit_bio(int rw, struct bio *bio)
                                kfree(mapped_datav);
                                goto leave;
                        }
-                        if ((BTRFSIC_PRINT_MASK_SUBMIT_BIO_BH |
+                        if (dev_state->state->print_mask &
-                             BTRFSIC_PRINT_MASK_VERBOSE) ==
+                            BTRFSIC_PRINT_MASK_SUBMIT_BIO_BH_VERBOSE)
-                            (dev_state->state->print_mask &
-                             (BTRFSIC_PRINT_MASK_SUBMIT_BIO_BH |
-                              BTRFSIC_PRINT_MASK_VERBOSE)))
                                printk(KERN_INFO
-                                       "#%u: page=%p, len=%u, offset=%u\n",
+                                       "#%u: bytenr=%llu, len=%u, offset=%u\n",
-                                       i, bio->bi_io_vec[i].bv_page,
+                                       i, cur_bytenr, bio->bi_io_vec[i].bv_len,
-                                       bio->bi_io_vec[i].bv_len,
                                       bio->bi_io_vec[i].bv_offset);
+                        cur_bytenr += bio->bi_io_vec[i].bv_len;
                }
                btrfsic_process_written_block(dev_state, dev_bytenr,
                                              mapped_datav, bio->bi_vcnt,
@@ -3097,10 +3090,20 @@ void btrfsic_submit_bio(int rw, struct bio *bio)
        }
 leave:
        mutex_unlock(&btrfsic_mutex);
+}
+void btrfsic_submit_bio(int rw, struct bio *bio)
+{
+        __btrfsic_submit_bio(rw, bio);
        submit_bio(rw, bio);
 }
+int btrfsic_submit_bio_wait(int rw, struct bio *bio)
+{
+        __btrfsic_submit_bio(rw, bio);
+        return submit_bio_wait(rw, bio);
+}
 int btrfsic_mount(struct btrfs_root *root,
                  struct btrfs_fs_devices *fs_devices,
                  int including_extent_data, u32 print_mask)
diff --git a/fs/btrfs/check-integrity.h b/fs/btrfs/check-integrity.h
index 8b59175cc502..13b8566c97ab 100644
--- a/fs/btrfs/check-integrity.h
+++ b/fs/btrfs/check-integrity.h
@@ -22,9 +22,11 @@
 #ifdef CONFIG_BTRFS_FS_CHECK_INTEGRITY
 int btrfsic_submit_bh(int rw, struct buffer_head *bh);
 void btrfsic_submit_bio(int rw, struct bio *bio);
+int btrfsic_submit_bio_wait(int rw, struct bio *bio);
 #else
 #define btrfsic_submit_bh submit_bh
 #define btrfsic_submit_bio submit_bio
+#define btrfsic_submit_bio_wait submit_bio_wait
 #endif
 int btrfsic_mount(struct btrfs_root *root,
diff --git a/fs/btrfs/ctree.h b/fs/btrfs/ctree.h
index f9aeb2759a64..54ab86127f7a 100644
--- a/fs/btrfs/ctree.h
+++ b/fs/btrfs/ctree.h
@@ -3613,9 +3613,6 @@ int btrfs_csum_file_blocks(struct btrfs_trans_handle *trans,
                           struct btrfs_ordered_sum *sums);
 int btrfs_csum_one_bio(struct btrfs_root *root, struct inode *inode,
                       struct bio *bio, u64 file_start, int contig);
-int btrfs_csum_truncate(struct btrfs_trans_handle *trans,
-                        struct btrfs_root *root, struct btrfs_path *path,
-                        u64 isize);
 int btrfs_lookup_csums_range(struct btrfs_root *root, u64 start, u64 end,
                             struct list_head *list, int search_commit);
 /* inode.c */
@@ -3744,9 +3741,6 @@ void btrfs_cleanup_defrag_inodes(struct btrfs_fs_info *fs_info);
 int btrfs_sync_file(struct file *file, loff_t start, loff_t end, int datasync);
 void btrfs_drop_extent_cache(struct inode *inode, u64 start, u64 end,
                             int skip_pinned);
-int btrfs_replace_extent_cache(struct inode *inode, struct extent_map *replace,
-                               u64 start, u64 end, int skip_pinned,
-                               int modified);
 extern const struct file_operations btrfs_file_operations;
 int __btrfs_drop_extents(struct btrfs_trans_handle *trans,
                         struct btrfs_root *root, struct inode *inode,
diff --git a/fs/btrfs/dev-replace.c b/fs/btrfs/dev-replace.c
index 342f9fd411e3..2cfc3dfff64f 100644
--- a/fs/btrfs/dev-replace.c
+++ b/fs/btrfs/dev-replace.c
@@ -366,7 +366,7 @@ int btrfs_dev_replace_start(struct btrfs_root *root,
        dev_replace->tgtdev = tgt_device;
        printk_in_rcu(KERN_INFO
-                      "btrfs: dev_replace from %s (devid %llu) to %s) started\n",
+                      "btrfs: dev_replace from %s (devid %llu) to %s started\n",
                      src_device->missing ? "<missing disk>" :
                        rcu_str_deref(src_device->name),
                      src_device->devid,
diff --git a/fs/btrfs/disk-io.c b/fs/btrfs/disk-io.c
index 4c4ed0bb3da1..8072cfa8a3b1 100644
--- a/fs/btrfs/disk-io.c
+++ b/fs/btrfs/disk-io.c
@@ -3517,7 +3517,6 @@ int btrfs_cleanup_fs_roots(struct btrfs_fs_info *fs_info)
 int btrfs_commit_super(struct btrfs_root *root)
 {
        struct btrfs_trans_handle *trans;
-        int ret;
        mutex_lock(&root->fs_info->cleaner_mutex);
        btrfs_run_delayed_iputs(root);
@@ -3531,25 +3530,7 @@ int btrfs_commit_super(struct btrfs_root *root)
        trans = btrfs_join_transaction(root);
        if (IS_ERR(trans))
                return PTR_ERR(trans);
-        ret = btrfs_commit_transaction(trans, root);
+        return btrfs_commit_transaction(trans, root);
-        if (ret)
-                return ret;
-        /* run commit again to drop the original snapshot */
-        trans = btrfs_join_transaction(root);
-        if (IS_ERR(trans))
-                return PTR_ERR(trans);
-        ret = btrfs_commit_transaction(trans, root);
-        if (ret)
-                return ret;
-        ret = btrfs_write_and_wait_transaction(NULL, root);
-        if (ret) {
-                btrfs_error(root->fs_info, ret,
-                            "Failed to sync btree inode to disk.");
-                return ret;
-        }
-        ret = write_ctree_super(NULL, root, 0);
-        return ret;
 }
 int close_ctree(struct btrfs_root *root)
diff --git a/fs/btrfs/extent-tree.c b/fs/btrfs/extent-tree.c
index 45d98d01028f..9c01509dd8ab 100644
--- a/fs/btrfs/extent-tree.c
+++ b/fs/btrfs/extent-tree.c
@@ -767,20 +767,19 @@ int btrfs_lookup_extent_info(struct btrfs_trans_handle *trans,
        if (!path)
                return -ENOMEM;
-        if (metadata) {
-                key.objectid = bytenr;
-                key.type = BTRFS_METADATA_ITEM_KEY;
-                key.offset = offset;
-        } else {
-                key.objectid = bytenr;
-                key.type = BTRFS_EXTENT_ITEM_KEY;
-                key.offset = offset;
-        }
        if (!trans) {
                path->skip_locking = 1;
                path->search_commit_root = 1;
        }
+search_again:
+        key.objectid = bytenr;
+        key.offset = offset;
+        if (metadata)
+                key.type = BTRFS_METADATA_ITEM_KEY;
+        else
+                key.type = BTRFS_EXTENT_ITEM_KEY;
 again:
        ret = btrfs_search_slot(trans, root->fs_info->extent_root,
                                &key, path, 0, 0);
@@ -788,7 +787,6 @@ again:
                goto out_free;
        if (ret > 0 && metadata && key.type == BTRFS_METADATA_ITEM_KEY) {
-                metadata = 0;
                if (path->slots[0]) {
                        path->slots[0]--;
                        btrfs_item_key_to_cpu(path->nodes[0], &key,
@@ -855,7 +853,7 @@ again:
                        mutex_lock(&head->mutex);
                        mutex_unlock(&head->mutex);
                        btrfs_put_delayed_ref(&head->node);
-                        goto again;
+                        goto search_again;
                }
                if (head->extent_op && head->extent_op->update_flags)
                        extent_flags |= head->extent_op->flags_to_set;
diff --git a/fs/btrfs/extent_io.c b/fs/btrfs/extent_io.c
index 856bc2b2192c..ff43802a7c88 100644
--- a/fs/btrfs/extent_io.c
+++ b/fs/btrfs/extent_io.c
@@ -1952,11 +1952,6 @@ static int free_io_failure(struct inode *inode, struct io_failure_record *rec,
        return err;
 }
-static void repair_io_failure_callback(struct bio *bio, int err)
-{
-        complete(bio->bi_private);
-}
 /*
 * this bypasses the standard btrfs submit functions deliberately, as
 * the standard behavior is to write all copies in a raid setup. here we only
@@ -1973,13 +1968,13 @@ int repair_io_failure(struct btrfs_fs_info *fs_info, u64 start,
 {
        struct bio *bio;
        struct btrfs_device *dev;
-        DECLARE_COMPLETION_ONSTACK(compl);
        u64 map_length = 0;
        u64 sector;
        struct btrfs_bio *bbio = NULL;
        struct btrfs_mapping_tree *map_tree = &fs_info->mapping_tree;
        int ret;
+        ASSERT(!(fs_info->sb->s_flags & MS_RDONLY));
        BUG_ON(!mirror_num);
        /* we can't repair anything in raid56 yet */
@@ -1989,8 +1984,6 @@ int repair_io_failure(struct btrfs_fs_info *fs_info, u64 start,
        bio = btrfs_io_bio_alloc(GFP_NOFS, 1);
        if (!bio)
                return -EIO;
-        bio->bi_private = &compl;
-        bio->bi_end_io = repair_io_failure_callback;
        bio->bi_size = 0;
        map_length = length;
@@ -2011,10 +2004,8 @@ int repair_io_failure(struct btrfs_fs_info *fs_info, u64 start,
        }
        bio->bi_bdev = dev->bdev;
        bio_add_page(bio, page, length, start - page_offset(page));
-        btrfsic_submit_bio(WRITE_SYNC, bio);
-        wait_for_completion(&compl);
-        if (!test_bit(BIO_UPTODATE, &bio->bi_flags)) {
+        if (btrfsic_submit_bio_wait(WRITE_SYNC, bio)) {
                /* try to remap that extent elsewhere? */
                bio_put(bio);
                btrfs_dev_stat_inc_and_print(dev, BTRFS_DEV_STAT_WRITE_ERRS);
@@ -2036,6 +2027,9 @@ int repair_eb_io_failure(struct btrfs_root *root, struct extent_buffer *eb,
        unsigned long i, num_pages = num_extent_pages(eb->start, eb->len);
        int ret = 0;
+        if (root->fs_info->sb->s_flags & MS_RDONLY)
+                return -EROFS;
        for (i = 0; i < num_pages; i++) {
                struct page *p = extent_buffer_page(eb, i);
                ret = repair_io_failure(root->fs_info, start, PAGE_CACHE_SIZE,
@@ -2057,12 +2051,12 @@ static int clean_io_failure(u64 start, struct page *page)
        u64 private;
        u64 private_failure;
        struct io_failure_record *failrec;
-        struct btrfs_fs_info *fs_info;
+        struct inode *inode = page->mapping->host;
+        struct btrfs_fs_info *fs_info = BTRFS_I(inode)->root->fs_info;
        struct extent_state *state;
        int num_copies;
        int did_repair = 0;
        int ret;
-        struct inode *inode = page->mapping->host;
        private = 0;
        ret = count_range_bits(&BTRFS_I(inode)->io_failure_tree, &private,
@@ -2085,6 +2079,8 @@ static int clean_io_failure(u64 start, struct page *page)
                did_repair = 1;
                goto out;
        }
+        if (fs_info->sb->s_flags & MS_RDONLY)
+                goto out;
        spin_lock(&BTRFS_I(inode)->io_tree.lock);
        state = find_first_extent_bit_state(&BTRFS_I(inode)->io_tree,
@@ -2094,7 +2090,6 @@ static int clean_io_failure(u64 start, struct page *page)
        if (state && state->start <= failrec->start &&
            state->end >= failrec->start + failrec->len - 1) {
-                fs_info = BTRFS_I(inode)->root->fs_info;
                num_copies = btrfs_num_copies(fs_info, failrec->logical,
                                              failrec->len);
                if (num_copies > 1)  {
diff --git a/fs/btrfs/inode.c b/fs/btrfs/inode.c
index da8d2f696ac5..f1a77449d032 100644
--- a/fs/btrfs/inode.c
+++ b/fs/btrfs/inode.c
@@ -2129,7 +2129,8 @@ static noinline bool record_extent_backrefs(struct btrfs_path *path,
                                                  old->extent_offset, fs_info,
                                                  path, record_one_backref,
                                                  old);
-                BUG_ON(ret < 0 && ret != -ENOENT);
+                if (ret < 0 && ret != -ENOENT)
+                        return false;
                /* no backref to be processed for this extent */
                if (!old->count) {
@@ -6186,8 +6187,7 @@ insert:
        write_unlock(&em_tree->lock);
 out:
-        if (em)
+        trace_btrfs_get_extent(root, em);
-                trace_btrfs_get_extent(root, em);
        if (path)
                btrfs_free_path(path);
diff --git a/fs/btrfs/ioctl.c b/fs/btrfs/ioctl.c
index a111622598b0..21da5762b0b1 100644
--- a/fs/btrfs/ioctl.c
+++ b/fs/btrfs/ioctl.c
@@ -2121,7 +2121,7 @@ static noinline int btrfs_ioctl_snap_destroy(struct file *file,
        err = mutex_lock_killable_nested(&dir->i_mutex, I_MUTEX_PARENT);
        if (err == -EINTR)
-                goto out;
+                goto out_drop_write;
        dentry = lookup_one_len(vol_args->name, parent, namelen);
        if (IS_ERR(dentry)) {
                err = PTR_ERR(dentry);
@@ -2284,6 +2284,7 @@ out_dput:
        dput(dentry);
 out_unlock_dir:
        mutex_unlock(&dir->i_mutex);
+out_drop_write:
        mnt_drop_write_file(file);
 out:
        kfree(vol_args);
diff --git a/fs/btrfs/ordered-data.c b/fs/btrfs/ordered-data.c
index 25a8f3812f14..69582d5b69d1 100644
--- a/fs/btrfs/ordered-data.c
+++ b/fs/btrfs/ordered-data.c
@@ -638,6 +638,7 @@ void btrfs_wait_ordered_roots(struct btrfs_fs_info *fs_info, int nr)
                        WARN_ON(nr < 0);
                }
        }
+        list_splice_tail(&splice, &fs_info->ordered_roots);
        spin_unlock(&fs_info->ordered_root_lock);
 }
@@ -803,7 +804,7 @@ int btrfs_wait_ordered_range(struct inode *inode, u64 start, u64 len)
                        btrfs_put_ordered_extent(ordered);
                        break;
                }
-                if (ordered->file_offset + ordered->len < start) {
+                if (ordered->file_offset + ordered->len <= start) {
                        btrfs_put_ordered_extent(ordered);
                        break;
                }
diff --git a/fs/btrfs/relocation.c b/fs/btrfs/relocation.c
index ce459a7cb16d..429c73c374b8 100644
--- a/fs/btrfs/relocation.c
+++ b/fs/btrfs/relocation.c
@@ -571,7 +571,9 @@ static int is_cowonly_root(u64 root_objectid)
            root_objectid == BTRFS_CHUNK_TREE_OBJECTID ||
            root_objectid == BTRFS_DEV_TREE_OBJECTID ||
            root_objectid == BTRFS_TREE_LOG_OBJECTID ||
-            root_objectid == BTRFS_CSUM_TREE_OBJECTID)
+            root_objectid == BTRFS_CSUM_TREE_OBJECTID ||
+            root_objectid == BTRFS_UUID_TREE_OBJECTID ||
+            root_objectid == BTRFS_QUOTA_TREE_OBJECTID)
                return 1;
        return 0;
 }
@@ -1264,10 +1266,10 @@ static int __must_check __add_reloc_root(struct btrfs_root *root)
 }
 /*
- * helper to update/delete the 'address of tree root -> reloc tree'
+ * helper to delete the 'address of tree root -> reloc tree'
 * mapping
 */
-static int __update_reloc_root(struct btrfs_root *root, int del)
+static void __del_reloc_root(struct btrfs_root *root)
 {
        struct rb_node *rb_node;
        struct mapping_node *node = NULL;
@@ -1275,7 +1277,7 @@ static int __update_reloc_root(struct btrfs_root *root, int del)
        spin_lock(&rc->reloc_root_tree.lock);
        rb_node = tree_search(&rc->reloc_root_tree.rb_root,
-                              root->commit_root->start);
+                              root->node->start);
        if (rb_node) {
                node = rb_entry(rb_node, struct mapping_node, rb_node);
                rb_erase(&node->rb_node, &rc->reloc_root_tree.rb_root);
@@ -1283,23 +1285,45 @@ static int __update_reloc_root(struct btrfs_root *root, int del)
        spin_unlock(&rc->reloc_root_tree.lock);
        if (!node)
-                return 0;
+                return;
        BUG_ON((struct btrfs_root *)node->data != root);
-        if (!del) {
+        spin_lock(&root->fs_info->trans_lock);
-                spin_lock(&rc->reloc_root_tree.lock);
+        list_del_init(&root->root_list);
-                node->bytenr = root->node->start;
+        spin_unlock(&root->fs_info->trans_lock);
-                rb_node = tree_insert(&rc->reloc_root_tree.rb_root,
+        kfree(node);
-                                      node->bytenr, &node->rb_node);
+}
-                spin_unlock(&rc->reloc_root_tree.lock);
-                if (rb_node)
+/*
-                        backref_tree_panic(rb_node, -EEXIST, node->bytenr);
+ * helper to update the 'address of tree root -> reloc tree'
-        } else {
+ * mapping
-                spin_lock(&root->fs_info->trans_lock);
+ */
-                list_del_init(&root->root_list);
+static int __update_reloc_root(struct btrfs_root *root, u64 new_bytenr)
-                spin_unlock(&root->fs_info->trans_lock);
+{
-                kfree(node);
+        struct rb_node *rb_node;
+        struct mapping_node *node = NULL;
+        struct reloc_control *rc = root->fs_info->reloc_ctl;
+        spin_lock(&rc->reloc_root_tree.lock);
+        rb_node = tree_search(&rc->reloc_root_tree.rb_root,
+                              root->node->start);
+        if (rb_node) {
+                node = rb_entry(rb_node, struct mapping_node, rb_node);
+                rb_erase(&node->rb_node, &rc->reloc_root_tree.rb_root);
        }
+        spin_unlock(&rc->reloc_root_tree.lock);
+        if (!node)
+                return 0;
+        BUG_ON((struct btrfs_root *)node->data != root);
+        spin_lock(&rc->reloc_root_tree.lock);
+        node->bytenr = new_bytenr;
+        rb_node = tree_insert(&rc->reloc_root_tree.rb_root,
+                              node->bytenr, &node->rb_node);
+        spin_unlock(&rc->reloc_root_tree.lock);
+        if (rb_node)
+                backref_tree_panic(rb_node, -EEXIST, node->bytenr);
        return 0;
 }
@@ -1420,7 +1444,6 @@ int btrfs_update_reloc_root(struct btrfs_trans_handle *trans,
 {
        struct btrfs_root *reloc_root;
        struct btrfs_root_item *root_item;
-        int del = 0;
        int ret;
        if (!root->reloc_root)
@@ -1432,11 +1455,9 @@ int btrfs_update_reloc_root(struct btrfs_trans_handle *trans,
        if (root->fs_info->reloc_ctl->merge_reloc_tree &&
            btrfs_root_refs(root_item) == 0) {
                root->reloc_root = NULL;
-                del = 1;
+                __del_reloc_root(reloc_root);
        }
-        __update_reloc_root(reloc_root, del);
        if (reloc_root->commit_root != reloc_root->node) {
                btrfs_set_root_node(root_item, reloc_root->node);
                free_extent_buffer(reloc_root->commit_root);
@@ -2287,7 +2308,7 @@ void free_reloc_roots(struct list_head *list)
        while (!list_empty(list)) {
                reloc_root = list_entry(list->next, struct btrfs_root,
                                        root_list);
-                __update_reloc_root(reloc_root, 1);
+                __del_reloc_root(reloc_root);
                free_extent_buffer(reloc_root->node);
                free_extent_buffer(reloc_root->commit_root);
                kfree(reloc_root);
@@ -2332,7 +2353,7 @@ again:
                        ret = merge_reloc_root(rc, root);
                        if (ret) {
-                                __update_reloc_root(reloc_root, 1);
+                                __del_reloc_root(reloc_root);
                                free_extent_buffer(reloc_root->node);
                                free_extent_buffer(reloc_root->commit_root);
                                kfree(reloc_root);
@@ -2388,6 +2409,13 @@ out:
                btrfs_std_error(root->fs_info, ret);
                if (!list_empty(&reloc_roots))
                        free_reloc_roots(&reloc_roots);
+                /* new reloc root may be added */
+                mutex_lock(&root->fs_info->reloc_mutex);
+                list_splice_init(&rc->reloc_roots, &reloc_roots);
+                mutex_unlock(&root->fs_info->reloc_mutex);
+                if (!list_empty(&reloc_roots))
+                        free_reloc_roots(&reloc_roots);
        }
        BUG_ON(!RB_EMPTY_ROOT(&rc->reloc_root_tree.rb_root));
@@ -4522,6 +4550,11 @@ int btrfs_reloc_cow_block(struct btrfs_trans_handle *trans,
        BUG_ON(rc->stage == UPDATE_DATA_PTRS &&
               root->root_key.objectid == BTRFS_DATA_RELOC_TREE_OBJECTID);
+        if (root->root_key.objectid == BTRFS_TREE_RELOC_OBJECTID) {
+                if (buf == root->node)
+                        __update_reloc_root(root, cow->start);
+        }
        level = btrfs_header_level(buf);
        if (btrfs_header_generation(buf) <=
            btrfs_root_last_snapshot(&root->root_item))
diff --git a/fs/btrfs/scrub.c b/fs/btrfs/scrub.c
index 2544805544f0..1fd3f33c330a 100644
--- a/fs/btrfs/scrub.c
+++ b/fs/btrfs/scrub.c
@@ -208,7 +208,6 @@ static void scrub_recheck_block_checksum(struct btrfs_fs_info *fs_info,
                                         int is_metadata, int have_csum,
                                         const u8 *csum, u64 generation,
                                         u16 csum_size);
-static void scrub_complete_bio_end_io(struct bio *bio, int err);
 static int scrub_repair_block_from_good_copy(struct scrub_block *sblock_bad,
                                             struct scrub_block *sblock_good,
                                             int force_write);
@@ -938,8 +937,10 @@ static int scrub_handle_errored_block(struct scrub_block *sblock_to_check)
                                BTRFS_DEV_STAT_CORRUPTION_ERRS);
        }
-        if (sctx->readonly && !sctx->is_dev_replace)
+        if (sctx->readonly) {
-                goto did_not_correct_error;
+                ASSERT(!sctx->is_dev_replace);
+                goto out;
+        }
        if (!is_metadata && !have_csum) {
                struct scrub_fixup_nodatasum *fixup_nodatasum;
@@ -1292,7 +1293,6 @@ static void scrub_recheck_block(struct btrfs_fs_info *fs_info,
        for (page_num = 0; page_num < sblock->page_count; page_num++) {
                struct bio *bio;
                struct scrub_page *page = sblock->pagev[page_num];
-                DECLARE_COMPLETION_ONSTACK(complete);
                if (page->dev->bdev == NULL) {
                        page->io_error = 1;
@@ -1309,18 +1309,11 @@ static void scrub_recheck_block(struct btrfs_fs_info *fs_info,
                }
                bio->bi_bdev = page->dev->bdev;
                bio->bi_sector = page->physical >> 9;
-                bio->bi_end_io = scrub_complete_bio_end_io;
-                bio->bi_private = &complete;
                bio_add_page(bio, page->page, PAGE_SIZE, 0);
-                btrfsic_submit_bio(READ, bio);
+                if (btrfsic_submit_bio_wait(READ, bio))
-                /* this will also unplug the queue */
-                wait_for_completion(&complete);
-                page->io_error = !test_bit(BIO_UPTODATE, &bio->bi_flags);
-                if (!test_bit(BIO_UPTODATE, &bio->bi_flags))
                        sblock->no_io_error_seen = 0;
                bio_put(bio);
        }
@@ -1389,11 +1382,6 @@ static void scrub_recheck_block_checksum(struct btrfs_fs_info *fs_info,
                sblock->checksum_error = 1;
 }
-static void scrub_complete_bio_end_io(struct bio *bio, int err)
-{
-        complete((struct completion *)bio->bi_private);
-}
 static int scrub_repair_block_from_good_copy(struct scrub_block *sblock_bad,
                                             struct scrub_block *sblock_good,
                                             int force_write)
@@ -1428,7 +1416,6 @@ static int scrub_repair_page_from_good_copy(struct scrub_block *sblock_bad,
            sblock_bad->checksum_error || page_bad->io_error) {
                struct bio *bio;
                int ret;
-                DECLARE_COMPLETION_ONSTACK(complete);
                if (!page_bad->dev->bdev) {
                        printk_ratelimited(KERN_WARNING
@@ -1441,19 +1428,14 @@ static int scrub_repair_page_from_good_copy(struct scrub_block *sblock_bad,
                        return -EIO;
                bio->bi_bdev = page_bad->dev->bdev;
                bio->bi_sector = page_bad->physical >> 9;
-                bio->bi_end_io = scrub_complete_bio_end_io;
-                bio->bi_private = &complete;
                ret = bio_add_page(bio, page_good->page, PAGE_SIZE, 0);
                if (PAGE_SIZE != ret) {
                        bio_put(bio);
                        return -EIO;
                }
-                btrfsic_submit_bio(WRITE, bio);
-                /* this will also unplug the queue */
+                if (btrfsic_submit_bio_wait(WRITE, bio)) {
-                wait_for_completion(&complete);
-                if (!bio_flagged(bio, BIO_UPTODATE)) {
                        btrfs_dev_stat_inc_and_print(page_bad->dev,
                                BTRFS_DEV_STAT_WRITE_ERRS);
                        btrfs_dev_replace_stats_inc(
@@ -3373,7 +3355,6 @@ static int write_page_nocow(struct scrub_ctx *sctx,
        struct bio *bio;
        struct btrfs_device *dev;
        int ret;
-        DECLARE_COMPLETION_ONSTACK(compl);
        dev = sctx->wr_ctx.tgtdev;
        if (!dev)
@@ -3390,8 +3371,6 @@ static int write_page_nocow(struct scrub_ctx *sctx,
                spin_unlock(&sctx->stat_lock);
                return -ENOMEM;
        }
-        bio->bi_private = &compl;
-        bio->bi_end_io = scrub_complete_bio_end_io;
        bio->bi_size = 0;
        bio->bi_sector = physical_for_dev_replace >> 9;
        bio->bi_bdev = dev->bdev;
@@ -3402,10 +3381,8 @@ leave_with_eio:
                btrfs_dev_stat_inc_and_print(dev, BTRFS_DEV_STAT_WRITE_ERRS);
                return -EIO;
        }
-        btrfsic_submit_bio(WRITE_SYNC, bio);
-        wait_for_completion(&compl);
-        if (!test_bit(BIO_UPTODATE, &bio->bi_flags))
+        if (btrfsic_submit_bio_wait(WRITE_SYNC, bio))
                goto leave_with_eio;
        bio_put(bio);
diff --git a/fs/btrfs/send.c b/fs/btrfs/send.c
index 6837fe87f3a6..945d1db98f26 100644
--- a/fs/btrfs/send.c
+++ b/fs/btrfs/send.c
@@ -4723,8 +4723,8 @@ long btrfs_ioctl_send(struct file *mnt_file, void __user *arg_)
        }
        if (!access_ok(VERIFY_READ, arg->clone_sources,
-                        sizeof(*arg->clone_sources *
+                        sizeof(*arg->clone_sources) *
-                        arg->clone_sources_count))) {
+                        arg->clone_sources_count)) {
                ret = -EFAULT;
                goto out;
        }
diff --git a/fs/btrfs/super.c b/fs/btrfs/super.c
index 2d8ac1bf0cf9..d71a11d13dfa 100644
--- a/fs/btrfs/super.c
+++ b/fs/btrfs/super.c
@@ -432,7 +432,6 @@ int btrfs_parse_options(struct btrfs_root *root, char *options)
                        } else {
                                printk(KERN_INFO "btrfs: setting nodatacow\n");
                        }
-                        info->compress_type = BTRFS_COMPRESS_NONE;
                        btrfs_clear_opt(info->mount_opt, COMPRESS);
                        btrfs_clear_opt(info->mount_opt, FORCE_COMPRESS);
                        btrfs_set_opt(info->mount_opt, NODATACOW);
@@ -461,7 +460,6 @@ int btrfs_parse_options(struct btrfs_root *root, char *options)
                                btrfs_set_fs_incompat(info, COMPRESS_LZO);
                        } else if (strncmp(args[0].from, "no", 2) == 0) {
                                compress_type = "no";
-                                info->compress_type = BTRFS_COMPRESS_NONE;
                                btrfs_clear_opt(info->mount_opt, COMPRESS);
                                btrfs_clear_opt(info->mount_opt, FORCE_COMPRESS);
                                compress_force = false;
@@ -474,9 +472,10 @@ int btrfs_parse_options(struct btrfs_root *root, char *options)
                                btrfs_set_opt(info->mount_opt, FORCE_COMPRESS);
                                pr_info("btrfs: force %s compression\n",
                                        compress_type);
-                        } else
+                        } else if (btrfs_test_opt(root, COMPRESS)) {
                                pr_info("btrfs: use %s compression\n",
                                        compress_type);
+                        }
                        break;
                case Opt_ssd:
                        printk(KERN_INFO "btrfs: use ssd allocation scheme\n");
diff --git a/fs/btrfs/transaction.c b/fs/btrfs/transaction.c
index 57c16b46afbd..c6a872a8a468 100644
--- a/fs/btrfs/transaction.c
+++ b/fs/btrfs/transaction.c
@@ -1480,7 +1480,7 @@ static void do_async_commit(struct work_struct *work)
         * We've got freeze protection passed with the transaction.
         * Tell lockdep about it.
         */
-        if (ac->newtrans->type < TRANS_JOIN_NOLOCK)
+        if (ac->newtrans->type & __TRANS_FREEZABLE)
                rwsem_acquire_read(
                     &ac->root->fs_info->sb->s_writers.lock_map[SB_FREEZE_FS-1],
                     0, 1, _THIS_IP_);
@@ -1521,7 +1521,7 @@ int btrfs_commit_transaction_async(struct btrfs_trans_handle *trans,
         * Tell lockdep we've released the freeze rwsem, since the
         * async commit thread will be the one to unlock it.
         */
-        if (trans->type < TRANS_JOIN_NOLOCK)
+        if (ac->newtrans->type & __TRANS_FREEZABLE)
                rwsem_release(
                        &root->fs_info->sb->s_writers.lock_map[SB_FREEZE_FS-1],
                        1, _THIS_IP_);
diff --git a/fs/btrfs/tree-log.c b/fs/btrfs/tree-log.c
index 744553c83fe2..9f7fc51ca334 100644
--- a/fs/btrfs/tree-log.c
+++ b/fs/btrfs/tree-log.c
@@ -3697,7 +3697,8 @@ static int btrfs_log_inode(struct btrfs_trans_handle *trans,
                        ret = btrfs_truncate_inode_items(trans, log,
                                                         inode, 0, 0);
                } else if (test_and_clear_bit(BTRFS_INODE_COPY_EVERYTHING,
-                                              &BTRFS_I(inode)->runtime_flags)) {
+                                              &BTRFS_I(inode)->runtime_flags) ||
+                           inode_only == LOG_INODE_EXISTS) {
                        if (inode_only == LOG_INODE_ALL)
                                fast_search = true;
                        max_key.type = BTRFS_XATTR_ITEM_KEY;
@@ -3801,7 +3802,7 @@ log_extents:
                        err = ret;
                        goto out_unlock;
                }
-        } else {
+        } else if (inode_only == LOG_INODE_ALL) {
                struct extent_map_tree *tree = &BTRFS_I(inode)->extent_tree;
                struct extent_map *em, *n;
diff --git a/fs/btrfs/volumes.c b/fs/btrfs/volumes.c
index 0db637097862..92303f42baaa 100644
--- a/fs/btrfs/volumes.c
+++ b/fs/btrfs/volumes.c
@@ -5394,7 +5394,7 @@ static int bio_size_ok(struct block_device *bdev, struct bio *bio,
 {
        struct bio_vec *prev;
        struct request_queue *q = bdev_get_queue(bdev);
-        unsigned short max_sectors = queue_max_sectors(q);
+        unsigned int max_sectors = queue_max_sectors(q);
        struct bvec_merge_data bvm = {
                .bi_bdev = bdev,
                .bi_sector = sector,
diff --git a/fs/ceph/addr.c b/fs/ceph/addr.c
index 6df8bd481425..1e561c059539 100644
--- a/fs/ceph/addr.c
+++ b/fs/ceph/addr.c
@@ -216,7 +216,7 @@ static int readpage_nounlock(struct file *filp, struct page *page)
        }
        SetPageUptodate(page);
-        if (err == 0)
+        if (err >= 0)
                ceph_readpage_to_fscache(inode, page);
 out:
diff --git a/fs/ceph/cache.c b/fs/ceph/cache.c
index 7db2e6ca4b8f..8c44fdd4e1c3 100644
--- a/fs/ceph/cache.c
+++ b/fs/ceph/cache.c
@@ -324,6 +324,9 @@ void ceph_invalidate_fscache_page(struct inode* inode, struct page *page)
 {
        struct ceph_inode_info *ci = ceph_inode(inode);
+        if (!PageFsCache(page))
+                return;
        fscache_wait_on_page_write(ci->fscache, page);
        fscache_uncache_page(ci->fscache, page);
 }
diff --git a/fs/ceph/caps.c b/fs/ceph/caps.c
index 13976c33332e..3c0a4bd74996 100644
--- a/fs/ceph/caps.c
+++ b/fs/ceph/caps.c
@@ -897,7 +897,7 @@ static int __ceph_is_any_caps(struct ceph_inode_info *ci)
 * caller should hold i_ceph_lock.
 * caller will not hold session s_mutex if called from destroy_inode.
 */
-void __ceph_remove_cap(struct ceph_cap *cap)
+void __ceph_remove_cap(struct ceph_cap *cap, bool queue_release)
 {
        struct ceph_mds_session *session = cap->session;
        struct ceph_inode_info *ci = cap->ci;
@@ -909,6 +909,16 @@ void __ceph_remove_cap(struct ceph_cap *cap)
        /* remove from session list */
        spin_lock(&session->s_cap_lock);
+        /*
+         * s_cap_reconnect is protected by s_cap_lock. no one changes
+         * s_cap_gen while session is in the reconnect state.
+         */
+        if (queue_release &&
+            (!session->s_cap_reconnect ||
+             cap->cap_gen == session->s_cap_gen))
+                __queue_cap_release(session, ci->i_vino.ino, cap->cap_id,
+                                    cap->mseq, cap->issue_seq);
        if (session->s_cap_iterator == cap) {
                /* not yet, we are iterating over this very cap */
                dout("__ceph_remove_cap  delaying %p removal from session %p\n",
@@ -1023,7 +1033,6 @@ void __queue_cap_release(struct ceph_mds_session *session,
        struct ceph_mds_cap_release *head;
        struct ceph_mds_cap_item *item;
-        spin_lock(&session->s_cap_lock);
        BUG_ON(!session->s_num_cap_releases);
        msg = list_first_entry(&session->s_cap_releases,
                               struct ceph_msg, list_head);
@@ -1052,7 +1061,6 @@ void __queue_cap_release(struct ceph_mds_session *session,
                     (int)CEPH_CAPS_PER_RELEASE,
                     (int)msg->front.iov_len);
        }
-        spin_unlock(&session->s_cap_lock);
 }
 /*
@@ -1067,12 +1075,8 @@ void ceph_queue_caps_release(struct inode *inode)
        p = rb_first(&ci->i_caps);
        while (p) {
                struct ceph_cap *cap = rb_entry(p, struct ceph_cap, ci_node);
-                struct ceph_mds_session *session = cap->session;
-                __queue_cap_release(session, ceph_ino(inode), cap->cap_id,
-                                    cap->mseq, cap->issue_seq);
                p = rb_next(p);
-                __ceph_remove_cap(cap);
+                __ceph_remove_cap(cap, true);
        }
 }
@@ -2791,7 +2795,7 @@ static void handle_cap_export(struct inode *inode, struct ceph_mds_caps *ex,
                        }
                        spin_unlock(&mdsc->cap_dirty_lock);
                }
-                __ceph_remove_cap(cap);
+                __ceph_remove_cap(cap, false);
        }
        /* else, we already released it */
@@ -2931,9 +2935,12 @@ void ceph_handle_caps(struct ceph_mds_session *session,
        if (!inode) {
                dout(" i don't have ino %llx\n", vino.ino);
-                if (op == CEPH_CAP_OP_IMPORT)
+                if (op == CEPH_CAP_OP_IMPORT) {
+                        spin_lock(&session->s_cap_lock);
                        __queue_cap_release(session, vino.ino, cap_id,
                                            mseq, seq);
+                        spin_unlock(&session->s_cap_lock);
+                }
                goto flush_cap_releases;
        }
diff --git a/fs/ceph/dir.c b/fs/ceph/dir.c
index 868b61d56cac..2a0bcaeb189a 100644
--- a/fs/ceph/dir.c
+++ b/fs/ceph/dir.c
@@ -352,8 +352,18 @@ more:
                }
                /* note next offset and last dentry name */
+                rinfo = &req->r_reply_info;
+                if (le32_to_cpu(rinfo->dir_dir->frag) != frag) {
+                        frag = le32_to_cpu(rinfo->dir_dir->frag);
+                        if (ceph_frag_is_leftmost(frag))
+                                fi->next_offset = 2;
+                        else
+                                fi->next_offset = 0;
+                        off = fi->next_offset;
+                }
                fi->offset = fi->next_offset;
                fi->last_readdir = req;
+                fi->frag = frag;
                if (req->r_reply_info.dir_end) {
                        kfree(fi->last_name);
@@ -363,7 +373,6 @@ more:
                        else
                                fi->next_offset = 0;
                } else {
-                        rinfo = &req->r_reply_info;
                        err = note_last_dentry(fi,
                                       rinfo->dir_dname[rinfo->dir_nr-1],
                                       rinfo->dir_dname_len[rinfo->dir_nr-1]);
diff --git a/fs/ceph/inode.c b/fs/ceph/inode.c
index 8549a48115f7..9a8e396aed89 100644
--- a/fs/ceph/inode.c
+++ b/fs/ceph/inode.c
@@ -577,6 +577,8 @@ static int fill_inode(struct inode *inode,
        int issued = 0, implemented;
        struct timespec mtime, atime, ctime;
        u32 nsplits;
+        struct ceph_inode_frag *frag;
+        struct rb_node *rb_node;
        struct ceph_buffer *xattr_blob = NULL;
        int err = 0;
        int queue_trunc = 0;
@@ -751,15 +753,38 @@ no_change:
        /* FIXME: move me up, if/when version reflects fragtree changes */
        nsplits = le32_to_cpu(info->fragtree.nsplits);
        mutex_lock(&ci->i_fragtree_mutex);
+        rb_node = rb_first(&ci->i_fragtree);
        for (i = 0; i < nsplits; i++) {
                u32 id = le32_to_cpu(info->fragtree.splits[i].frag);
-                struct ceph_inode_frag *frag = __get_or_create_frag(ci, id);
+                frag = NULL;
+                while (rb_node) {
-                if (IS_ERR(frag))
+                        frag = rb_entry(rb_node, struct ceph_inode_frag, node);
-                        continue;
+                        if (ceph_frag_compare(frag->frag, id) >= 0) {
+                                if (frag->frag != id)
+                                        frag = NULL;
+                                else
+                                        rb_node = rb_next(rb_node);
+                                break;
+                        }
+                        rb_node = rb_next(rb_node);
+                        rb_erase(&frag->node, &ci->i_fragtree);
+                        kfree(frag);
+                        frag = NULL;
+                }
+                if (!frag) {
+                        frag = __get_or_create_frag(ci, id);
+                        if (IS_ERR(frag))
+                                continue;
+                }
                frag->split_by = le32_to_cpu(info->fragtree.splits[i].by);
                dout(" frag %x split by %d\n", frag->frag, frag->split_by);
        }
+        while (rb_node) {
+                frag = rb_entry(rb_node, struct ceph_inode_frag, node);
+                rb_node = rb_next(rb_node);
+                rb_erase(&frag->node, &ci->i_fragtree);
+                kfree(frag);
+        }
        mutex_unlock(&ci->i_fragtree_mutex);
        /* were we issued a capability? */
@@ -1250,8 +1275,20 @@ int ceph_readdir_prepopulate(struct ceph_mds_request *req,
        int err = 0, i;
        struct inode *snapdir = NULL;
        struct ceph_mds_request_head *rhead = req->r_request->front.iov_base;
-        u64 frag = le32_to_cpu(rhead->args.readdir.frag);
        struct ceph_dentry_info *di;
+        u64 r_readdir_offset = req->r_readdir_offset;
+        u32 frag = le32_to_cpu(rhead->args.readdir.frag);
+        if (rinfo->dir_dir &&
+            le32_to_cpu(rinfo->dir_dir->frag) != frag) {
+                dout("readdir_prepopulate got new frag %x -> %x\n",
+                     frag, le32_to_cpu(rinfo->dir_dir->frag));
+                frag = le32_to_cpu(rinfo->dir_dir->frag);
+                if (ceph_frag_is_leftmost(frag))
+                        r_readdir_offset = 2;
+                else
+                        r_readdir_offset = 0;
+        }
        if (req->r_aborted)
                return readdir_prepopulate_inodes_only(req, session);
@@ -1315,7 +1352,7 @@ retry_lookup:
                }
                di = dn->d_fsdata;
-                di->offset = ceph_make_fpos(frag, i + req->r_readdir_offset);
+                di->offset = ceph_make_fpos(frag, i + r_readdir_offset);
                /* inode */
                if (dn->d_inode) {
diff --git a/fs/ceph/mds_client.c b/fs/ceph/mds_client.c
index b7bda5d9611d..d90861f45210 100644
--- a/fs/ceph/mds_client.c
+++ b/fs/ceph/mds_client.c
@@ -43,6 +43,7 @@
 */
 struct ceph_reconnect_state {
+        int nr_caps;
        struct ceph_pagelist *pagelist;
        bool flock;
 };
@@ -443,6 +444,7 @@ static struct ceph_mds_session *register_session(struct ceph_mds_client *mdsc,
        INIT_LIST_HEAD(&s->s_waiting);
        INIT_LIST_HEAD(&s->s_unsafe);
        s->s_num_cap_releases = 0;
+        s->s_cap_reconnect = 0;
        s->s_cap_iterator = NULL;
        INIT_LIST_HEAD(&s->s_cap_releases);
        INIT_LIST_HEAD(&s->s_cap_releases_done);
@@ -642,6 +644,8 @@ static void __unregister_request(struct ceph_mds_client *mdsc,
                req->r_unsafe_dir = NULL;
        }
+        complete_all(&req->r_safe_completion);
        ceph_mdsc_put_request(req);
 }
@@ -986,7 +990,7 @@ static int remove_session_caps_cb(struct inode *inode, struct ceph_cap *cap,
        dout("removing cap %p, ci is %p, inode is %p\n",
             cap, ci, &ci->vfs_inode);
        spin_lock(&ci->i_ceph_lock);
-        __ceph_remove_cap(cap);
+        __ceph_remove_cap(cap, false);
        if (!__ceph_is_any_real_caps(ci)) {
                struct ceph_mds_client *mdsc =
                        ceph_sb_to_client(inode->i_sb)->mdsc;
@@ -1231,9 +1235,7 @@ static int trim_caps_cb(struct inode *inode, struct ceph_cap *cap, void *arg)
        session->s_trim_caps--;
        if (oissued) {
                /* we aren't the only cap.. just remove us */
-                __queue_cap_release(session, ceph_ino(inode), cap->cap_id,
+                __ceph_remove_cap(cap, true);
-                                    cap->mseq, cap->issue_seq);
-                __ceph_remove_cap(cap);
        } else {
                /* try to drop referring dentries */
                spin_unlock(&ci->i_ceph_lock);
@@ -1416,7 +1418,6 @@ static void discard_cap_releases(struct ceph_mds_client *mdsc,
        unsigned num;
        dout("discard_cap_releases mds%d\n", session->s_mds);
-        spin_lock(&session->s_cap_lock);
        /* zero out the in-progress message */
        msg = list_first_entry(&session->s_cap_releases,
@@ -1443,8 +1444,6 @@ static void discard_cap_releases(struct ceph_mds_client *mdsc,
                msg->front.iov_len = sizeof(*head);
                list_add(&msg->list_head, &session->s_cap_releases);
        }
-        spin_unlock(&session->s_cap_lock);
 }
 /*
@@ -1875,8 +1874,11 @@ static int __do_request(struct ceph_mds_client *mdsc,
        int mds = -1;
        int err = -EAGAIN;
-        if (req->r_err || req->r_got_result)
+        if (req->r_err || req->r_got_result) {
+                if (req->r_aborted)
+                        __unregister_request(mdsc, req);
                goto out;
+        }
        if (req->r_timeout &&
            time_after_eq(jiffies, req->r_started + req->r_timeout)) {
@@ -2186,7 +2188,6 @@ static void handle_reply(struct ceph_mds_session *session, struct ceph_msg *msg)
        if (head->safe) {
                req->r_got_safe = true;
                __unregister_request(mdsc, req);
-                complete_all(&req->r_safe_completion);
                if (req->r_got_unsafe) {
                        /*
@@ -2238,8 +2239,7 @@ static void handle_reply(struct ceph_mds_session *session, struct ceph_msg *msg)
        err = ceph_fill_trace(mdsc->fsc->sb, req, req->r_session);
        if (err == 0) {
                if (result == 0 && (req->r_op == CEPH_MDS_OP_READDIR ||
-                                    req->r_op == CEPH_MDS_OP_LSSNAP) &&
+                                    req->r_op == CEPH_MDS_OP_LSSNAP))
-                    rinfo->dir_nr)
                        ceph_readdir_prepopulate(req, req->r_session);
                ceph_unreserve_caps(mdsc, &req->r_caps_reservation);
        }
@@ -2490,6 +2490,7 @@ static int encode_caps_cb(struct inode *inode, struct ceph_cap *cap,
        cap->seq = 0;        /* reset cap seq */
        cap->issue_seq = 0;  /* and issue_seq */
        cap->mseq = 0;       /* and migrate_seq */
+        cap->cap_gen = cap->session->s_cap_gen;
        if (recon_state->flock) {
                rec.v2.cap_id = cpu_to_le64(cap->cap_id);
@@ -2552,6 +2553,8 @@ encode_again:
        } else {
                err = ceph_pagelist_append(pagelist, &rec, reclen);
        }
+        recon_state->nr_caps++;
 out_free:
        kfree(path);
 out_dput:
@@ -2579,6 +2582,7 @@ static void send_mds_reconnect(struct ceph_mds_client *mdsc,
        struct rb_node *p;
        int mds = session->s_mds;
        int err = -ENOMEM;
+        int s_nr_caps;
        struct ceph_pagelist *pagelist;
        struct ceph_reconnect_state recon_state;
@@ -2610,20 +2614,38 @@ static void send_mds_reconnect(struct ceph_mds_client *mdsc,
        dout("session %p state %s\n", session,
             session_state_name(session->s_state));
+        spin_lock(&session->s_gen_ttl_lock);
+        session->s_cap_gen++;
+        spin_unlock(&session->s_gen_ttl_lock);
+        spin_lock(&session->s_cap_lock);
+        /*
+         * notify __ceph_remove_cap() that we are composing cap reconnect.
+         * If a cap get released before being added to the cap reconnect,
+         * __ceph_remove_cap() should skip queuing cap release.
+         */
+        session->s_cap_reconnect = 1;
        /* drop old cap expires; we're about to reestablish that state */
        discard_cap_releases(mdsc, session);
+        spin_unlock(&session->s_cap_lock);
        /* traverse this session's caps */
-        err = ceph_pagelist_encode_32(pagelist, session->s_nr_caps);
+        s_nr_caps = session->s_nr_caps;
+        err = ceph_pagelist_encode_32(pagelist, s_nr_caps);
        if (err)
                goto fail;
+        recon_state.nr_caps = 0;
        recon_state.pagelist = pagelist;
        recon_state.flock = session->s_con.peer_features & CEPH_FEATURE_FLOCK;
        err = iterate_session_caps(session, encode_caps_cb, &recon_state);
        if (err < 0)
                goto fail;
+        spin_lock(&session->s_cap_lock);
+        session->s_cap_reconnect = 0;
+        spin_unlock(&session->s_cap_lock);
        /*
         * snaprealms.  we provide mds with the ino, seq (version), and
         * parent for all of our realms.  If the mds has any newer info,
@@ -2646,11 +2668,18 @@ static void send_mds_reconnect(struct ceph_mds_client *mdsc,
        if (recon_state.flock)
                reply->hdr.version = cpu_to_le16(2);
-        if (pagelist->length) {
-                /* set up outbound data if we have any */
+        /* raced with cap release? */
-                reply->hdr.data_len = cpu_to_le32(pagelist->length);
+        if (s_nr_caps != recon_state.nr_caps) {
-                ceph_msg_data_add_pagelist(reply, pagelist);
+                struct page *page = list_first_entry(&pagelist->head,
+                                                     struct page, lru);
+                __le32 *addr = kmap_atomic(page);
+                *addr = cpu_to_le32(recon_state.nr_caps);
+                kunmap_atomic(addr);
        }
+        reply->hdr.data_len = cpu_to_le32(pagelist->length);
+        ceph_msg_data_add_pagelist(reply, pagelist);
        ceph_con_send(&session->s_con, reply);
        mutex_unlock(&session->s_mutex);
diff --git a/fs/ceph/mds_client.h b/fs/ceph/mds_client.h
index c2a19fbbe517..4c053d099ae4 100644
--- a/fs/ceph/mds_client.h
+++ b/fs/ceph/mds_client.h
@@ -132,6 +132,7 @@ struct ceph_mds_session {
        struct list_head  s_caps;     /* all caps issued by this session */
        int               s_nr_caps, s_trim_caps;
        int               s_num_cap_releases;
+        int               s_cap_reconnect;
        struct list_head  s_cap_releases; /* waiting cap_release messages */
        struct list_head  s_cap_releases_done; /* ready to send */
        struct ceph_cap  *s_cap_iterator;
diff --git a/fs/ceph/super.h b/fs/ceph/super.h
index 6014b0a3c405..ef4ac38bb614 100644
--- a/fs/ceph/super.h
+++ b/fs/ceph/super.h
@@ -741,13 +741,7 @@ extern int ceph_add_cap(struct inode *inode,
                        int fmode, unsigned issued, unsigned wanted,
                        unsigned cap, unsigned seq, u64 realmino, int flags,
                        struct ceph_cap_reservation *caps_reservation);
-extern void __ceph_remove_cap(struct ceph_cap *cap);
+extern void __ceph_remove_cap(struct ceph_cap *cap, bool queue_release);
-static inline void ceph_remove_cap(struct ceph_cap *cap)
-{
-        spin_lock(&cap->ci->i_ceph_lock);
-        __ceph_remove_cap(cap);
-        spin_unlock(&cap->ci->i_ceph_lock);
-}
 extern void ceph_put_cap(struct ceph_mds_client *mdsc,
                         struct ceph_cap *cap);
diff --git a/fs/cifs/cifsglob.h b/fs/cifs/cifsglob.h
index d9ea7ada1378..f918a998a087 100644
--- a/fs/cifs/cifsglob.h
+++ b/fs/cifs/cifsglob.h
@@ -384,6 +384,7 @@ struct smb_version_operations {
        int (*clone_range)(const unsigned int, struct cifsFileInfo *src_file,
                        struct cifsFileInfo *target_file, u64 src_off, u64 len,
                        u64 dest_off);
+        int (*validate_negotiate)(const unsigned int, struct cifs_tcon *);
 };
 struct smb_version_values {
diff --git a/fs/cifs/ioctl.c b/fs/cifs/ioctl.c
index 409b45eefe70..77492301cc2b 100644
--- a/fs/cifs/ioctl.c
+++ b/fs/cifs/ioctl.c
@@ -26,13 +26,15 @@
 #include <linux/mount.h>
 #include <linux/mm.h>
 #include <linux/pagemap.h>
-#include <linux/btrfs.h>
 #include "cifspdu.h"
 #include "cifsglob.h"
 #include "cifsproto.h"
 #include "cifs_debug.h"
 #include "cifsfs.h"
+#define CIFS_IOCTL_MAGIC        0xCF
+#define CIFS_IOC_COPYCHUNK_FILE _IOW(CIFS_IOCTL_MAGIC, 3, int)
 static long cifs_ioctl_clone(unsigned int xid, struct file *dst_file,
                        unsigned long srcfd, u64 off, u64 len, u64 destoff)
 {
@@ -213,7 +215,7 @@ long cifs_ioctl(struct file *filep, unsigned int command, unsigned long arg)
                                cifs_dbg(FYI, "set compress flag rc %d\n", rc);
                        }
                        break;
-                case BTRFS_IOC_CLONE:
+                case CIFS_IOC_COPYCHUNK_FILE:
                        rc = cifs_ioctl_clone(xid, filep, arg, 0, 0, 0);
                        break;
                default:
diff --git a/fs/cifs/smb2ops.c b/fs/cifs/smb2ops.c
index 11dde4b24f8a..757da3e54d3d 100644
--- a/fs/cifs/smb2ops.c
+++ b/fs/cifs/smb2ops.c
@@ -532,7 +532,10 @@ smb2_clone_range(const unsigned int xid,
        int rc;
        unsigned int ret_data_len;
        struct copychunk_ioctl *pcchunk;
-        char *retbuf = NULL;
+        struct copychunk_ioctl_rsp *retbuf = NULL;
+        struct cifs_tcon *tcon;
+        int chunks_copied = 0;
+        bool chunk_sizes_updated = false;
        pcchunk = kmalloc(sizeof(struct copychunk_ioctl), GFP_KERNEL);
@@ -547,27 +550,96 @@ smb2_clone_range(const unsigned int xid,
        /* Note: request_res_key sets res_key null only if rc !=0 */
        if (rc)
-                return rc;
+                goto cchunk_out;
        /* For now array only one chunk long, will make more flexible later */
        pcchunk->ChunkCount = __constant_cpu_to_le32(1);
        pcchunk->Reserved = 0;
-        pcchunk->SourceOffset = cpu_to_le64(src_off);
-        pcchunk->TargetOffset = cpu_to_le64(dest_off);
-        pcchunk->Length = cpu_to_le32(len);
        pcchunk->Reserved2 = 0;
-        /* Request that server copy to target from src file identified by key */
+        tcon = tlink_tcon(trgtfile->tlink);
-        rc = SMB2_ioctl(xid, tlink_tcon(trgtfile->tlink),
-                        trgtfile->fid.persistent_fid,
-                        trgtfile->fid.volatile_fid, FSCTL_SRV_COPYCHUNK_WRITE,
-                        true /* is_fsctl */, (char *)pcchunk,
-                        sizeof(struct copychunk_ioctl), &retbuf, &ret_data_len);
-        /* BB need to special case rc = EINVAL to alter chunk size */
+        while (len > 0) {
+                pcchunk->SourceOffset = cpu_to_le64(src_off);
+                pcchunk->TargetOffset = cpu_to_le64(dest_off);
+                pcchunk->Length =
+                        cpu_to_le32(min_t(u32, len, tcon->max_bytes_chunk));
-        cifs_dbg(FYI, "rc %d data length out %d\n", rc, ret_data_len);
+                /* Request server copy to target from src identified by key */
+                rc = SMB2_ioctl(xid, tcon, trgtfile->fid.persistent_fid,
+                        trgtfile->fid.volatile_fid, FSCTL_SRV_COPYCHUNK_WRITE,
+                        true /* is_fsctl */, (char *)pcchunk,
+                        sizeof(struct copychunk_ioctl), (char **)&retbuf,
+                        &ret_data_len);
+                if (rc == 0) {
+                        if (ret_data_len !=
+                                        sizeof(struct copychunk_ioctl_rsp)) {
+                                cifs_dbg(VFS, "invalid cchunk response size\n");
+                                rc = -EIO;
+                                goto cchunk_out;
+                        }
+                        if (retbuf->TotalBytesWritten == 0) {
+                                cifs_dbg(FYI, "no bytes copied\n");
+                                rc = -EIO;
+                                goto cchunk_out;
+                        }
+                        /*
+                         * Check if server claimed to write more than we asked
+                         */
+                        if (le32_to_cpu(retbuf->TotalBytesWritten) >
+                            le32_to_cpu(pcchunk->Length)) {
+                                cifs_dbg(VFS, "invalid copy chunk response\n");
+                                rc = -EIO;
+                                goto cchunk_out;
+                        }
+                        if (le32_to_cpu(retbuf->ChunksWritten) != 1) {
+                                cifs_dbg(VFS, "invalid num chunks written\n");
+                                rc = -EIO;
+                                goto cchunk_out;
+                        }
+                        chunks_copied++;
+                        src_off += le32_to_cpu(retbuf->TotalBytesWritten);
+                        dest_off += le32_to_cpu(retbuf->TotalBytesWritten);
+                        len -= le32_to_cpu(retbuf->TotalBytesWritten);
+                        cifs_dbg(FYI, "Chunks %d PartialChunk %d Total %d\n",
+                                le32_to_cpu(retbuf->ChunksWritten),
+                                le32_to_cpu(retbuf->ChunkBytesWritten),
+                                le32_to_cpu(retbuf->TotalBytesWritten));
+                } else if (rc == -EINVAL) {
+                        if (ret_data_len != sizeof(struct copychunk_ioctl_rsp))
+                                goto cchunk_out;
+                        cifs_dbg(FYI, "MaxChunks %d BytesChunk %d MaxCopy %d\n",
+                                le32_to_cpu(retbuf->ChunksWritten),
+                                le32_to_cpu(retbuf->ChunkBytesWritten),
+                                le32_to_cpu(retbuf->TotalBytesWritten));
+                        /*
+                         * Check if this is the first request using these sizes,
+                         * (ie check if copy succeed once with original sizes
+                         * and check if the server gave us different sizes after
+                         * we already updated max sizes on previous request).
+                         * if not then why is the server returning an error now
+                         */
+                        if ((chunks_copied != 0) || chunk_sizes_updated)
+                                goto cchunk_out;
+                        /* Check that server is not asking us to grow size */
+                        if (le32_to_cpu(retbuf->ChunkBytesWritten) <
+                                        tcon->max_bytes_chunk)
+                                tcon->max_bytes_chunk =
+                                        le32_to_cpu(retbuf->ChunkBytesWritten);
+                        else
+                                goto cchunk_out; /* server gave us bogus size */
+                        /* No need to change MaxChunks since already set to 1 */
+                        chunk_sizes_updated = true;
+                }
+        }
+cchunk_out:
        kfree(pcchunk);
        return rc;
 }
@@ -1247,6 +1319,7 @@ struct smb_version_operations smb30_operations = {
        .create_lease_buf = smb3_create_lease_buf,
        .parse_lease_buf = smb3_parse_lease_buf,
        .clone_range = smb2_clone_range,
+        .validate_negotiate = smb3_validate_negotiate,
 };
 struct smb_version_values smb20_values = {
diff --git a/fs/cifs/smb2pdu.c b/fs/cifs/smb2pdu.c
index d65270c290a1..2013234b73ad 100644
--- a/fs/cifs/smb2pdu.c
+++ b/fs/cifs/smb2pdu.c
@@ -454,6 +454,81 @@ neg_exit:
        return rc;
 }
+int smb3_validate_negotiate(const unsigned int xid, struct cifs_tcon *tcon)
+{
+        int rc = 0;
+        struct validate_negotiate_info_req vneg_inbuf;
+        struct validate_negotiate_info_rsp *pneg_rsp;
+        u32 rsplen;
+        cifs_dbg(FYI, "validate negotiate\n");
+        /*
+         * validation ioctl must be signed, so no point sending this if we
+         * can not sign it.  We could eventually change this to selectively
+         * sign just this, the first and only signed request on a connection.
+         * This is good enough for now since a user who wants better security
+         * would also enable signing on the mount. Having validation of
+         * negotiate info for signed connections helps reduce attack vectors
+         */
+        if (tcon->ses->server->sign == false)
+                return 0; /* validation requires signing */
+        vneg_inbuf.Capabilities =
+                        cpu_to_le32(tcon->ses->server->vals->req_capabilities);
+        memcpy(vneg_inbuf.Guid, cifs_client_guid, SMB2_CLIENT_GUID_SIZE);
+        if (tcon->ses->sign)
+                vneg_inbuf.SecurityMode =
+                        cpu_to_le16(SMB2_NEGOTIATE_SIGNING_REQUIRED);
+        else if (global_secflags & CIFSSEC_MAY_SIGN)
+                vneg_inbuf.SecurityMode =
+                        cpu_to_le16(SMB2_NEGOTIATE_SIGNING_ENABLED);
+        else
+                vneg_inbuf.SecurityMode = 0;
+        vneg_inbuf.DialectCount = cpu_to_le16(1);
+        vneg_inbuf.Dialects[0] =
+                cpu_to_le16(tcon->ses->server->vals->protocol_id);
+        rc = SMB2_ioctl(xid, tcon, NO_FILE_ID, NO_FILE_ID,
+                FSCTL_VALIDATE_NEGOTIATE_INFO, true /* is_fsctl */,
+                (char *)&vneg_inbuf, sizeof(struct validate_negotiate_info_req),
+                (char **)&pneg_rsp, &rsplen);
+        if (rc != 0) {
+                cifs_dbg(VFS, "validate protocol negotiate failed: %d\n", rc);
+                return -EIO;
+        }
+        if (rsplen != sizeof(struct validate_negotiate_info_rsp)) {
+                cifs_dbg(VFS, "invalid size of protocol negotiate response\n");
+                return -EIO;
+        }
+        /* check validate negotiate info response matches what we got earlier */
+        if (pneg_rsp->Dialect !=
+                        cpu_to_le16(tcon->ses->server->vals->protocol_id))
+                goto vneg_out;
+        if (pneg_rsp->SecurityMode != cpu_to_le16(tcon->ses->server->sec_mode))
+                goto vneg_out;
+        /* do not validate server guid because not saved at negprot time yet */
+        if ((le32_to_cpu(pneg_rsp->Capabilities) | SMB2_NT_FIND |
+              SMB2_LARGE_FILES) != tcon->ses->server->capabilities)
+                goto vneg_out;
+        /* validate negotiate successful */
+        cifs_dbg(FYI, "validate negotiate info successful\n");
+        return 0;
+vneg_out:
+        cifs_dbg(VFS, "protocol revalidation - security settings mismatch\n");
+        return -EIO;
+}
 int
 SMB2_sess_setup(const unsigned int xid, struct cifs_ses *ses,
                const struct nls_table *nls_cp)
@@ -829,6 +904,8 @@ SMB2_tcon(const unsigned int xid, struct cifs_ses *ses, const char *tree,
            ((tcon->share_flags & SHI1005_FLAGS_DFS) == 0))
                cifs_dbg(VFS, "DFS capability contradicts DFS flag\n");
        init_copy_chunk_defaults(tcon);
+        if (tcon->ses->server->ops->validate_negotiate)
+                rc = tcon->ses->server->ops->validate_negotiate(xid, tcon);
 tcon_exit:
        free_rsp_buf(resp_buftype, rsp);
        kfree(unc_path);
@@ -1214,10 +1291,17 @@ SMB2_ioctl(const unsigned int xid, struct cifs_tcon *tcon, u64 persistent_fid,
        rc = SendReceive2(xid, ses, iov, num_iovecs, &resp_buftype, 0);
        rsp = (struct smb2_ioctl_rsp *)iov[0].iov_base;
-        if (rc != 0) {
+        if ((rc != 0) && (rc != -EINVAL)) {
                if (tcon)
                        cifs_stats_fail_inc(tcon, SMB2_IOCTL_HE);
                goto ioctl_exit;
+        } else if (rc == -EINVAL) {
+                if ((opcode != FSCTL_SRV_COPYCHUNK_WRITE) &&
+                    (opcode != FSCTL_SRV_COPYCHUNK)) {
+                        if (tcon)
+                                cifs_stats_fail_inc(tcon, SMB2_IOCTL_HE);
+                        goto ioctl_exit;
+                }
        }
        /* check if caller wants to look at return data or just return rc */
@@ -2154,11 +2238,9 @@ send_set_info(const unsigned int xid, struct cifs_tcon *tcon,
        rc = SendReceive2(xid, ses, iov, num, &resp_buftype, 0);
        rsp = (struct smb2_set_info_rsp *)iov[0].iov_base;
-        if (rc != 0) {
+        if (rc != 0)
                cifs_stats_fail_inc(tcon, SMB2_SET_INFO_HE);
-                goto out;
-        }
-out:
        free_rsp_buf(resp_buftype, rsp);
        kfree(iov);
        return rc;
diff --git a/fs/cifs/smb2pdu.h b/fs/cifs/smb2pdu.h
index f88320bbb477..2022c542ea3a 100644
--- a/fs/cifs/smb2pdu.h
+++ b/fs/cifs/smb2pdu.h
@@ -577,13 +577,19 @@ struct copychunk_ioctl_rsp {
        __le32 TotalBytesWritten;
 } __packed;
-/* Response and Request are the same format */
+struct validate_negotiate_info_req {
-struct validate_negotiate_info {
        __le32 Capabilities;
        __u8   Guid[SMB2_CLIENT_GUID_SIZE];
        __le16 SecurityMode;
        __le16 DialectCount;
-        __le16 Dialect[1];
+        __le16 Dialects[1]; /* dialect (someday maybe list) client asked for */
+} __packed;
+struct validate_negotiate_info_rsp {
+        __le32 Capabilities;
+        __u8   Guid[SMB2_CLIENT_GUID_SIZE];
+        __le16 SecurityMode;
+        __le16 Dialect; /* Dialect in use for the connection */
 } __packed;
 #define RSS_CAPABLE     0x00000001
diff --git a/fs/cifs/smb2proto.h b/fs/cifs/smb2proto.h
index b4eea105b08c..93adc64666f3 100644
--- a/fs/cifs/smb2proto.h
+++ b/fs/cifs/smb2proto.h
@@ -162,5 +162,6 @@ extern int smb2_lockv(const unsigned int xid, struct cifs_tcon *tcon,
                      struct smb2_lock_element *buf);
 extern int SMB2_lease_break(const unsigned int xid, struct cifs_tcon *tcon,
                            __u8 *lease_key, const __le32 lease_state);
+extern int smb3_validate_negotiate(const unsigned int, struct cifs_tcon *);
 #endif                  /* _SMB2PROTO_H */
diff --git a/fs/cifs/smbfsctl.h b/fs/cifs/smbfsctl.h
index a4b2391fe66e..0e538b5c9622 100644
--- a/fs/cifs/smbfsctl.h
+++ b/fs/cifs/smbfsctl.h
@@ -90,7 +90,7 @@
 #define FSCTL_LMR_REQUEST_RESILIENCY 0x001401D4 /* BB add struct */
 #define FSCTL_LMR_GET_LINK_TRACK_INF 0x001400E8 /* BB add struct */
 #define FSCTL_LMR_SET_LINK_TRACK_INF 0x001400EC /* BB add struct */
-#define FSCTL_VALIDATE_NEGOTIATE_INFO 0x00140204 /* BB add struct */
+#define FSCTL_VALIDATE_NEGOTIATE_INFO 0x00140204
 /* Perform server-side data movement */
 #define FSCTL_SRV_COPYCHUNK 0x001440F2
 #define FSCTL_SRV_COPYCHUNK_WRITE 0x001480F2
diff --git a/fs/configfs/dir.c b/fs/configfs/dir.c
index 277bd1be21fd..e081acbac2e7 100644
--- a/fs/configfs/dir.c
+++ b/fs/configfs/dir.c
@@ -56,29 +56,28 @@ static void configfs_d_iput(struct dentry * dentry,
        struct configfs_dirent *sd = dentry->d_fsdata;
        if (sd) {
-                BUG_ON(sd->s_dentry != dentry);
                /* Coordinate with configfs_readdir */
                spin_lock(&configfs_dirent_lock);
-                sd->s_dentry = NULL;
+                /* Coordinate with configfs_attach_attr where will increase
+                 * sd->s_count and update sd->s_dentry to new allocated one.
+                 * Only set sd->dentry to null when this dentry is the only
+                 * sd owner.
+                 * If not do so, configfs_d_iput may run just after
+                 * configfs_attach_attr and set sd->s_dentry to null
+                 * even it's still in use.
+                 */
+                if (atomic_read(&sd->s_count) <= 2)
+                        sd->s_dentry = NULL;
                spin_unlock(&configfs_dirent_lock);
                configfs_put(sd);
        }
        iput(inode);
 }
-/*
- * We _must_ delete our dentries on last dput, as the chain-to-parent
- * behavior is required to clear the parents of default_groups.
- */
-static int configfs_d_delete(const struct dentry *dentry)
-{
-        return 1;
-}
 const struct dentry_operations configfs_dentry_ops = {
        .d_iput         = configfs_d_iput,
-        /* simple_delete_dentry() isn't exported */
+        .d_delete       = always_delete_dentry,
-        .d_delete       = configfs_d_delete,
 };
 #ifdef CONFIG_LOCKDEP
@@ -426,8 +425,11 @@ static int configfs_attach_attr(struct configfs_dirent * sd, struct dentry * den
        struct configfs_attribute * attr = sd->s_element;
        int error;
+        spin_lock(&configfs_dirent_lock);
        dentry->d_fsdata = configfs_get(sd);
        sd->s_dentry = dentry;
+        spin_unlock(&configfs_dirent_lock);
        error = configfs_create(dentry, (attr->ca_mode & S_IALLUGO) | S_IFREG,
                                configfs_init_file);
        if (error) {
diff --git a/fs/coredump.c b/fs/coredump.c
index 62406b6959b6..bc3fbcd32558 100644
--- a/fs/coredump.c
+++ b/fs/coredump.c
@@ -695,7 +695,7 @@ int dump_emit(struct coredump_params *cprm, const void *addr, int nr)
        while (nr) {
                if (dump_interrupted())
                        return 0;
-                n = vfs_write(file, addr, nr, &pos);
+                n = __kernel_write(file, addr, nr, &pos);
                if (n <= 0)
                        return 0;
                file->f_pos = pos;
@@ -733,7 +733,7 @@ int dump_align(struct coredump_params *cprm, int align)
 {
        unsigned mod = cprm->written & (align - 1);
        if (align & (align - 1))
-                return -EINVAL;
+                return 0;
-        return mod ? dump_skip(cprm, align - mod) : 0;
+        return mod ? dump_skip(cprm, align - mod) : 1;
 }
 EXPORT_SYMBOL(dump_align);
diff --git a/fs/dcache.c b/fs/dcache.c
index 0a38ef8d7f00..6055d61811d3 100644
--- a/fs/dcache.c
+++ b/fs/dcache.c
@@ -88,35 +88,6 @@ EXPORT_SYMBOL(rename_lock);
 static struct kmem_cache *dentry_cache __read_mostly;
-/**
- * read_seqbegin_or_lock - begin a sequence number check or locking block
- * @lock: sequence lock
- * @seq : sequence number to be checked
- *
- * First try it once optimistically without taking the lock. If that fails,
- * take the lock. The sequence number is also used as a marker for deciding
- * whether to be a reader (even) or writer (odd).
- * N.B. seq must be initialized to an even number to begin with.
- */
-static inline void read_seqbegin_or_lock(seqlock_t *lock, int *seq)
-{
-        if (!(*seq & 1))        /* Even */
-                *seq = read_seqbegin(lock);
-        else                    /* Odd */
-                read_seqlock_excl(lock);
-}
-static inline int need_seqretry(seqlock_t *lock, int seq)
-{
-        return !(seq & 1) && read_seqretry(lock, seq);
-}
-static inline void done_seqretry(seqlock_t *lock, int seq)
-{
-        if (seq & 1)
-                read_sequnlock_excl(lock);
-}
 /*
 * This is the single most critical data structure when it comes
 * to the dcache: the hashtable for lookups. Somebody should try
@@ -125,8 +96,6 @@ static inline void done_seqretry(seqlock_t *lock, int seq)
 * This hash-function tries to avoid losing too many bits of hash
 * information, yet avoid using a prime hash-size or similar.
 */
-#define D_HASHBITS     d_hash_shift
-#define D_HASHMASK     d_hash_mask
 static unsigned int d_hash_mask __read_mostly;
 static unsigned int d_hash_shift __read_mostly;
@@ -137,8 +106,8 @@ static inline struct hlist_bl_head *d_hash(const struct dentry *parent,
                                        unsigned int hash)
 {
        hash += (unsigned long) parent / L1_CACHE_BYTES;
-        hash = hash + (hash >> D_HASHBITS);
+        hash = hash + (hash >> d_hash_shift);
-        return dentry_hashtable + (hash & D_HASHMASK);
+        return dentry_hashtable + (hash & d_hash_mask);
 }
 /* Statistics gathering. */
@@ -223,7 +192,7 @@ static inline int dentry_string_cmp(const unsigned char *cs, const unsigned char
                if (!tcount)
                        return 0;
        }
-        mask = ~(~0ul << tcount*8);
+        mask = bytemask_from_count(tcount);
        return unlikely(!!((a ^ b) & mask));
 }
@@ -469,7 +438,7 @@ static struct dentry *d_kill(struct dentry *dentry, struct dentry *parent)
 {
        list_del(&dentry->d_u.d_child);
        /*
-         * Inform try_to_ascend() that we are no longer attached to the
+         * Inform d_walk() that we are no longer attached to the
         * dentry tree
         */
        dentry->d_flags |= DCACHE_DENTRY_KILLED;
@@ -1069,34 +1038,6 @@ void shrink_dcache_sb(struct super_block *sb)
 }
 EXPORT_SYMBOL(shrink_dcache_sb);
-/*
- * This tries to ascend one level of parenthood, but
- * we can race with renaming, so we need to re-check
- * the parenthood after dropping the lock and check
- * that the sequence number still matches.
- */
-static struct dentry *try_to_ascend(struct dentry *old, unsigned seq)
-{
-        struct dentry *new = old->d_parent;
-        rcu_read_lock();
-        spin_unlock(&old->d_lock);
-        spin_lock(&new->d_lock);
-        /*
-         * might go back up the wrong parent if we have had a rename
-         * or deletion
-         */
-        if (new != old->d_parent ||
-                 (old->d_flags & DCACHE_DENTRY_KILLED) ||
-                 need_seqretry(&rename_lock, seq)) {
-                spin_unlock(&new->d_lock);
-                new = NULL;
-        }
-        rcu_read_unlock();
-        return new;
-}
 /**
 * enum d_walk_ret - action to talke during tree walk
 * @D_WALK_CONTINUE:    contrinue walk
@@ -1185,9 +1126,24 @@ resume:
         */
        if (this_parent != parent) {
                struct dentry *child = this_parent;
-                this_parent = try_to_ascend(this_parent, seq);
+                this_parent = child->d_parent;
-                if (!this_parent)
+                rcu_read_lock();
+                spin_unlock(&child->d_lock);
+                spin_lock(&this_parent->d_lock);
+                /*
+                 * might go back up the wrong parent if we have had a rename
+                 * or deletion
+                 */
+                if (this_parent != child->d_parent ||
+                         (child->d_flags & DCACHE_DENTRY_KILLED) ||
+                         need_seqretry(&rename_lock, seq)) {
+                        spin_unlock(&this_parent->d_lock);
+                        rcu_read_unlock();
                        goto rename_retry;
+                }
+                rcu_read_unlock();
                next = child->d_u.d_child.next;
                goto resume;
        }
diff --git a/fs/ecryptfs/file.c b/fs/ecryptfs/file.c
index 2229a74aeeed..b1eaa7a1f82c 100644
--- a/fs/ecryptfs/file.c
+++ b/fs/ecryptfs/file.c
@@ -313,11 +313,9 @@ static int ecryptfs_fasync(int fd, struct file *file, int flag)
 static long
 ecryptfs_unlocked_ioctl(struct file *file, unsigned int cmd, unsigned long arg)
 {
-        struct file *lower_file = NULL;
+        struct file *lower_file = ecryptfs_file_to_lower(file);
        long rc = -ENOTTY;
-        if (ecryptfs_file_to_private(file))
-                lower_file = ecryptfs_file_to_lower(file);
        if (lower_file->f_op->unlocked_ioctl)
                rc = lower_file->f_op->unlocked_ioctl(lower_file, cmd, arg);
        return rc;
@@ -327,11 +325,9 @@ ecryptfs_unlocked_ioctl(struct file *file, unsigned int cmd, unsigned long arg)
 static long
 ecryptfs_compat_ioctl(struct file *file, unsigned int cmd, unsigned long arg)
 {
-        struct file *lower_file = NULL;
+        struct file *lower_file = ecryptfs_file_to_lower(file);
        long rc = -ENOIOCTLCMD;
-        if (ecryptfs_file_to_private(file))
-                lower_file = ecryptfs_file_to_lower(file);
        if (lower_file->f_op && lower_file->f_op->compat_ioctl)
                rc = lower_file->f_op->compat_ioctl(lower_file, cmd, arg);
        return rc;
diff --git a/fs/efivarfs/super.c b/fs/efivarfs/super.c
index a8766b880c07..becc725a1953 100644
--- a/fs/efivarfs/super.c
+++ b/fs/efivarfs/super.c
@@ -83,19 +83,10 @@ static int efivarfs_d_hash(const struct dentry *dentry, struct qstr *qstr)
        return 0;
 }
-/*
- * Retaining negative dentries for an in-memory filesystem just wastes
- * memory and lookup time: arrange for them to be deleted immediately.
- */
-static int efivarfs_delete_dentry(const struct dentry *dentry)
-{
-        return 1;
-}
 static struct dentry_operations efivarfs_d_ops = {
        .d_compare = efivarfs_d_compare,
        .d_hash = efivarfs_d_hash,
-        .d_delete = efivarfs_delete_dentry,
+        .d_delete = always_delete_dentry,
 };
 static struct dentry *efivarfs_alloc_dentry(struct dentry *parent, char *name)
diff --git a/fs/eventpoll.c b/fs/eventpoll.c
index 79b65c3b9e87..8b5e2584c840 100644
--- a/fs/eventpoll.c
+++ b/fs/eventpoll.c
@@ -1852,8 +1852,7 @@ SYSCALL_DEFINE4(epoll_ctl, int, epfd, int, op, int, fd,
                goto error_tgt_fput;
        /* Check if EPOLLWAKEUP is allowed */
-        if ((epds.events & EPOLLWAKEUP) && !capable(CAP_BLOCK_SUSPEND))
+        ep_take_care_of_epollwakeup(&epds);
-                epds.events &= ~EPOLLWAKEUP;
        /*
         * We have to check that the file structure underneath the file descriptor
diff --git a/fs/exec.c b/fs/exec.c
index 977319fd77f3..7ea097f6b341 100644
--- a/fs/exec.c
+++ b/fs/exec.c
@@ -1380,10 +1380,6 @@ int search_binary_handler(struct linux_binprm *bprm)
        if (retval)
                return retval;
-        retval = audit_bprm(bprm);
-        if (retval)
-                return retval;
        retval = -ENOENT;
 retry:
        read_lock(&binfmt_lock);
@@ -1431,6 +1427,7 @@ static int exec_binprm(struct linux_binprm *bprm)
        ret = search_binary_handler(bprm);
        if (ret >= 0) {
+                audit_bprm(bprm);
                trace_sched_process_exec(current, old_pid, bprm);
                ptrace_event(PTRACE_EVENT_EXEC, old_vpid);
                current->did_exec = 1;
diff --git a/fs/gfs2/glock.c b/fs/gfs2/glock.c
index e66a8009aff1..c8420f7e4db6 100644
--- a/fs/gfs2/glock.c
+++ b/fs/gfs2/glock.c
@@ -1899,7 +1899,8 @@ static int gfs2_glock_iter_next(struct gfs2_glock_iter *gi)
                        gi->nhash = 0;
                }
        /* Skip entries for other sb and dead entries */
-        } while (gi->sdp != gi->gl->gl_sbd || __lockref_is_dead(&gl->gl_lockref));
+        } while (gi->sdp != gi->gl->gl_sbd ||
+                 __lockref_is_dead(&gi->gl->gl_lockref));
        return 0;
 }
diff --git a/fs/gfs2/inode.c b/fs/gfs2/inode.c
index 1615df16cf4e..7119504159f1 100644
--- a/fs/gfs2/inode.c
+++ b/fs/gfs2/inode.c
@@ -1171,8 +1171,11 @@ static int gfs2_atomic_open(struct inode *dir, struct dentry *dentry,
        if (d != NULL)
                dentry = d;
        if (dentry->d_inode) {
-                if (!(*opened & FILE_OPENED))
+                if (!(*opened & FILE_OPENED)) {
+                        if (d == NULL)
+                                dget(dentry);
                        return finish_no_open(file, dentry);
+                }
                dput(d);
                return 0;
        }
diff --git a/fs/gfs2/lock_dlm.c b/fs/gfs2/lock_dlm.c
index c8423d6de6c3..2a6ba06bee6f 100644
--- a/fs/gfs2/lock_dlm.c
+++ b/fs/gfs2/lock_dlm.c
@@ -466,19 +466,19 @@ static void gdlm_cancel(struct gfs2_glock *gl)
 static void control_lvb_read(struct lm_lockstruct *ls, uint32_t *lvb_gen,
                             char *lvb_bits)
 {
-        uint32_t gen;
+        __le32 gen;
        memcpy(lvb_bits, ls->ls_control_lvb, GDLM_LVB_SIZE);
-        memcpy(&gen, lvb_bits, sizeof(uint32_t));
+        memcpy(&gen, lvb_bits, sizeof(__le32));
        *lvb_gen = le32_to_cpu(gen);
 }
 static void control_lvb_write(struct lm_lockstruct *ls, uint32_t lvb_gen,
                              char *lvb_bits)
 {
-        uint32_t gen;
+        __le32 gen;
        memcpy(ls->ls_control_lvb, lvb_bits, GDLM_LVB_SIZE);
        gen = cpu_to_le32(lvb_gen);
-        memcpy(ls->ls_control_lvb, &gen, sizeof(uint32_t));
+        memcpy(ls->ls_control_lvb, &gen, sizeof(__le32));
 }
 static int all_jid_bits_clear(char *lvb)
diff --git a/fs/gfs2/quota.c b/fs/gfs2/quota.c
index 453b50eaddec..98236d0df3ca 100644
--- a/fs/gfs2/quota.c
+++ b/fs/gfs2/quota.c
@@ -667,7 +667,7 @@ static int gfs2_adjust_quota(struct gfs2_inode *ip, loff_t loc,
        struct buffer_head *bh;
        struct page *page;
        void *kaddr, *ptr;
-        struct gfs2_quota q, *qp;
+        struct gfs2_quota q;
        int err, nbytes;
        u64 size;
@@ -683,28 +683,25 @@ static int gfs2_adjust_quota(struct gfs2_inode *ip, loff_t loc,
                return err;
        err = -EIO;
-        qp = &q;
+        be64_add_cpu(&q.qu_value, change);
-        qp->qu_value = be64_to_cpu(qp->qu_value);
+        qd->qd_qb.qb_value = q.qu_value;
-        qp->qu_value += change;
-        qp->qu_value = cpu_to_be64(qp->qu_value);
-        qd->qd_qb.qb_value = qp->qu_value;
        if (fdq) {
                if (fdq->d_fieldmask & FS_DQ_BSOFT) {
-                        qp->qu_warn = cpu_to_be64(fdq->d_blk_softlimit >> sdp->sd_fsb2bb_shift);
+                        q.qu_warn = cpu_to_be64(fdq->d_blk_softlimit >> sdp->sd_fsb2bb_shift);
-                        qd->qd_qb.qb_warn = qp->qu_warn;
+                        qd->qd_qb.qb_warn = q.qu_warn;
                }
                if (fdq->d_fieldmask & FS_DQ_BHARD) {
-                        qp->qu_limit = cpu_to_be64(fdq->d_blk_hardlimit >> sdp->sd_fsb2bb_shift);
+                        q.qu_limit = cpu_to_be64(fdq->d_blk_hardlimit >> sdp->sd_fsb2bb_shift);
-                        qd->qd_qb.qb_limit = qp->qu_limit;
+                        qd->qd_qb.qb_limit = q.qu_limit;
                }
                if (fdq->d_fieldmask & FS_DQ_BCOUNT) {
-                        qp->qu_value = cpu_to_be64(fdq->d_bcount >> sdp->sd_fsb2bb_shift);
+                        q.qu_value = cpu_to_be64(fdq->d_bcount >> sdp->sd_fsb2bb_shift);
-                        qd->qd_qb.qb_value = qp->qu_value;
+                        qd->qd_qb.qb_value = q.qu_value;
                }
        }
        /* Write the quota into the quota file on disk */
-        ptr = qp;
+        ptr = &q;
        nbytes = sizeof(struct gfs2_quota);
 get_a_page:
        page = find_or_create_page(mapping, index, GFP_NOFS);
diff --git a/fs/gfs2/rgrp.c b/fs/gfs2/rgrp.c
index 4d83abdd5635..c8d6161bd682 100644
--- a/fs/gfs2/rgrp.c
+++ b/fs/gfs2/rgrp.c
@@ -1127,7 +1127,7 @@ int gfs2_rgrp_bh_get(struct gfs2_rgrpd *rgd)
                rgd->rd_flags |= (GFS2_RDF_UPTODATE | GFS2_RDF_CHECK);
                rgd->rd_free_clone = rgd->rd_free;
        }
-        if (be32_to_cpu(GFS2_MAGIC) != rgd->rd_rgl->rl_magic) {
+        if (cpu_to_be32(GFS2_MAGIC) != rgd->rd_rgl->rl_magic) {
                rgd->rd_rgl->rl_unlinked = cpu_to_be32(count_unlinked(rgd));
                gfs2_rgrp_ondisk2lvb(rgd->rd_rgl,
                                     rgd->rd_bits[0].bi_bh->b_data);
@@ -1161,7 +1161,7 @@ int update_rgrp_lvb(struct gfs2_rgrpd *rgd)
        if (rgd->rd_flags & GFS2_RDF_UPTODATE)
                return 0;
-        if (be32_to_cpu(GFS2_MAGIC) != rgd->rd_rgl->rl_magic)
+        if (cpu_to_be32(GFS2_MAGIC) != rgd->rd_rgl->rl_magic)
                return gfs2_rgrp_bh_get(rgd);
        rl_flags = be32_to_cpu(rgd->rd_rgl->rl_flags);
diff --git a/fs/hfsplus/wrapper.c b/fs/hfsplus/wrapper.c
index b51a6079108d..e9a97a0d4314 100644
--- a/fs/hfsplus/wrapper.c
+++ b/fs/hfsplus/wrapper.c
@@ -24,13 +24,6 @@ struct hfsplus_wd {
        u16 embed_count;
 };
-static void hfsplus_end_io_sync(struct bio *bio, int err)
-{
-        if (err)
-                clear_bit(BIO_UPTODATE, &bio->bi_flags);
-        complete(bio->bi_private);
-}
 /*
 * hfsplus_submit_bio - Perfrom block I/O
 * @sb: super block of volume for I/O
@@ -53,7 +46,6 @@ static void hfsplus_end_io_sync(struct bio *bio, int err)
 int hfsplus_submit_bio(struct super_block *sb, sector_t sector,
                void *buf, void **data, int rw)
 {
-        DECLARE_COMPLETION_ONSTACK(wait);
        struct bio *bio;
        int ret = 0;
        u64 io_size;
@@ -73,8 +65,6 @@ int hfsplus_submit_bio(struct super_block *sb, sector_t sector,
        bio = bio_alloc(GFP_NOIO, 1);
        bio->bi_sector = sector;
        bio->bi_bdev = sb->s_bdev;
-        bio->bi_end_io = hfsplus_end_io_sync;
-        bio->bi_private = &wait;
        if (!(rw & WRITE) && data)
                *data = (u8 *)buf + offset;
@@ -93,12 +83,7 @@ int hfsplus_submit_bio(struct super_block *sb, sector_t sector,
                buf = (u8 *)buf + len;
        }
-        submit_bio(rw, bio);
+        ret = submit_bio_wait(rw, bio);
-        wait_for_completion(&wait);
-        if (!bio_flagged(bio, BIO_UPTODATE))
-                ret = -EIO;
 out:
        bio_put(bio);
        return ret < 0 ? ret : 0;
diff --git a/fs/hostfs/hostfs_kern.c b/fs/hostfs/hostfs_kern.c
index 25437280a207..db23ce1bd903 100644
--- a/fs/hostfs/hostfs_kern.c
+++ b/fs/hostfs/hostfs_kern.c
@@ -33,15 +33,6 @@ static inline struct hostfs_inode_info *HOSTFS_I(struct inode *inode)
 #define FILE_HOSTFS_I(file) HOSTFS_I(file_inode(file))
-static int hostfs_d_delete(const struct dentry *dentry)
-{
-        return 1;
-}
-static const struct dentry_operations hostfs_dentry_ops = {
-        .d_delete               = hostfs_d_delete,
-};
 /* Changed in hostfs_args before the kernel starts running */
 static char *root_ino = "";
 static int append = 0;
@@ -925,7 +916,7 @@ static int hostfs_fill_sb_common(struct super_block *sb, void *d, int silent)
        sb->s_blocksize_bits = 10;
        sb->s_magic = HOSTFS_SUPER_MAGIC;
        sb->s_op = &hostfs_sbops;
-        sb->s_d_op = &hostfs_dentry_ops;
+        sb->s_d_op = &simple_dentry_operations;
        sb->s_maxbytes = MAX_LFS_FILESIZE;
        /* NULL is printed as <NULL> by sprintf: avoid that. */
diff --git a/fs/libfs.c b/fs/libfs.c
index 5de06947ba5e..a1844244246f 100644
--- a/fs/libfs.c
+++ b/fs/libfs.c
@@ -47,10 +47,16 @@ EXPORT_SYMBOL(simple_statfs);
 * Retaining negative dentries for an in-memory filesystem just wastes
 * memory and lookup time: arrange for them to be deleted immediately.
 */
-static int simple_delete_dentry(const struct dentry *dentry)
+int always_delete_dentry(const struct dentry *dentry)
 {
        return 1;
 }
+EXPORT_SYMBOL(always_delete_dentry);
+const struct dentry_operations simple_dentry_operations = {
+        .d_delete = always_delete_dentry,
+};
+EXPORT_SYMBOL(simple_dentry_operations);
 /*
 * Lookup the data. This is trivial - if the dentry didn't already
@@ -58,10 +64,6 @@ static int simple_delete_dentry(const struct dentry *dentry)
 */
 struct dentry *simple_lookup(struct inode *dir, struct dentry *dentry, unsigned int flags)
 {
-        static const struct dentry_operations simple_dentry_operations = {
-                .d_delete = simple_delete_dentry,
-        };
        if (dentry->d_name.len > NAME_MAX)
                return ERR_PTR(-ENAMETOOLONG);
        if (!dentry->d_sb->s_d_op)
diff --git a/fs/logfs/dev_bdev.c b/fs/logfs/dev_bdev.c
index 550475ca6a0e..0f95f0d0b313 100644
--- a/fs/logfs/dev_bdev.c
+++ b/fs/logfs/dev_bdev.c
@@ -14,16 +14,10 @@
 #define PAGE_OFS(ofs) ((ofs) & (PAGE_SIZE-1))
-static void request_complete(struct bio *bio, int err)
-{
-        complete((struct completion *)bio->bi_private);
-}
 static int sync_request(struct page *page, struct block_device *bdev, int rw)
 {
        struct bio bio;
        struct bio_vec bio_vec;
-        struct completion complete;
        bio_init(&bio);
        bio.bi_max_vecs = 1;
@@ -35,13 +29,8 @@ static int sync_request(struct page *page, struct block_device *bdev, int rw)
        bio.bi_size = PAGE_SIZE;
        bio.bi_bdev = bdev;
        bio.bi_sector = page->index * (PAGE_SIZE >> 9);
-        init_completion(&complete);
-        bio.bi_private = &complete;
-        bio.bi_end_io = request_complete;
-        submit_bio(rw, &bio);
+        return submit_bio_wait(rw, &bio);
-        wait_for_completion(&complete);
-        return test_bit(BIO_UPTODATE, &bio.bi_flags) ? 0 : -EIO;
 }
 static int bdev_readpage(void *_sb, struct page *page)
diff --git a/fs/namei.c b/fs/namei.c
index e029a4cbff7d..3531deebad30 100644
--- a/fs/namei.c
+++ b/fs/namei.c
@@ -513,8 +513,7 @@ static int unlazy_walk(struct nameidata *nd, struct dentry *dentry)
        if (!lockref_get_not_dead(&parent->d_lockref)) {
                nd->path.dentry = NULL; 
-                rcu_read_unlock();
+                goto out;
-                return -ECHILD;
        }
        /*
@@ -1599,11 +1598,6 @@ static inline int nested_symlink(struct path *path, struct nameidata *nd)
 *   do a "get_unaligned()" if this helps and is sufficiently
 *   fast.
 *
- * - Little-endian machines (so that we can generate the mask
- *   of low bytes efficiently). Again, we *could* do a byte
- *   swapping load on big-endian architectures if that is not
- *   expensive enough to make the optimization worthless.
- *
 * - non-CONFIG_DEBUG_PAGEALLOC configurations (so that we
 *   do not trap on the (extremely unlikely) case of a page
 *   crossing operation.
@@ -1647,7 +1641,7 @@ unsigned int full_name_hash(const unsigned char *name, unsigned int len)
                if (!len)
                        goto done;
        }
-        mask = ~(~0ul << len*8);
+        mask = bytemask_from_count(len);
        hash += mask & a;
 done:
        return fold_hash(hash);
@@ -2435,6 +2429,7 @@ static int may_delete(struct inode *dir, struct dentry *victim, bool isdir)
 */
 static inline int may_create(struct inode *dir, struct dentry *child)
 {
+        audit_inode_child(dir, child, AUDIT_TYPE_CHILD_CREATE);
        if (child->d_inode)
                return -EEXIST;
        if (IS_DEADDIR(dir))
diff --git a/fs/nfs/blocklayout/blocklayout.h b/fs/nfs/blocklayout/blocklayout.h
index 8485978993e8..9838fb020473 100644
--- a/fs/nfs/blocklayout/blocklayout.h
+++ b/fs/nfs/blocklayout/blocklayout.h
@@ -36,6 +36,7 @@
 #include <linux/nfs_fs.h>
 #include <linux/sunrpc/rpc_pipe_fs.h>
+#include "../nfs4_fs.h"
 #include "../pnfs.h"
 #include "../netns.h"
diff --git a/fs/nfs/blocklayout/extents.c b/fs/nfs/blocklayout/extents.c
index 9c3e117c3ed1..4d0161442565 100644
--- a/fs/nfs/blocklayout/extents.c
+++ b/fs/nfs/blocklayout/extents.c
@@ -44,7 +44,7 @@
 static inline sector_t normalize(sector_t s, int base)
 {
        sector_t tmp = s; /* Since do_div modifies its argument */
-        return s - do_div(tmp, base);
+        return s - sector_div(tmp, base);
 }
 static inline sector_t normalize_up(sector_t s, int base)
diff --git a/fs/nfs/dns_resolve.c b/fs/nfs/dns_resolve.c
index fc0f95ec7358..d25f10fb4926 100644
--- a/fs/nfs/dns_resolve.c
+++ b/fs/nfs/dns_resolve.c
@@ -46,7 +46,9 @@ ssize_t nfs_dns_resolve_name(struct net *net, char *name, size_t namelen,
 #include <linux/sunrpc/cache.h>
 #include <linux/sunrpc/svcauth.h>
 #include <linux/sunrpc/rpc_pipe_fs.h>
+#include <linux/nfs_fs.h>
+#include "nfs4_fs.h"
 #include "dns_resolve.h"
 #include "cache_lib.h"
 #include "netns.h"
diff --git a/fs/nfs/inode.c b/fs/nfs/inode.c
index 18ab2da4eeb6..00ad1c2b217d 100644
--- a/fs/nfs/inode.c
+++ b/fs/nfs/inode.c
@@ -312,7 +312,7 @@ struct nfs4_label *nfs4_label_alloc(struct nfs_server *server, gfp_t flags)
 }
 EXPORT_SYMBOL_GPL(nfs4_label_alloc);
 #else
-void inline nfs_setsecurity(struct inode *inode, struct nfs_fattr *fattr,
+void nfs_setsecurity(struct inode *inode, struct nfs_fattr *fattr,
                                        struct nfs4_label *label)
 {
 }
diff --git a/fs/nfs/internal.h b/fs/nfs/internal.h
index bca6a3e3c49c..8b5cc04a8611 100644
--- a/fs/nfs/internal.h
+++ b/fs/nfs/internal.h
@@ -269,6 +269,21 @@ extern const u32 nfs41_maxgetdevinfo_overhead;
 extern struct rpc_procinfo nfs4_procedures[];
 #endif
+#ifdef CONFIG_NFS_V4_SECURITY_LABEL
+extern struct nfs4_label *nfs4_label_alloc(struct nfs_server *server, gfp_t flags);
+static inline void nfs4_label_free(struct nfs4_label *label)
+{
+        if (label) {
+                kfree(label->label);
+                kfree(label);
+        }
+        return;
+}
+#else
+static inline struct nfs4_label *nfs4_label_alloc(struct nfs_server *server, gfp_t flags) { return NULL; }
+static inline void nfs4_label_free(void *label) {}
+#endif /* CONFIG_NFS_V4_SECURITY_LABEL */
 /* proc.c */
 void nfs_close_context(struct nfs_open_context *ctx, int is_sync);
 extern struct nfs_client *nfs_init_client(struct nfs_client *clp,
diff --git a/fs/nfs/nfs4_fs.h b/fs/nfs/nfs4_fs.h
index 3ce79b04522e..5609edc742a0 100644
--- a/fs/nfs/nfs4_fs.h
+++ b/fs/nfs/nfs4_fs.h
@@ -9,6 +9,14 @@
 #ifndef __LINUX_FS_NFS_NFS4_FS_H
 #define __LINUX_FS_NFS_NFS4_FS_H
+#if defined(CONFIG_NFS_V4_2)
+#define NFS4_MAX_MINOR_VERSION 2
+#elif defined(CONFIG_NFS_V4_1)
+#define NFS4_MAX_MINOR_VERSION 1
+#else
+#define NFS4_MAX_MINOR_VERSION 0
+#endif
 #if IS_ENABLED(CONFIG_NFS_V4)
 #define NFS4_MAX_LOOP_ON_RECOVER (10)
diff --git a/fs/nfs/nfs4proc.c b/fs/nfs/nfs4proc.c
index 659990c0109e..15052b81df42 100644
--- a/fs/nfs/nfs4proc.c
+++ b/fs/nfs/nfs4proc.c
@@ -2518,9 +2518,8 @@ static void nfs4_close_done(struct rpc_task *task, void *data)
                                                     calldata->roc_barrier);
                        nfs_set_open_stateid(state, &calldata->res.stateid, 0);
                        renew_lease(server, calldata->timestamp);
-                        nfs4_close_clear_stateid_flags(state,
-                                        calldata->arg.fmode);
                        break;
+                case -NFS4ERR_ADMIN_REVOKED:
                case -NFS4ERR_STALE_STATEID:
                case -NFS4ERR_OLD_STATEID:
                case -NFS4ERR_BAD_STATEID:
@@ -2528,9 +2527,13 @@ static void nfs4_close_done(struct rpc_task *task, void *data)
                        if (calldata->arg.fmode == 0)
                                break;
                default:
-                        if (nfs4_async_handle_error(task, server, state) == -EAGAIN)
+                        if (nfs4_async_handle_error(task, server, state) == -EAGAIN) {
                                rpc_restart_call_prepare(task);
+                                goto out_release;
+                        }
        }
+        nfs4_close_clear_stateid_flags(state, calldata->arg.fmode);
+out_release:
        nfs_release_seqid(calldata->arg.seqid);
        nfs_refresh_inode(calldata->inode, calldata->res.fattr);
        dprintk("%s: done, ret = %d!\n", __func__, task->tk_status);
@@ -4802,7 +4805,7 @@ nfs4_async_handle_error(struct rpc_task *task, const struct nfs_server *server,
                        dprintk("%s ERROR %d, Reset session\n", __func__,
                                task->tk_status);
                        nfs4_schedule_session_recovery(clp->cl_session, task->tk_status);
-                        goto restart_call;
+                        goto wait_on_recovery;
 #endif /* CONFIG_NFS_V4_1 */
                case -NFS4ERR_DELAY:
                        nfs_inc_server_stats(server, NFSIOS_DELAY);
@@ -4987,11 +4990,17 @@ static void nfs4_delegreturn_done(struct rpc_task *task, void *calldata)
        trace_nfs4_delegreturn_exit(&data->args, &data->res, task->tk_status);
        switch (task->tk_status) {
-        case -NFS4ERR_STALE_STATEID:
-        case -NFS4ERR_EXPIRED:
        case 0:
                renew_lease(data->res.server, data->timestamp);
                break;
+        case -NFS4ERR_ADMIN_REVOKED:
+        case -NFS4ERR_DELEG_REVOKED:
+        case -NFS4ERR_BAD_STATEID:
+        case -NFS4ERR_OLD_STATEID:
+        case -NFS4ERR_STALE_STATEID:
+        case -NFS4ERR_EXPIRED:
+                task->tk_status = 0;
+                break;
        default:
                if (nfs4_async_handle_error(task, data->res.server, NULL) ==
                                -EAGAIN) {
@@ -7589,7 +7598,14 @@ static void nfs4_layoutreturn_done(struct rpc_task *task, void *calldata)
                return;
        server = NFS_SERVER(lrp->args.inode);
-        if (nfs4_async_handle_error(task, server, NULL) == -EAGAIN) {
+        switch (task->tk_status) {
+        default:
+                task->tk_status = 0;
+        case 0:
+                break;
+        case -NFS4ERR_DELAY:
+                if (nfs4_async_handle_error(task, server, NULL) != -EAGAIN)
+                        break;
                rpc_restart_call_prepare(task);
                return;
        }
diff --git a/fs/nfsd/nfs4xdr.c b/fs/nfsd/nfs4xdr.c
index 088de1355e93..ee7237f99f54 100644
--- a/fs/nfsd/nfs4xdr.c
+++ b/fs/nfsd/nfs4xdr.c
@@ -141,8 +141,8 @@ xdr_error:					\
 static void next_decode_page(struct nfsd4_compoundargs *argp)
 {
-        argp->pagelist++;
        argp->p = page_address(argp->pagelist[0]);
+        argp->pagelist++;
        if (argp->pagelen < PAGE_SIZE) {
                argp->end = argp->p + (argp->pagelen>>2);
                argp->pagelen = 0;
@@ -1229,6 +1229,7 @@ nfsd4_decode_write(struct nfsd4_compoundargs *argp, struct nfsd4_write *write)
                len -= pages * PAGE_SIZE;
                argp->p = (__be32 *)page_address(argp->pagelist[0]);
+                argp->pagelist++;
                argp->end = argp->p + XDR_QUADLEN(PAGE_SIZE);
        }
        argp->p += XDR_QUADLEN(len);
diff --git a/fs/nfsd/nfscache.c b/fs/nfsd/nfscache.c
index 9186c7ce0b14..b6af150c96b8 100644
--- a/fs/nfsd/nfscache.c
+++ b/fs/nfsd/nfscache.c
@@ -132,6 +132,13 @@ nfsd_reply_cache_alloc(void)
 }
 static void
+nfsd_reply_cache_unhash(struct svc_cacherep *rp)
+{
+        hlist_del_init(&rp->c_hash);
+        list_del_init(&rp->c_lru);
+}
+static void
 nfsd_reply_cache_free_locked(struct svc_cacherep *rp)
 {
        if (rp->c_type == RC_REPLBUFF && rp->c_replvec.iov_base) {
@@ -417,7 +424,7 @@ nfsd_cache_lookup(struct svc_rqst *rqstp)
                rp = list_first_entry(&lru_head, struct svc_cacherep, c_lru);
                if (nfsd_cache_entry_expired(rp) ||
                    num_drc_entries >= max_drc_entries) {
-                        lru_put_end(rp);
+                        nfsd_reply_cache_unhash(rp);
                        prune_cache_entries();
                        goto search_cache;
                }
diff --git a/fs/nfsd/vfs.c b/fs/nfsd/vfs.c
index 94b5f5d2bfed..7eea63cada1d 100644
--- a/fs/nfsd/vfs.c
+++ b/fs/nfsd/vfs.c
@@ -298,41 +298,12 @@ commit_metadata(struct svc_fh *fhp)
 }
 /*
- * Set various file attributes.
+ * Go over the attributes and take care of the small differences between
- * N.B. After this call fhp needs an fh_put
+ * NFS semantics and what Linux expects.
 */
-__be32
+static void
-nfsd_setattr(struct svc_rqst *rqstp, struct svc_fh *fhp, struct iattr *iap,
+nfsd_sanitize_attrs(struct inode *inode, struct iattr *iap)
-             int check_guard, time_t guardtime)
 {
-        struct dentry   *dentry;
-        struct inode    *inode;
-        int             accmode = NFSD_MAY_SATTR;
-        umode_t         ftype = 0;
-        __be32          err;
-        int             host_err;
-        int             size_change = 0;
-        if (iap->ia_valid & (ATTR_ATIME | ATTR_MTIME | ATTR_SIZE))
-                accmode |= NFSD_MAY_WRITE|NFSD_MAY_OWNER_OVERRIDE;
-        if (iap->ia_valid & ATTR_SIZE)
-                ftype = S_IFREG;
-        /* Get inode */
-        err = fh_verify(rqstp, fhp, ftype, accmode);
-        if (err)
-                goto out;
-        dentry = fhp->fh_dentry;
-        inode = dentry->d_inode;
-        /* Ignore any mode updates on symlinks */
-        if (S_ISLNK(inode->i_mode))
-                iap->ia_valid &= ~ATTR_MODE;
-        if (!iap->ia_valid)
-                goto out;
        /*
         * NFSv2 does not differentiate between "set-[ac]time-to-now"
         * which only requires access, and "set-[ac]time-to-X" which
@@ -342,8 +313,7 @@ nfsd_setattr(struct svc_rqst *rqstp, struct svc_fh *fhp, struct iattr *iap,
         * convert to "set to now" instead of "set to explicit time"
         *
         * We only call inode_change_ok as the last test as technically
-         * it is not an interface that we should be using.  It is only
+         * it is not an interface that we should be using.
-         * valid if the filesystem does not define it's own i_op->setattr.
         */
 #define BOTH_TIME_SET (ATTR_ATIME_SET | ATTR_MTIME_SET)
 #define MAX_TOUCH_TIME_ERROR (30*60)
@@ -369,30 +339,6 @@ nfsd_setattr(struct svc_rqst *rqstp, struct svc_fh *fhp, struct iattr *iap,
                        iap->ia_valid &= ~BOTH_TIME_SET;
                }
        }
-            
-        /*
-         * The size case is special.
-         * It changes the file as well as the attributes.
-         */
-        if (iap->ia_valid & ATTR_SIZE) {
-                if (iap->ia_size < inode->i_size) {
-                        err = nfsd_permission(rqstp, fhp->fh_export, dentry,
-                                        NFSD_MAY_TRUNC|NFSD_MAY_OWNER_OVERRIDE);
-                        if (err)
-                                goto out;
-                }
-                host_err = get_write_access(inode);
-                if (host_err)
-                        goto out_nfserr;
-                size_change = 1;
-                host_err = locks_verify_truncate(inode, NULL, iap->ia_size);
-                if (host_err) {
-                        put_write_access(inode);
-                        goto out_nfserr;
-                }
-        }
        /* sanitize the mode change */
        if (iap->ia_valid & ATTR_MODE) {
@@ -415,32 +361,111 @@ nfsd_setattr(struct svc_rqst *rqstp, struct svc_fh *fhp, struct iattr *iap,
                        iap->ia_valid |= (ATTR_KILL_SUID | ATTR_KILL_SGID);
                }
        }
+}
-        /* Change the attributes. */
+static __be32
+nfsd_get_write_access(struct svc_rqst *rqstp, struct svc_fh *fhp,
+                struct iattr *iap)
+{
+        struct inode *inode = fhp->fh_dentry->d_inode;
+        int host_err;
-        iap->ia_valid |= ATTR_CTIME;
+        if (iap->ia_size < inode->i_size) {
+                __be32 err;
-        err = nfserr_notsync;
+                err = nfsd_permission(rqstp, fhp->fh_export, fhp->fh_dentry,
-        if (!check_guard || guardtime == inode->i_ctime.tv_sec) {
+                                NFSD_MAY_TRUNC | NFSD_MAY_OWNER_OVERRIDE);
-                host_err = nfsd_break_lease(inode);
+                if (err)
-                if (host_err)
+                        return err;
-                        goto out_nfserr;
+        }
-                fh_lock(fhp);
-                host_err = notify_change(dentry, iap, NULL);
+        host_err = get_write_access(inode);
-                err = nfserrno(host_err);
+        if (host_err)
-                fh_unlock(fhp);
+                goto out_nfserrno;
+        host_err = locks_verify_truncate(inode, NULL, iap->ia_size);
+        if (host_err)
+                goto out_put_write_access;
+        return 0;
+out_put_write_access:
+        put_write_access(inode);
+out_nfserrno:
+        return nfserrno(host_err);
+}
+/*
+ * Set various file attributes.  After this call fhp needs an fh_put.
+ */
+__be32
+nfsd_setattr(struct svc_rqst *rqstp, struct svc_fh *fhp, struct iattr *iap,
+             int check_guard, time_t guardtime)
+{
+        struct dentry   *dentry;
+        struct inode    *inode;
+        int             accmode = NFSD_MAY_SATTR;
+        umode_t         ftype = 0;
+        __be32          err;
+        int             host_err;
+        int             size_change = 0;
+        if (iap->ia_valid & (ATTR_ATIME | ATTR_MTIME | ATTR_SIZE))
+                accmode |= NFSD_MAY_WRITE|NFSD_MAY_OWNER_OVERRIDE;
+        if (iap->ia_valid & ATTR_SIZE)
+                ftype = S_IFREG;
+        /* Get inode */
+        err = fh_verify(rqstp, fhp, ftype, accmode);
+        if (err)
+                goto out;
+        dentry = fhp->fh_dentry;
+        inode = dentry->d_inode;
+        /* Ignore any mode updates on symlinks */
+        if (S_ISLNK(inode->i_mode))
+                iap->ia_valid &= ~ATTR_MODE;
+        if (!iap->ia_valid)
+                goto out;
+        nfsd_sanitize_attrs(inode, iap);
+        /*
+         * The size case is special, it changes the file in addition to the
+         * attributes.
+         */
+        if (iap->ia_valid & ATTR_SIZE) {
+                err = nfsd_get_write_access(rqstp, fhp, iap);
+                if (err)
+                        goto out;
+                size_change = 1;
        }
+        iap->ia_valid |= ATTR_CTIME;
+        if (check_guard && guardtime != inode->i_ctime.tv_sec) {
+                err = nfserr_notsync;
+                goto out_put_write_access;
+        }
+        host_err = nfsd_break_lease(inode);
+        if (host_err)
+                goto out_put_write_access_nfserror;
+        fh_lock(fhp);
+        host_err = notify_change(dentry, iap, NULL);
+        fh_unlock(fhp);
+out_put_write_access_nfserror:
+        err = nfserrno(host_err);
+out_put_write_access:
        if (size_change)
                put_write_access(inode);
        if (!err)
                commit_metadata(fhp);
 out:
        return err;
-out_nfserr:
-        err = nfserrno(host_err);
-        goto out;
 }
 #if defined(CONFIG_NFSD_V2_ACL) || \
diff --git a/fs/pipe.c b/fs/pipe.c
index d2c45e14e6d8..0e0752ef2715 100644
--- a/fs/pipe.c
+++ b/fs/pipe.c
@@ -726,11 +726,25 @@ pipe_poll(struct file *filp, poll_table *wait)
        return mask;
 }
+static void put_pipe_info(struct inode *inode, struct pipe_inode_info *pipe)
+{
+        int kill = 0;
+        spin_lock(&inode->i_lock);
+        if (!--pipe->files) {
+                inode->i_pipe = NULL;
+                kill = 1;
+        }
+        spin_unlock(&inode->i_lock);
+        if (kill)
+                free_pipe_info(pipe);
+}
 static int
 pipe_release(struct inode *inode, struct file *file)
 {
-        struct pipe_inode_info *pipe = inode->i_pipe;
+        struct pipe_inode_info *pipe = file->private_data;
-        int kill = 0;
        __pipe_lock(pipe);
        if (file->f_mode & FMODE_READ)
@@ -743,17 +757,9 @@ pipe_release(struct inode *inode, struct file *file)
                kill_fasync(&pipe->fasync_readers, SIGIO, POLL_IN);
                kill_fasync(&pipe->fasync_writers, SIGIO, POLL_OUT);
        }
-        spin_lock(&inode->i_lock);
-        if (!--pipe->files) {
-                inode->i_pipe = NULL;
-                kill = 1;
-        }
-        spin_unlock(&inode->i_lock);
        __pipe_unlock(pipe);
-        if (kill)
+        put_pipe_info(inode, pipe);
-                free_pipe_info(pipe);
        return 0;
 }
@@ -1014,7 +1020,6 @@ static int fifo_open(struct inode *inode, struct file *filp)
 {
        struct pipe_inode_info *pipe;
        bool is_pipe = inode->i_sb->s_magic == PIPEFS_MAGIC;
-        int kill = 0;
        int ret;
        filp->f_version = 0;
@@ -1130,15 +1135,9 @@ err_wr:
        goto err;
 err:
-        spin_lock(&inode->i_lock);
-        if (!--pipe->files) {
-                inode->i_pipe = NULL;
-                kill = 1;
-        }
-        spin_unlock(&inode->i_lock);
        __pipe_unlock(pipe);
-        if (kill)
-                free_pipe_info(pipe);
+        put_pipe_info(inode, pipe);
        return ret;
 }
diff --git a/fs/proc/base.c b/fs/proc/base.c
index 1485e38daaa3..03c8d747be48 100644
--- a/fs/proc/base.c
+++ b/fs/proc/base.c
@@ -1151,10 +1151,16 @@ static ssize_t proc_loginuid_write(struct file * file, const char __user * buf,
                goto out_free_page;
        }
-        kloginuid = make_kuid(file->f_cred->user_ns, loginuid);
-        if (!uid_valid(kloginuid)) {
+        /* is userspace tring to explicitly UNSET the loginuid? */
-                length = -EINVAL;
+        if (loginuid == AUDIT_UID_UNSET) {
-                goto out_free_page;
+                kloginuid = INVALID_UID;
+        } else {
+                kloginuid = make_kuid(file->f_cred->user_ns, loginuid);
+                if (!uid_valid(kloginuid)) {
+                        length = -EINVAL;
+                        goto out_free_page;
+                }
        }
        length = audit_set_loginuid(kloginuid);
diff --git a/fs/proc/generic.c b/fs/proc/generic.c
index 737e15615b04..cca93b6fb9a9 100644
--- a/fs/proc/generic.c
+++ b/fs/proc/generic.c
@@ -175,22 +175,6 @@ static const struct inode_operations proc_link_inode_operations = {
 };
 /*
- * As some entries in /proc are volatile, we want to 
- * get rid of unused dentries.  This could be made 
- * smarter: we could keep a "volatile" flag in the 
- * inode to indicate which ones to keep.
- */
-static int proc_delete_dentry(const struct dentry * dentry)
-{
-        return 1;
-}
-static const struct dentry_operations proc_dentry_operations =
-{
-        .d_delete       = proc_delete_dentry,
-};
-/*
 * Don't create negative dentries here, return -ENOENT by hand
 * instead.
 */
@@ -209,7 +193,7 @@ struct dentry *proc_lookup_de(struct proc_dir_entry *de, struct inode *dir,
                        inode = proc_get_inode(dir->i_sb, de);
                        if (!inode)
                                return ERR_PTR(-ENOMEM);
-                        d_set_d_op(dentry, &proc_dentry_operations);
+                        d_set_d_op(dentry, &simple_dentry_operations);
                        d_add(dentry, inode);
                        return NULL;
                }
diff --git a/fs/proc/inode.c b/fs/proc/inode.c
index 28955d4b7218..124fc43c7090 100644
--- a/fs/proc/inode.c
+++ b/fs/proc/inode.c
@@ -292,16 +292,20 @@ proc_reg_get_unmapped_area(struct file *file, unsigned long orig_addr,
 {
        struct proc_dir_entry *pde = PDE(file_inode(file));
        unsigned long rv = -EIO;
-        unsigned long (*get_area)(struct file *, unsigned long, unsigned long,
-                                  unsigned long, unsigned long) = NULL;
        if (use_pde(pde)) {
+                typeof(proc_reg_get_unmapped_area) *get_area;
+                get_area = pde->proc_fops->get_unmapped_area;
 #ifdef CONFIG_MMU
-                get_area = current->mm->get_unmapped_area;
+                if (!get_area)
+                        get_area = current->mm->get_unmapped_area;
 #endif
-                if (pde->proc_fops->get_unmapped_area)
-                        get_area = pde->proc_fops->get_unmapped_area;
                if (get_area)
                        rv = get_area(file, orig_addr, len, pgoff, flags);
+                else
+                        rv = orig_addr;
                unuse_pde(pde);
        }
        return rv;
diff --git a/fs/proc/namespaces.c b/fs/proc/namespaces.c
index 49a7fff2e83a..9ae46b87470d 100644
--- a/fs/proc/namespaces.c
+++ b/fs/proc/namespaces.c
@@ -42,12 +42,6 @@ static const struct inode_operations ns_inode_operations = {
        .setattr        = proc_setattr,
 };
-static int ns_delete_dentry(const struct dentry *dentry)
-{
-        /* Don't cache namespace inodes when not in use */
-        return 1;
-}
 static char *ns_dname(struct dentry *dentry, char *buffer, int buflen)
 {
        struct inode *inode = dentry->d_inode;
@@ -59,7 +53,7 @@ static char *ns_dname(struct dentry *dentry, char *buffer, int buflen)
 const struct dentry_operations ns_dentry_operations =
 {
-        .d_delete       = ns_delete_dentry,
+        .d_delete       = always_delete_dentry,
        .d_dname        = ns_dname,
 };
diff --git a/fs/squashfs/Kconfig b/fs/squashfs/Kconfig
index c70111ebefd4..b6fa8657dcbc 100644
--- a/fs/squashfs/Kconfig
+++ b/fs/squashfs/Kconfig
@@ -25,6 +25,78 @@ config SQUASHFS
          If unsure, say N.
+choice
+        prompt "File decompression options"
+        depends on SQUASHFS
+        help
+          Squashfs now supports two options for decompressing file
+          data.  Traditionally Squashfs has decompressed into an
+          intermediate buffer and then memcopied it into the page cache.
+          Squashfs now supports the ability to decompress directly into
+          the page cache.
+          If unsure, select "Decompress file data into an intermediate buffer"
+config SQUASHFS_FILE_CACHE
+        bool "Decompress file data into an intermediate buffer"
+        help
+          Decompress file data into an intermediate buffer and then
+          memcopy it into the page cache.
+config SQUASHFS_FILE_DIRECT
+        bool "Decompress files directly into the page cache"
+        help
+          Directly decompress file data into the page cache.
+          Doing so can significantly improve performance because
+          it eliminates a memcpy and it also removes the lock contention
+          on the single buffer.
+endchoice
+choice
+        prompt "Decompressor parallelisation options"
+        depends on SQUASHFS
+        help
+          Squashfs now supports three parallelisation options for
+          decompression.  Each one exhibits various trade-offs between
+          decompression performance and CPU and memory usage.
+          If in doubt, select "Single threaded compression"
+config SQUASHFS_DECOMP_SINGLE
+        bool "Single threaded compression"
+        help
+          Traditionally Squashfs has used single-threaded decompression.
+          Only one block (data or metadata) can be decompressed at any
+          one time.  This limits CPU and memory usage to a minimum.
+config SQUASHFS_DECOMP_MULTI
+        bool "Use multiple decompressors for parallel I/O"
+        help
+          By default Squashfs uses a single decompressor but it gives
+          poor performance on parallel I/O workloads when using multiple CPU
+          machines due to waiting on decompressor availability.
+          If you have a parallel I/O workload and your system has enough memory,
+          using this option may improve overall I/O performance.
+          This decompressor implementation uses up to two parallel
+          decompressors per core.  It dynamically allocates decompressors
+          on a demand basis.
+config SQUASHFS_DECOMP_MULTI_PERCPU
+        bool "Use percpu multiple decompressors for parallel I/O"
+        help
+          By default Squashfs uses a single decompressor but it gives
+          poor performance on parallel I/O workloads when using multiple CPU
+          machines due to waiting on decompressor availability.
+          This decompressor implementation uses a maximum of one
+          decompressor per core.  It uses percpu variables to ensure
+          decompression is load-balanced across the cores.
+endchoice
 config SQUASHFS_XATTR
        bool "Squashfs XATTR support"
        depends on SQUASHFS
diff --git a/fs/squashfs/Makefile b/fs/squashfs/Makefile
index 110b0476f3b4..4132520b4ff2 100644
--- a/fs/squashfs/Makefile
+++ b/fs/squashfs/Makefile
@@ -5,6 +5,11 @@
 obj-$(CONFIG_SQUASHFS) += squashfs.o
 squashfs-y += block.o cache.o dir.o export.o file.o fragment.o id.o inode.o
 squashfs-y += namei.o super.o symlink.o decompressor.o
+squashfs-$(CONFIG_SQUASHFS_FILE_CACHE) += file_cache.o
+squashfs-$(CONFIG_SQUASHFS_FILE_DIRECT) += file_direct.o page_actor.o
+squashfs-$(CONFIG_SQUASHFS_DECOMP_SINGLE) += decompressor_single.o
+squashfs-$(CONFIG_SQUASHFS_DECOMP_MULTI) += decompressor_multi.o
+squashfs-$(CONFIG_SQUASHFS_DECOMP_MULTI_PERCPU) += decompressor_multi_percpu.o
 squashfs-$(CONFIG_SQUASHFS_XATTR) += xattr.o xattr_id.o
 squashfs-$(CONFIG_SQUASHFS_LZO) += lzo_wrapper.o
 squashfs-$(CONFIG_SQUASHFS_XZ) += xz_wrapper.o
diff --git a/fs/squashfs/block.c b/fs/squashfs/block.c
index 41d108ecc9be..0cea9b9236d0 100644
--- a/fs/squashfs/block.c
+++ b/fs/squashfs/block.c
@@ -36,6 +36,7 @@
 #include "squashfs_fs_sb.h"
 #include "squashfs.h"
 #include "decompressor.h"
+#include "page_actor.h"
 /*
 * Read the metadata block length, this is stored in the first two
@@ -86,16 +87,16 @@ static struct buffer_head *get_block_length(struct super_block *sb,
 * generated a larger block - this does occasionally happen with compression
 * algorithms).
 */
-int squashfs_read_data(struct super_block *sb, void **buffer, u64 index,
+int squashfs_read_data(struct super_block *sb, u64 index, int length,
-                        int length, u64 *next_index, int srclength, int pages)
+                u64 *next_index, struct squashfs_page_actor *output)
 {
        struct squashfs_sb_info *msblk = sb->s_fs_info;
        struct buffer_head **bh;
        int offset = index & ((1 << msblk->devblksize_log2) - 1);
        u64 cur_index = index >> msblk->devblksize_log2;
-        int bytes, compressed, b = 0, k = 0, page = 0, avail;
+        int bytes, compressed, b = 0, k = 0, avail, i;
-        bh = kcalloc(((srclength + msblk->devblksize - 1)
+        bh = kcalloc(((output->length + msblk->devblksize - 1)
                >> msblk->devblksize_log2) + 1, sizeof(*bh), GFP_KERNEL);
        if (bh == NULL)
                return -ENOMEM;
@@ -111,9 +112,9 @@ int squashfs_read_data(struct super_block *sb, void **buffer, u64 index,
                        *next_index = index + length;
                TRACE("Block @ 0x%llx, %scompressed size %d, src size %d\n",
-                        index, compressed ? "" : "un", length, srclength);
+                        index, compressed ? "" : "un", length, output->length);
-                if (length < 0 || length > srclength ||
+                if (length < 0 || length > output->length ||
                                (index + length) > msblk->bytes_used)
                        goto read_failure;
@@ -145,7 +146,7 @@ int squashfs_read_data(struct super_block *sb, void **buffer, u64 index,
                TRACE("Block @ 0x%llx, %scompressed size %d\n", index,
                                compressed ? "" : "un", length);
-                if (length < 0 || length > srclength ||
+                if (length < 0 || length > output->length ||
                                        (index + length) > msblk->bytes_used)
                        goto block_release;
@@ -158,9 +159,15 @@ int squashfs_read_data(struct super_block *sb, void **buffer, u64 index,
                ll_rw_block(READ, b - 1, bh + 1);
        }
+        for (i = 0; i < b; i++) {
+                wait_on_buffer(bh[i]);
+                if (!buffer_uptodate(bh[i]))
+                        goto block_release;
+        }
        if (compressed) {
-                length = squashfs_decompress(msblk, buffer, bh, b, offset,
+                length = squashfs_decompress(msblk, bh, b, offset, length,
-                         length, srclength, pages);
+                        output);
                if (length < 0)
                        goto read_failure;
        } else {
@@ -168,22 +175,20 @@ int squashfs_read_data(struct super_block *sb, void **buffer, u64 index,
                 * Block is uncompressed.
                 */
                int in, pg_offset = 0;
+                void *data = squashfs_first_page(output);
                for (bytes = length; k < b; k++) {
                        in = min(bytes, msblk->devblksize - offset);
                        bytes -= in;
-                        wait_on_buffer(bh[k]);
-                        if (!buffer_uptodate(bh[k]))
-                                goto block_release;
                        while (in) {
                                if (pg_offset == PAGE_CACHE_SIZE) {
-                                        page++;
+                                        data = squashfs_next_page(output);
                                        pg_offset = 0;
                                }
                                avail = min_t(int, in, PAGE_CACHE_SIZE -
                                                pg_offset);
-                                memcpy(buffer[page] + pg_offset,
+                                memcpy(data + pg_offset, bh[k]->b_data + offset,
-                                                bh[k]->b_data + offset, avail);
+                                                avail);
                                in -= avail;
                                pg_offset += avail;
                                offset += avail;
@@ -191,6 +196,7 @@ int squashfs_read_data(struct super_block *sb, void **buffer, u64 index,
                        offset = 0;
                        put_bh(bh[k]);
                }
+                squashfs_finish_page(output);
        }
        kfree(bh);
diff --git a/fs/squashfs/cache.c b/fs/squashfs/cache.c
index af0b73802592..1cb70a0b2168 100644
--- a/fs/squashfs/cache.c
+++ b/fs/squashfs/cache.c
@@ -56,6 +56,7 @@
 #include "squashfs_fs.h"
 #include "squashfs_fs_sb.h"
 #include "squashfs.h"
+#include "page_actor.h"
 /*
 * Look-up block in cache, and increment usage count.  If not in cache, read
@@ -119,9 +120,8 @@ struct squashfs_cache_entry *squashfs_cache_get(struct super_block *sb,
                        entry->error = 0;
                        spin_unlock(&cache->lock);
-                        entry->length = squashfs_read_data(sb, entry->data,
+                        entry->length = squashfs_read_data(sb, block, length,
-                                block, length, &entry->next_index,
+                                &entry->next_index, entry->actor);
-                                cache->block_size, cache->pages);
                        spin_lock(&cache->lock);
@@ -220,6 +220,7 @@ void squashfs_cache_delete(struct squashfs_cache *cache)
                                kfree(cache->entry[i].data[j]);
                        kfree(cache->entry[i].data);
                }
+                kfree(cache->entry[i].actor);
        }
        kfree(cache->entry);
@@ -280,6 +281,13 @@ struct squashfs_cache *squashfs_cache_init(char *name, int entries,
                                goto cleanup;
                        }
                }
+                entry->actor = squashfs_page_actor_init(entry->data,
+                                                cache->pages, 0);
+                if (entry->actor == NULL) {
+                        ERROR("Failed to allocate %s cache entry\n", name);
+                        goto cleanup;
+                }
        }
        return cache;
@@ -410,6 +418,7 @@ void *squashfs_read_table(struct super_block *sb, u64 block, int length)
        int pages = (length + PAGE_CACHE_SIZE - 1) >> PAGE_CACHE_SHIFT;
        int i, res;
        void *table, *buffer, **data;
+        struct squashfs_page_actor *actor;
        table = buffer = kmalloc(length, GFP_KERNEL);
        if (table == NULL)
@@ -421,19 +430,28 @@ void *squashfs_read_table(struct super_block *sb, u64 block, int length)
                goto failed;
        }
+        actor = squashfs_page_actor_init(data, pages, length);
+        if (actor == NULL) {
+                res = -ENOMEM;
+                goto failed2;
+        }
        for (i = 0; i < pages; i++, buffer += PAGE_CACHE_SIZE)
                data[i] = buffer;
-        res = squashfs_read_data(sb, data, block, length |
+        res = squashfs_read_data(sb, block, length |
-                SQUASHFS_COMPRESSED_BIT_BLOCK, NULL, length, pages);
+                SQUASHFS_COMPRESSED_BIT_BLOCK, NULL, actor);
        kfree(data);
+        kfree(actor);
        if (res < 0)
                goto failed;
        return table;
+failed2:
+        kfree(data);
 failed:
        kfree(table);
        return ERR_PTR(res);
diff --git a/fs/squashfs/decompressor.c b/fs/squashfs/decompressor.c
index 3f6271d86abc..ac22fe73b0ad 100644
--- a/fs/squashfs/decompressor.c
+++ b/fs/squashfs/decompressor.c
@@ -30,6 +30,7 @@
 #include "squashfs_fs_sb.h"
 #include "decompressor.h"
 #include "squashfs.h"
+#include "page_actor.h"
 /*
 * This file (and decompressor.h) implements a decompressor framework for
@@ -37,29 +38,29 @@
 */
 static const struct squashfs_decompressor squashfs_lzma_unsupported_comp_ops = {
-        NULL, NULL, NULL, LZMA_COMPRESSION, "lzma", 0
+        NULL, NULL, NULL, NULL, LZMA_COMPRESSION, "lzma", 0
 };
 #ifndef CONFIG_SQUASHFS_LZO
 static const struct squashfs_decompressor squashfs_lzo_comp_ops = {
-        NULL, NULL, NULL, LZO_COMPRESSION, "lzo", 0
+        NULL, NULL, NULL, NULL, LZO_COMPRESSION, "lzo", 0
 };
 #endif
 #ifndef CONFIG_SQUASHFS_XZ
 static const struct squashfs_decompressor squashfs_xz_comp_ops = {
-        NULL, NULL, NULL, XZ_COMPRESSION, "xz", 0
+        NULL, NULL, NULL, NULL, XZ_COMPRESSION, "xz", 0
 };
 #endif
 #ifndef CONFIG_SQUASHFS_ZLIB
 static const struct squashfs_decompressor squashfs_zlib_comp_ops = {
-        NULL, NULL, NULL, ZLIB_COMPRESSION, "zlib", 0
+        NULL, NULL, NULL, NULL, ZLIB_COMPRESSION, "zlib", 0
 };
 #endif
 static const struct squashfs_decompressor squashfs_unknown_comp_ops = {
-        NULL, NULL, NULL, 0, "unknown", 0
+        NULL, NULL, NULL, NULL, 0, "unknown", 0
 };
 static const struct squashfs_decompressor *decompressor[] = {
@@ -83,10 +84,11 @@ const struct squashfs_decompressor *squashfs_lookup_decompressor(int id)
 }
-void *squashfs_decompressor_init(struct super_block *sb, unsigned short flags)
+static void *get_comp_opts(struct super_block *sb, unsigned short flags)
 {
        struct squashfs_sb_info *msblk = sb->s_fs_info;
-        void *strm, *buffer = NULL;
+        void *buffer = NULL, *comp_opts;
+        struct squashfs_page_actor *actor = NULL;
        int length = 0;
        /*
@@ -94,23 +96,46 @@ void *squashfs_decompressor_init(struct super_block *sb, unsigned short flags)
         */
        if (SQUASHFS_COMP_OPTS(flags)) {
                buffer = kmalloc(PAGE_CACHE_SIZE, GFP_KERNEL);
-                if (buffer == NULL)
+                if (buffer == NULL) {
-                        return ERR_PTR(-ENOMEM);
+                        comp_opts = ERR_PTR(-ENOMEM);
+                        goto out;
+                }
+                actor = squashfs_page_actor_init(&buffer, 1, 0);
+                if (actor == NULL) {
+                        comp_opts = ERR_PTR(-ENOMEM);
+                        goto out;
+                }
-                length = squashfs_read_data(sb, &buffer,
+                length = squashfs_read_data(sb,
-                        sizeof(struct squashfs_super_block), 0, NULL,
+                        sizeof(struct squashfs_super_block), 0, NULL, actor);
-                        PAGE_CACHE_SIZE, 1);
                if (length < 0) {
-                        strm = ERR_PTR(length);
+                        comp_opts = ERR_PTR(length);
-                        goto finished;
+                        goto out;
                }
        }
-        strm = msblk->decompressor->init(msblk, buffer, length);
+        comp_opts = squashfs_comp_opts(msblk, buffer, length);
-finished:
+out:
+        kfree(actor);
        kfree(buffer);
+        return comp_opts;
+}
+void *squashfs_decompressor_setup(struct super_block *sb, unsigned short flags)
+{
+        struct squashfs_sb_info *msblk = sb->s_fs_info;
+        void *stream, *comp_opts = get_comp_opts(sb, flags);
+        if (IS_ERR(comp_opts))
+                return comp_opts;
+        stream = squashfs_decompressor_create(msblk, comp_opts);
+        if (IS_ERR(stream))
+                kfree(comp_opts);
-        return strm;
+        return stream;
 }
diff --git a/fs/squashfs/decompressor.h b/fs/squashfs/decompressor.h
index 330073e29029..af0985321808 100644
--- a/fs/squashfs/decompressor.h
+++ b/fs/squashfs/decompressor.h
@@ -24,28 +24,22 @@
 */
 struct squashfs_decompressor {
-        void    *(*init)(struct squashfs_sb_info *, void *, int);
+        void    *(*init)(struct squashfs_sb_info *, void *);
+        void    *(*comp_opts)(struct squashfs_sb_info *, void *, int);
        void    (*free)(void *);
-        int     (*decompress)(struct squashfs_sb_info *, void **,
+        int     (*decompress)(struct squashfs_sb_info *, void *,
-                struct buffer_head **, int, int, int, int, int);
+                struct buffer_head **, int, int, int,
+                struct squashfs_page_actor *);
        int     id;
        char    *name;
        int     supported;
 };
-static inline void squashfs_decompressor_free(struct squashfs_sb_info *msblk,
+static inline void *squashfs_comp_opts(struct squashfs_sb_info *msblk,
-        void *s)
+                                                        void *buff, int length)
 {
-        if (msblk->decompressor)
+        return msblk->decompressor->comp_opts ?
-                msblk->decompressor->free(s);
+                msblk->decompressor->comp_opts(msblk, buff, length) : NULL;
-}
-static inline int squashfs_decompress(struct squashfs_sb_info *msblk,
-        void **buffer, struct buffer_head **bh, int b, int offset, int length,
-        int srclength, int pages)
-{
-        return msblk->decompressor->decompress(msblk, buffer, bh, b, offset,
-                length, srclength, pages);
 }
 #ifdef CONFIG_SQUASHFS_XZ
diff --git a/fs/squashfs/decompressor_multi.c b/fs/squashfs/decompressor_multi.c
new file mode 100644
index 000000000000..d6008a636479
--- /dev/null
+++ b/fs/squashfs/decompressor_multi.c
@@ -0,0 +1,198 @@
+/*
+ *  Copyright (c) 2013
+ *  Minchan Kim <minchan@kernel.org>
+ *
+ *  This work is licensed under the terms of the GNU GPL, version 2. See
+ *  the COPYING file in the top-level directory.
+ */
+#include <linux/types.h>
+#include <linux/mutex.h>
+#include <linux/slab.h>
+#include <linux/buffer_head.h>
+#include <linux/sched.h>
+#include <linux/wait.h>
+#include <linux/cpumask.h>
+#include "squashfs_fs.h"
+#include "squashfs_fs_sb.h"
+#include "decompressor.h"
+#include "squashfs.h"
+/*
+ * This file implements multi-threaded decompression in the
+ * decompressor framework
+ */
+/*
+ * The reason that multiply two is that a CPU can request new I/O
+ * while it is waiting previous request.
+ */
+#define MAX_DECOMPRESSOR        (num_online_cpus() * 2)
+int squashfs_max_decompressors(void)
+{
+        return MAX_DECOMPRESSOR;
+}
+struct squashfs_stream {
+        void                    *comp_opts;
+        struct list_head        strm_list;
+        struct mutex            mutex;
+        int                     avail_decomp;
+        wait_queue_head_t       wait;
+};
+struct decomp_stream {
+        void *stream;
+        struct list_head list;
+};
+static void put_decomp_stream(struct decomp_stream *decomp_strm,
+                                struct squashfs_stream *stream)
+{
+        mutex_lock(&stream->mutex);
+        list_add(&decomp_strm->list, &stream->strm_list);
+        mutex_unlock(&stream->mutex);
+        wake_up(&stream->wait);
+}
+void *squashfs_decompressor_create(struct squashfs_sb_info *msblk,
+                                void *comp_opts)
+{
+        struct squashfs_stream *stream;
+        struct decomp_stream *decomp_strm = NULL;
+        int err = -ENOMEM;
+        stream = kzalloc(sizeof(*stream), GFP_KERNEL);
+        if (!stream)
+                goto out;
+        stream->comp_opts = comp_opts;
+        mutex_init(&stream->mutex);
+        INIT_LIST_HEAD(&stream->strm_list);
+        init_waitqueue_head(&stream->wait);
+        /*
+         * We should have a decompressor at least as default
+         * so if we fail to allocate new decompressor dynamically,
+         * we could always fall back to default decompressor and
+         * file system works.
+         */
+        decomp_strm = kmalloc(sizeof(*decomp_strm), GFP_KERNEL);
+        if (!decomp_strm)
+                goto out;
+        decomp_strm->stream = msblk->decompressor->init(msblk,
+                                                stream->comp_opts);
+        if (IS_ERR(decomp_strm->stream)) {
+                err = PTR_ERR(decomp_strm->stream);
+                goto out;
+        }
+        list_add(&decomp_strm->list, &stream->strm_list);
+        stream->avail_decomp = 1;
+        return stream;
+out:
+        kfree(decomp_strm);
+        kfree(stream);
+        return ERR_PTR(err);
+}
+void squashfs_decompressor_destroy(struct squashfs_sb_info *msblk)
+{
+        struct squashfs_stream *stream = msblk->stream;
+        if (stream) {
+                struct decomp_stream *decomp_strm;
+                while (!list_empty(&stream->strm_list)) {
+                        decomp_strm = list_entry(stream->strm_list.prev,
+                                                struct decomp_stream, list);
+                        list_del(&decomp_strm->list);
+                        msblk->decompressor->free(decomp_strm->stream);
+                        kfree(decomp_strm);
+                        stream->avail_decomp--;
+                }
+                WARN_ON(stream->avail_decomp);
+                kfree(stream->comp_opts);
+                kfree(stream);
+        }
+}
+static struct decomp_stream *get_decomp_stream(struct squashfs_sb_info *msblk,
+                                        struct squashfs_stream *stream)
+{
+        struct decomp_stream *decomp_strm;
+        while (1) {
+                mutex_lock(&stream->mutex);
+                /* There is available decomp_stream */
+                if (!list_empty(&stream->strm_list)) {
+                        decomp_strm = list_entry(stream->strm_list.prev,
+                                struct decomp_stream, list);
+                        list_del(&decomp_strm->list);
+                        mutex_unlock(&stream->mutex);
+                        break;
+                }
+                /*
+                 * If there is no available decomp and already full,
+                 * let's wait for releasing decomp from other users.
+                 */
+                if (stream->avail_decomp >= MAX_DECOMPRESSOR)
+                        goto wait;
+                /* Let's allocate new decomp */
+                decomp_strm = kmalloc(sizeof(*decomp_strm), GFP_KERNEL);
+                if (!decomp_strm)
+                        goto wait;
+                decomp_strm->stream = msblk->decompressor->init(msblk,
+                                                stream->comp_opts);
+                if (IS_ERR(decomp_strm->stream)) {
+                        kfree(decomp_strm);
+                        goto wait;
+                }
+                stream->avail_decomp++;
+                WARN_ON(stream->avail_decomp > MAX_DECOMPRESSOR);
+                mutex_unlock(&stream->mutex);
+                break;
+wait:
+                /*
+                 * If system memory is tough, let's for other's
+                 * releasing instead of hurting VM because it could
+                 * make page cache thrashing.
+                 */
+                mutex_unlock(&stream->mutex);
+                wait_event(stream->wait,
+                        !list_empty(&stream->strm_list));
+        }
+        return decomp_strm;
+}
+int squashfs_decompress(struct squashfs_sb_info *msblk, struct buffer_head **bh,
+        int b, int offset, int length, struct squashfs_page_actor *output)
+{
+        int res;
+        struct squashfs_stream *stream = msblk->stream;
+        struct decomp_stream *decomp_stream = get_decomp_stream(msblk, stream);
+        res = msblk->decompressor->decompress(msblk, decomp_stream->stream,
+                bh, b, offset, length, output);
+        put_decomp_stream(decomp_stream, stream);
+        if (res < 0)
+                ERROR("%s decompression failed, data probably corrupt\n",
+                        msblk->decompressor->name);
+        return res;
+}
diff --git a/fs/squashfs/decompressor_multi_percpu.c b/fs/squashfs/decompressor_multi_percpu.c
new file mode 100644
index 000000000000..23a9c28ad8ea
--- /dev/null
+++ b/fs/squashfs/decompressor_multi_percpu.c
@@ -0,0 +1,97 @@
+/*
+ * Copyright (c) 2013
+ * Phillip Lougher <phillip@squashfs.org.uk>
+ *
+ * This work is licensed under the terms of the GNU GPL, version 2. See
+ * the COPYING file in the top-level directory.
+ */
+#include <linux/types.h>
+#include <linux/slab.h>
+#include <linux/percpu.h>
+#include <linux/buffer_head.h>
+#include "squashfs_fs.h"
+#include "squashfs_fs_sb.h"
+#include "decompressor.h"
+#include "squashfs.h"
+/*
+ * This file implements multi-threaded decompression using percpu
+ * variables, one thread per cpu core.
+ */
+struct squashfs_stream {
+        void            *stream;
+};
+void *squashfs_decompressor_create(struct squashfs_sb_info *msblk,
+                                                void *comp_opts)
+{
+        struct squashfs_stream *stream;
+        struct squashfs_stream __percpu *percpu;
+        int err, cpu;
+        percpu = alloc_percpu(struct squashfs_stream);
+        if (percpu == NULL)
+                return ERR_PTR(-ENOMEM);
+        for_each_possible_cpu(cpu) {
+                stream = per_cpu_ptr(percpu, cpu);
+                stream->stream = msblk->decompressor->init(msblk, comp_opts);
+                if (IS_ERR(stream->stream)) {
+                        err = PTR_ERR(stream->stream);
+                        goto out;
+                }
+        }
+        kfree(comp_opts);
+        return (__force void *) percpu;
+out:
+        for_each_possible_cpu(cpu) {
+                stream = per_cpu_ptr(percpu, cpu);
+                if (!IS_ERR_OR_NULL(stream->stream))
+                        msblk->decompressor->free(stream->stream);
+        }
+        free_percpu(percpu);
+        return ERR_PTR(err);
+}
+void squashfs_decompressor_destroy(struct squashfs_sb_info *msblk)
+{
+        struct squashfs_stream __percpu *percpu =
+                        (struct squashfs_stream __percpu *) msblk->stream;
+        struct squashfs_stream *stream;
+        int cpu;
+        if (msblk->stream) {
+                for_each_possible_cpu(cpu) {
+                        stream = per_cpu_ptr(percpu, cpu);
+                        msblk->decompressor->free(stream->stream);
+                }
+                free_percpu(percpu);
+        }
+}
+int squashfs_decompress(struct squashfs_sb_info *msblk, struct buffer_head **bh,
+        int b, int offset, int length, struct squashfs_page_actor *output)
+{
+        struct squashfs_stream __percpu *percpu =
+                        (struct squashfs_stream __percpu *) msblk->stream;
+        struct squashfs_stream *stream = get_cpu_ptr(percpu);
+        int res = msblk->decompressor->decompress(msblk, stream->stream, bh, b,
+                offset, length, output);
+        put_cpu_ptr(stream);
+        if (res < 0)
+                ERROR("%s decompression failed, data probably corrupt\n",
+                        msblk->decompressor->name);
+        return res;
+}
+int squashfs_max_decompressors(void)
+{
+        return num_possible_cpus();
+}
diff --git a/fs/squashfs/decompressor_single.c b/fs/squashfs/decompressor_single.c
new file mode 100644
index 000000000000..a6c75929a00e
--- /dev/null
+++ b/fs/squashfs/decompressor_single.c
@@ -0,0 +1,85 @@
+/*
+ * Copyright (c) 2013
+ * Phillip Lougher <phillip@squashfs.org.uk>
+ *
+ * This work is licensed under the terms of the GNU GPL, version 2. See
+ * the COPYING file in the top-level directory.
+ */
+#include <linux/types.h>
+#include <linux/mutex.h>
+#include <linux/slab.h>
+#include <linux/buffer_head.h>
+#include "squashfs_fs.h"
+#include "squashfs_fs_sb.h"
+#include "decompressor.h"
+#include "squashfs.h"
+/*
+ * This file implements single-threaded decompression in the
+ * decompressor framework
+ */
+struct squashfs_stream {
+        void            *stream;
+        struct mutex    mutex;
+};
+void *squashfs_decompressor_create(struct squashfs_sb_info *msblk,
+                                                void *comp_opts)
+{
+        struct squashfs_stream *stream;
+        int err = -ENOMEM;
+        stream = kmalloc(sizeof(*stream), GFP_KERNEL);
+        if (stream == NULL)
+                goto out;
+        stream->stream = msblk->decompressor->init(msblk, comp_opts);
+        if (IS_ERR(stream->stream)) {
+                err = PTR_ERR(stream->stream);
+                goto out;
+        }
+        kfree(comp_opts);
+        mutex_init(&stream->mutex);
+        return stream;
+out:
+        kfree(stream);
+        return ERR_PTR(err);
+}
+void squashfs_decompressor_destroy(struct squashfs_sb_info *msblk)
+{
+        struct squashfs_stream *stream = msblk->stream;
+        if (stream) {
+                msblk->decompressor->free(stream->stream);
+                kfree(stream);
+        }
+}
+int squashfs_decompress(struct squashfs_sb_info *msblk, struct buffer_head **bh,
+        int b, int offset, int length, struct squashfs_page_actor *output)
+{
+        int res;
+        struct squashfs_stream *stream = msblk->stream;
+        mutex_lock(&stream->mutex);
+        res = msblk->decompressor->decompress(msblk, stream->stream, bh, b,
+                offset, length, output);
+        mutex_unlock(&stream->mutex);
+        if (res < 0)
+                ERROR("%s decompression failed, data probably corrupt\n",
+                        msblk->decompressor->name);
+        return res;
+}
+int squashfs_max_decompressors(void)
+{
+        return 1;
+}
diff --git a/fs/squashfs/file.c b/fs/squashfs/file.c
index 8ca62c28fe12..e5c9689062ba 100644
--- a/fs/squashfs/file.c
+++ b/fs/squashfs/file.c
@@ -370,77 +370,15 @@ static int read_blocklist(struct inode *inode, int index, u64 *block)
        return le32_to_cpu(size);
 }
+/* Copy data into page cache  */
-static int squashfs_readpage(struct file *file, struct page *page)
+void squashfs_copy_cache(struct page *page, struct squashfs_cache_entry *buffer,
+        int bytes, int offset)
 {
        struct inode *inode = page->mapping->host;
        struct squashfs_sb_info *msblk = inode->i_sb->s_fs_info;
-        int bytes, i, offset = 0, sparse = 0;
-        struct squashfs_cache_entry *buffer = NULL;
        void *pageaddr;
+        int i, mask = (1 << (msblk->block_log - PAGE_CACHE_SHIFT)) - 1;
-        int mask = (1 << (msblk->block_log - PAGE_CACHE_SHIFT)) - 1;
+        int start_index = page->index & ~mask, end_index = start_index | mask;
-        int index = page->index >> (msblk->block_log - PAGE_CACHE_SHIFT);
-        int start_index = page->index & ~mask;
-        int end_index = start_index | mask;
-        int file_end = i_size_read(inode) >> msblk->block_log;
-        TRACE("Entered squashfs_readpage, page index %lx, start block %llx\n",
-                                page->index, squashfs_i(inode)->start);
-        if (page->index >= ((i_size_read(inode) + PAGE_CACHE_SIZE - 1) >>
-                                        PAGE_CACHE_SHIFT))
-                goto out;
-        if (index < file_end || squashfs_i(inode)->fragment_block ==
-                                        SQUASHFS_INVALID_BLK) {
-                /*
-                 * Reading a datablock from disk.  Need to read block list
-                 * to get location and block size.
-                 */
-                u64 block = 0;
-                int bsize = read_blocklist(inode, index, &block);
-                if (bsize < 0)
-                        goto error_out;
-                if (bsize == 0) { /* hole */
-                        bytes = index == file_end ?
-                                (i_size_read(inode) & (msblk->block_size - 1)) :
-                                 msblk->block_size;
-                        sparse = 1;
-                } else {
-                        /*
-                         * Read and decompress datablock.
-                         */
-                        buffer = squashfs_get_datablock(inode->i_sb,
-                                                                block, bsize);
-                        if (buffer->error) {
-                                ERROR("Unable to read page, block %llx, size %x"
-                                        "\n", block, bsize);
-                                squashfs_cache_put(buffer);
-                                goto error_out;
-                        }
-                        bytes = buffer->length;
-                }
-        } else {
-                /*
-                 * Datablock is stored inside a fragment (tail-end packed
-                 * block).
-                 */
-                buffer = squashfs_get_fragment(inode->i_sb,
-                                squashfs_i(inode)->fragment_block,
-                                squashfs_i(inode)->fragment_size);
-                if (buffer->error) {
-                        ERROR("Unable to read page, block %llx, size %x\n",
-                                squashfs_i(inode)->fragment_block,
-                                squashfs_i(inode)->fragment_size);
-                        squashfs_cache_put(buffer);
-                        goto error_out;
-                }
-                bytes = i_size_read(inode) & (msblk->block_size - 1);
-                offset = squashfs_i(inode)->fragment_offset;
-        }
        /*
         * Loop copying datablock into pages.  As the datablock likely covers
@@ -451,7 +389,7 @@ static int squashfs_readpage(struct file *file, struct page *page)
        for (i = start_index; i <= end_index && bytes > 0; i++,
                        bytes -= PAGE_CACHE_SIZE, offset += PAGE_CACHE_SIZE) {
                struct page *push_page;
-                int avail = sparse ? 0 : min_t(int, bytes, PAGE_CACHE_SIZE);
+                int avail = buffer ? min_t(int, bytes, PAGE_CACHE_SIZE) : 0;
                TRACE("bytes %d, i %d, available_bytes %d\n", bytes, i, avail);
@@ -475,11 +413,75 @@ skip_page:
                if (i != page->index)
                        page_cache_release(push_page);
        }
+}
+/* Read datablock stored packed inside a fragment (tail-end packed block) */
+static int squashfs_readpage_fragment(struct page *page)
+{
+        struct inode *inode = page->mapping->host;
+        struct squashfs_sb_info *msblk = inode->i_sb->s_fs_info;
+        struct squashfs_cache_entry *buffer = squashfs_get_fragment(inode->i_sb,
+                squashfs_i(inode)->fragment_block,
+                squashfs_i(inode)->fragment_size);
+        int res = buffer->error;
+        if (res)
+                ERROR("Unable to read page, block %llx, size %x\n",
+                        squashfs_i(inode)->fragment_block,
+                        squashfs_i(inode)->fragment_size);
+        else
+                squashfs_copy_cache(page, buffer, i_size_read(inode) &
+                        (msblk->block_size - 1),
+                        squashfs_i(inode)->fragment_offset);
+        squashfs_cache_put(buffer);
+        return res;
+}
-        if (!sparse)
+static int squashfs_readpage_sparse(struct page *page, int index, int file_end)
-                squashfs_cache_put(buffer);
+{
+        struct inode *inode = page->mapping->host;
+        struct squashfs_sb_info *msblk = inode->i_sb->s_fs_info;
+        int bytes = index == file_end ?
+                        (i_size_read(inode) & (msblk->block_size - 1)) :
+                         msblk->block_size;
+        squashfs_copy_cache(page, NULL, bytes, 0);
        return 0;
+}
+static int squashfs_readpage(struct file *file, struct page *page)
+{
+        struct inode *inode = page->mapping->host;
+        struct squashfs_sb_info *msblk = inode->i_sb->s_fs_info;
+        int index = page->index >> (msblk->block_log - PAGE_CACHE_SHIFT);
+        int file_end = i_size_read(inode) >> msblk->block_log;
+        int res;
+        void *pageaddr;
+        TRACE("Entered squashfs_readpage, page index %lx, start block %llx\n",
+                                page->index, squashfs_i(inode)->start);
+        if (page->index >= ((i_size_read(inode) + PAGE_CACHE_SIZE - 1) >>
+                                        PAGE_CACHE_SHIFT))
+                goto out;
+        if (index < file_end || squashfs_i(inode)->fragment_block ==
+                                        SQUASHFS_INVALID_BLK) {
+                u64 block = 0;
+                int bsize = read_blocklist(inode, index, &block);
+                if (bsize < 0)
+                        goto error_out;
+                if (bsize == 0)
+                        res = squashfs_readpage_sparse(page, index, file_end);
+                else
+                        res = squashfs_readpage_block(page, block, bsize);
+        } else
+                res = squashfs_readpage_fragment(page);
+        if (!res)
+                return 0;
 error_out:
        SetPageError(page);
diff --git a/fs/squashfs/file_cache.c b/fs/squashfs/file_cache.c
new file mode 100644
index 000000000000..f2310d2a2019
--- /dev/null
+++ b/fs/squashfs/file_cache.c
@@ -0,0 +1,38 @@
+/*
+ * Copyright (c) 2013
+ * Phillip Lougher <phillip@squashfs.org.uk>
+ *
+ * This work is licensed under the terms of the GNU GPL, version 2. See
+ * the COPYING file in the top-level directory.
+ */
+#include <linux/fs.h>
+#include <linux/vfs.h>
+#include <linux/kernel.h>
+#include <linux/slab.h>
+#include <linux/string.h>
+#include <linux/pagemap.h>
+#include <linux/mutex.h>
+#include "squashfs_fs.h"
+#include "squashfs_fs_sb.h"
+#include "squashfs_fs_i.h"
+#include "squashfs.h"
+/* Read separately compressed datablock and memcopy into page cache */
+int squashfs_readpage_block(struct page *page, u64 block, int bsize)
+{
+        struct inode *i = page->mapping->host;
+        struct squashfs_cache_entry *buffer = squashfs_get_datablock(i->i_sb,
+                block, bsize);
+        int res = buffer->error;
+        if (res)
+                ERROR("Unable to read page, block %llx, size %x\n", block,
+                        bsize);
+        else
+                squashfs_copy_cache(page, buffer, buffer->length, 0);
+        squashfs_cache_put(buffer);
+        return res;
+}
diff --git a/fs/squashfs/file_direct.c b/fs/squashfs/file_direct.c
new file mode 100644
index 000000000000..62a0de6632e1
--- /dev/null
+++ b/fs/squashfs/file_direct.c
@@ -0,0 +1,176 @@
+/*
+ * Copyright (c) 2013
+ * Phillip Lougher <phillip@squashfs.org.uk>
+ *
+ * This work is licensed under the terms of the GNU GPL, version 2. See
+ * the COPYING file in the top-level directory.
+ */
+#include <linux/fs.h>
+#include <linux/vfs.h>
+#include <linux/kernel.h>
+#include <linux/slab.h>
+#include <linux/string.h>
+#include <linux/pagemap.h>
+#include <linux/mutex.h>
+#include "squashfs_fs.h"
+#include "squashfs_fs_sb.h"
+#include "squashfs_fs_i.h"
+#include "squashfs.h"
+#include "page_actor.h"
+static int squashfs_read_cache(struct page *target_page, u64 block, int bsize,
+        int pages, struct page **page);
+/* Read separately compressed datablock directly into page cache */
+int squashfs_readpage_block(struct page *target_page, u64 block, int bsize)
+{
+        struct inode *inode = target_page->mapping->host;
+        struct squashfs_sb_info *msblk = inode->i_sb->s_fs_info;
+        int file_end = (i_size_read(inode) - 1) >> PAGE_CACHE_SHIFT;
+        int mask = (1 << (msblk->block_log - PAGE_CACHE_SHIFT)) - 1;
+        int start_index = target_page->index & ~mask;
+        int end_index = start_index | mask;
+        int i, n, pages, missing_pages, bytes, res = -ENOMEM;
+        struct page **page;
+        struct squashfs_page_actor *actor;
+        void *pageaddr;
+        if (end_index > file_end)
+                end_index = file_end;
+        pages = end_index - start_index + 1;
+        page = kmalloc(sizeof(void *) * pages, GFP_KERNEL);
+        if (page == NULL)
+                return res;
+        /*
+         * Create a "page actor" which will kmap and kunmap the
+         * page cache pages appropriately within the decompressor
+         */
+        actor = squashfs_page_actor_init_special(page, pages, 0);
+        if (actor == NULL)
+                goto out;
+        /* Try to grab all the pages covered by the Squashfs block */
+        for (missing_pages = 0, i = 0, n = start_index; i < pages; i++, n++) {
+                page[i] = (n == target_page->index) ? target_page :
+                        grab_cache_page_nowait(target_page->mapping, n);
+                if (page[i] == NULL) {
+                        missing_pages++;
+                        continue;
+                }
+                if (PageUptodate(page[i])) {
+                        unlock_page(page[i]);
+                        page_cache_release(page[i]);
+                        page[i] = NULL;
+                        missing_pages++;
+                }
+        }
+        if (missing_pages) {
+                /*
+                 * Couldn't get one or more pages, this page has either
+                 * been VM reclaimed, but others are still in the page cache
+                 * and uptodate, or we're racing with another thread in
+                 * squashfs_readpage also trying to grab them.  Fall back to
+                 * using an intermediate buffer.
+                 */
+                res = squashfs_read_cache(target_page, block, bsize, pages,
+                                                                page);
+                if (res < 0)
+                        goto mark_errored;
+                goto out;
+        }
+        /* Decompress directly into the page cache buffers */
+        res = squashfs_read_data(inode->i_sb, block, bsize, NULL, actor);
+        if (res < 0)
+                goto mark_errored;
+        /* Last page may have trailing bytes not filled */
+        bytes = res % PAGE_CACHE_SIZE;
+        if (bytes) {
+                pageaddr = kmap_atomic(page[pages - 1]);
+                memset(pageaddr + bytes, 0, PAGE_CACHE_SIZE - bytes);
+                kunmap_atomic(pageaddr);
+        }
+        /* Mark pages as uptodate, unlock and release */
+        for (i = 0; i < pages; i++) {
+                flush_dcache_page(page[i]);
+                SetPageUptodate(page[i]);
+                unlock_page(page[i]);
+                if (page[i] != target_page)
+                        page_cache_release(page[i]);
+        }
+        kfree(actor);
+        kfree(page);
+        return 0;
+mark_errored:
+        /* Decompression failed, mark pages as errored.  Target_page is
+         * dealt with by the caller
+         */
+        for (i = 0; i < pages; i++) {
+                if (page[i] == NULL || page[i] == target_page)
+                        continue;
+                flush_dcache_page(page[i]);
+                SetPageError(page[i]);
+                unlock_page(page[i]);
+                page_cache_release(page[i]);
+        }
+out:
+        kfree(actor);
+        kfree(page);
+        return res;
+}
+static int squashfs_read_cache(struct page *target_page, u64 block, int bsize,
+        int pages, struct page **page)
+{
+        struct inode *i = target_page->mapping->host;
+        struct squashfs_cache_entry *buffer = squashfs_get_datablock(i->i_sb,
+                                                 block, bsize);
+        int bytes = buffer->length, res = buffer->error, n, offset = 0;
+        void *pageaddr;
+        if (res) {
+                ERROR("Unable to read page, block %llx, size %x\n", block,
+                        bsize);
+                goto out;
+        }
+        for (n = 0; n < pages && bytes > 0; n++,
+                        bytes -= PAGE_CACHE_SIZE, offset += PAGE_CACHE_SIZE) {
+                int avail = min_t(int, bytes, PAGE_CACHE_SIZE);
+                if (page[n] == NULL)
+                        continue;
+                pageaddr = kmap_atomic(page[n]);
+                squashfs_copy_data(pageaddr, buffer, offset, avail);
+                memset(pageaddr + avail, 0, PAGE_CACHE_SIZE - avail);
+                kunmap_atomic(pageaddr);
+                flush_dcache_page(page[n]);
+                SetPageUptodate(page[n]);
+                unlock_page(page[n]);
+                if (page[n] != target_page)
+                        page_cache_release(page[n]);
+        }
+out:
+        squashfs_cache_put(buffer);
+        return res;
+}
diff --git a/fs/squashfs/lzo_wrapper.c b/fs/squashfs/lzo_wrapper.c
index 00f4dfc5f088..244b9fbfff7b 100644
--- a/fs/squashfs/lzo_wrapper.c
+++ b/fs/squashfs/lzo_wrapper.c
@@ -31,13 +31,14 @@
 #include "squashfs_fs_sb.h"
 #include "squashfs.h"
 #include "decompressor.h"
+#include "page_actor.h"
 struct squashfs_lzo {
        void    *input;
        void    *output;
 };
-static void *lzo_init(struct squashfs_sb_info *msblk, void *buff, int len)
+static void *lzo_init(struct squashfs_sb_info *msblk, void *buff)
 {
        int block_size = max_t(int, msblk->block_size, SQUASHFS_METADATA_SIZE);
@@ -74,22 +75,16 @@ static void lzo_free(void *strm)
 }
-static int lzo_uncompress(struct squashfs_sb_info *msblk, void **buffer,
+static int lzo_uncompress(struct squashfs_sb_info *msblk, void *strm,
-        struct buffer_head **bh, int b, int offset, int length, int srclength,
+        struct buffer_head **bh, int b, int offset, int length,
-        int pages)
+        struct squashfs_page_actor *output)
 {
-        struct squashfs_lzo *stream = msblk->stream;
+        struct squashfs_lzo *stream = strm;
-        void *buff = stream->input;
+        void *buff = stream->input, *data;
        int avail, i, bytes = length, res;
-        size_t out_len = srclength;
+        size_t out_len = output->length;
-        mutex_lock(&msblk->read_data_mutex);
        for (i = 0; i < b; i++) {
-                wait_on_buffer(bh[i]);
-                if (!buffer_uptodate(bh[i]))
-                        goto block_release;
                avail = min(bytes, msblk->devblksize - offset);
                memcpy(buff, bh[i]->b_data + offset, avail);
                buff += avail;
@@ -104,24 +99,24 @@ static int lzo_uncompress(struct squashfs_sb_info *msblk, void **buffer,
                goto failed;
        res = bytes = (int)out_len;
-        for (i = 0, buff = stream->output; bytes && i < pages; i++) {
+        data = squashfs_first_page(output);
-                avail = min_t(int, bytes, PAGE_CACHE_SIZE);
+        buff = stream->output;
-                memcpy(buffer[i], buff, avail);
+        while (data) {
-                buff += avail;
+                if (bytes <= PAGE_CACHE_SIZE) {
-                bytes -= avail;
+                        memcpy(data, buff, bytes);
+                        break;
+                } else {
+                        memcpy(data, buff, PAGE_CACHE_SIZE);
+                        buff += PAGE_CACHE_SIZE;
+                        bytes -= PAGE_CACHE_SIZE;
+                        data = squashfs_next_page(output);
+                }
        }
+        squashfs_finish_page(output);
-        mutex_unlock(&msblk->read_data_mutex);
        return res;
-block_release:
-        for (; i < b; i++)
-                put_bh(bh[i]);
 failed:
-        mutex_unlock(&msblk->read_data_mutex);
-        ERROR("lzo decompression failed, data probably corrupt\n");
        return -EIO;
 }
diff --git a/fs/squashfs/page_actor.c b/fs/squashfs/page_actor.c
new file mode 100644
index 000000000000..5a1c11f56441
--- /dev/null
+++ b/fs/squashfs/page_actor.c
@@ -0,0 +1,100 @@
+/*
+ * Copyright (c) 2013
+ * Phillip Lougher <phillip@squashfs.org.uk>
+ *
+ * This work is licensed under the terms of the GNU GPL, version 2. See
+ * the COPYING file in the top-level directory.
+ */
+#include <linux/kernel.h>
+#include <linux/slab.h>
+#include <linux/pagemap.h>
+#include "page_actor.h"
+/*
+ * This file contains implementations of page_actor for decompressing into
+ * an intermediate buffer, and for decompressing directly into the
+ * page cache.
+ *
+ * Calling code should avoid sleeping between calls to squashfs_first_page()
+ * and squashfs_finish_page().
+ */
+/* Implementation of page_actor for decompressing into intermediate buffer */
+static void *cache_first_page(struct squashfs_page_actor *actor)
+{
+        actor->next_page = 1;
+        return actor->buffer[0];
+}
+static void *cache_next_page(struct squashfs_page_actor *actor)
+{
+        if (actor->next_page == actor->pages)
+                return NULL;
+        return actor->buffer[actor->next_page++];
+}
+static void cache_finish_page(struct squashfs_page_actor *actor)
+{
+        /* empty */
+}
+struct squashfs_page_actor *squashfs_page_actor_init(void **buffer,
+        int pages, int length)
+{
+        struct squashfs_page_actor *actor = kmalloc(sizeof(*actor), GFP_KERNEL);
+        if (actor == NULL)
+                return NULL;
+        actor->length = length ? : pages * PAGE_CACHE_SIZE;
+        actor->buffer = buffer;
+        actor->pages = pages;
+        actor->next_page = 0;
+        actor->squashfs_first_page = cache_first_page;
+        actor->squashfs_next_page = cache_next_page;
+        actor->squashfs_finish_page = cache_finish_page;
+        return actor;
+}
+/* Implementation of page_actor for decompressing directly into page cache. */
+static void *direct_first_page(struct squashfs_page_actor *actor)
+{
+        actor->next_page = 1;
+        return actor->pageaddr = kmap_atomic(actor->page[0]);
+}
+static void *direct_next_page(struct squashfs_page_actor *actor)
+{
+        if (actor->pageaddr)
+                kunmap_atomic(actor->pageaddr);
+        return actor->pageaddr = actor->next_page == actor->pages ? NULL :
+                kmap_atomic(actor->page[actor->next_page++]);
+}
+static void direct_finish_page(struct squashfs_page_actor *actor)
+{
+        if (actor->pageaddr)
+                kunmap_atomic(actor->pageaddr);
+}
+struct squashfs_page_actor *squashfs_page_actor_init_special(struct page **page,
+        int pages, int length)
+{
+        struct squashfs_page_actor *actor = kmalloc(sizeof(*actor), GFP_KERNEL);
+        if (actor == NULL)
+                return NULL;
+        actor->length = length ? : pages * PAGE_CACHE_SIZE;
+        actor->page = page;
+        actor->pages = pages;
+        actor->next_page = 0;
+        actor->pageaddr = NULL;
+        actor->squashfs_first_page = direct_first_page;
+        actor->squashfs_next_page = direct_next_page;
+        actor->squashfs_finish_page = direct_finish_page;
+        return actor;
+}
diff --git a/fs/squashfs/page_actor.h b/fs/squashfs/page_actor.h
new file mode 100644
index 000000000000..26dd82008b82
--- /dev/null
+++ b/fs/squashfs/page_actor.h
@@ -0,0 +1,81 @@
+#ifndef PAGE_ACTOR_H
+#define PAGE_ACTOR_H
+/*
+ * Copyright (c) 2013
+ * Phillip Lougher <phillip@squashfs.org.uk>
+ *
+ * This work is licensed under the terms of the GNU GPL, version 2. See
+ * the COPYING file in the top-level directory.
+ */
+#ifndef CONFIG_SQUASHFS_FILE_DIRECT
+struct squashfs_page_actor {
+        void    **page;
+        int     pages;
+        int     length;
+        int     next_page;
+};
+static inline struct squashfs_page_actor *squashfs_page_actor_init(void **page,
+        int pages, int length)
+{
+        struct squashfs_page_actor *actor = kmalloc(sizeof(*actor), GFP_KERNEL);
+        if (actor == NULL)
+                return NULL;
+        actor->length = length ? : pages * PAGE_CACHE_SIZE;
+        actor->page = page;
+        actor->pages = pages;
+        actor->next_page = 0;
+        return actor;
+}
+static inline void *squashfs_first_page(struct squashfs_page_actor *actor)
+{
+        actor->next_page = 1;
+        return actor->page[0];
+}
+static inline void *squashfs_next_page(struct squashfs_page_actor *actor)
+{
+        return actor->next_page == actor->pages ? NULL :
+                actor->page[actor->next_page++];
+}
+static inline void squashfs_finish_page(struct squashfs_page_actor *actor)
+{
+        /* empty */
+}
+#else
+struct squashfs_page_actor {
+        union {
+                void            **buffer;
+                struct page     **page;
+        };
+        void    *pageaddr;
+        void    *(*squashfs_first_page)(struct squashfs_page_actor *);
+        void    *(*squashfs_next_page)(struct squashfs_page_actor *);
+        void    (*squashfs_finish_page)(struct squashfs_page_actor *);
+        int     pages;
+        int     length;
+        int     next_page;
+};
+extern struct squashfs_page_actor *squashfs_page_actor_init(void **, int, int);
+extern struct squashfs_page_actor *squashfs_page_actor_init_special(struct page
+                                                         **, int, int);
+static inline void *squashfs_first_page(struct squashfs_page_actor *actor)
+{
+        return actor->squashfs_first_page(actor);
+}
+static inline void *squashfs_next_page(struct squashfs_page_actor *actor)
+{
+        return actor->squashfs_next_page(actor);
+}
+static inline void squashfs_finish_page(struct squashfs_page_actor *actor)
+{
+        actor->squashfs_finish_page(actor);
+}
+#endif
+#endif
diff --git a/fs/squashfs/squashfs.h b/fs/squashfs/squashfs.h
index d1266516ed08..9e1bb79f7e6f 100644
--- a/fs/squashfs/squashfs.h
+++ b/fs/squashfs/squashfs.h
@@ -28,8 +28,8 @@
 #define WARNING(s, args...)     pr_warning("SQUASHFS: "s, ## args)
 /* block.c */
-extern int squashfs_read_data(struct super_block *, void **, u64, int, u64 *,
+extern int squashfs_read_data(struct super_block *, u64, int, u64 *,
-                                int, int);
+                                struct squashfs_page_actor *);
 /* cache.c */
 extern struct squashfs_cache *squashfs_cache_init(char *, int, int);
@@ -48,7 +48,14 @@ extern void *squashfs_read_table(struct super_block *, u64, int);
 /* decompressor.c */
 extern const struct squashfs_decompressor *squashfs_lookup_decompressor(int);
-extern void *squashfs_decompressor_init(struct super_block *, unsigned short);
+extern void *squashfs_decompressor_setup(struct super_block *, unsigned short);
+/* decompressor_xxx.c */
+extern void *squashfs_decompressor_create(struct squashfs_sb_info *, void *);
+extern void squashfs_decompressor_destroy(struct squashfs_sb_info *);
+extern int squashfs_decompress(struct squashfs_sb_info *, struct buffer_head **,
+        int, int, int, struct squashfs_page_actor *);
+extern int squashfs_max_decompressors(void);
 /* export.c */
 extern __le64 *squashfs_read_inode_lookup_table(struct super_block *, u64, u64,
@@ -59,6 +66,13 @@ extern int squashfs_frag_lookup(struct super_block *, unsigned int, u64 *);
 extern __le64 *squashfs_read_fragment_index_table(struct super_block *,
                                u64, u64, unsigned int);
+/* file.c */
+void squashfs_copy_cache(struct page *, struct squashfs_cache_entry *, int,
+                                int);
+/* file_xxx.c */
+extern int squashfs_readpage_block(struct page *, u64, int);
 /* id.c */
 extern int squashfs_get_id(struct super_block *, unsigned int, unsigned int *);
 extern __le64 *squashfs_read_id_index_table(struct super_block *, u64, u64,
diff --git a/fs/squashfs/squashfs_fs_sb.h b/fs/squashfs/squashfs_fs_sb.h
index 52934a22f296..1da565cb50c3 100644
--- a/fs/squashfs/squashfs_fs_sb.h
+++ b/fs/squashfs/squashfs_fs_sb.h
@@ -50,6 +50,7 @@ struct squashfs_cache_entry {
        wait_queue_head_t       wait_queue;
        struct squashfs_cache   *cache;
        void                    **data;
+        struct squashfs_page_actor      *actor;
 };
 struct squashfs_sb_info {
@@ -63,10 +64,9 @@ struct squashfs_sb_info {
        __le64                                  *id_table;
        __le64                                  *fragment_index;
        __le64                                  *xattr_id_table;
-        struct mutex                            read_data_mutex;
        struct mutex                            meta_index_mutex;
        struct meta_index                       *meta_index;
-        void                                    *stream;
+        struct squashfs_stream                  *stream;
        __le64                                  *inode_lookup_table;
        u64                                     inode_table;
        u64                                     directory_table;
diff --git a/fs/squashfs/super.c b/fs/squashfs/super.c
index 60553a9053ca..202df6312d4e 100644
--- a/fs/squashfs/super.c
+++ b/fs/squashfs/super.c
@@ -98,7 +98,6 @@ static int squashfs_fill_super(struct super_block *sb, void *data, int silent)
        msblk->devblksize = sb_min_blocksize(sb, SQUASHFS_DEVBLK_SIZE);
        msblk->devblksize_log2 = ffz(~msblk->devblksize);
-        mutex_init(&msblk->read_data_mutex);
        mutex_init(&msblk->meta_index_mutex);
        /*
@@ -206,13 +205,14 @@ static int squashfs_fill_super(struct super_block *sb, void *data, int silent)
                goto failed_mount;
        /* Allocate read_page block */
-        msblk->read_page = squashfs_cache_init("data", 1, msblk->block_size);
+        msblk->read_page = squashfs_cache_init("data",
+                squashfs_max_decompressors(), msblk->block_size);
        if (msblk->read_page == NULL) {
                ERROR("Failed to allocate read_page block\n");
                goto failed_mount;
        }
-        msblk->stream = squashfs_decompressor_init(sb, flags);
+        msblk->stream = squashfs_decompressor_setup(sb, flags);
        if (IS_ERR(msblk->stream)) {
                err = PTR_ERR(msblk->stream);
                msblk->stream = NULL;
@@ -336,7 +336,7 @@ failed_mount:
        squashfs_cache_delete(msblk->block_cache);
        squashfs_cache_delete(msblk->fragment_cache);
        squashfs_cache_delete(msblk->read_page);
-        squashfs_decompressor_free(msblk, msblk->stream);
+        squashfs_decompressor_destroy(msblk);
        kfree(msblk->inode_lookup_table);
        kfree(msblk->fragment_index);
        kfree(msblk->id_table);
@@ -383,7 +383,7 @@ static void squashfs_put_super(struct super_block *sb)
                squashfs_cache_delete(sbi->block_cache);
                squashfs_cache_delete(sbi->fragment_cache);
                squashfs_cache_delete(sbi->read_page);
-                squashfs_decompressor_free(sbi, sbi->stream);
+                squashfs_decompressor_destroy(sbi);
                kfree(sbi->id_table);
                kfree(sbi->fragment_index);
                kfree(sbi->meta_index);
diff --git a/fs/squashfs/xz_wrapper.c b/fs/squashfs/xz_wrapper.c
index 1760b7d108f6..c609624e4b8a 100644
--- a/fs/squashfs/xz_wrapper.c
+++ b/fs/squashfs/xz_wrapper.c
@@ -32,44 +32,70 @@
 #include "squashfs_fs_sb.h"
 #include "squashfs.h"
 #include "decompressor.h"
+#include "page_actor.h"
 struct squashfs_xz {
        struct xz_dec *state;
        struct xz_buf buf;
 };
-struct comp_opts {
+struct disk_comp_opts {
        __le32 dictionary_size;
        __le32 flags;
 };
-static void *squashfs_xz_init(struct squashfs_sb_info *msblk, void *buff,
+struct comp_opts {
-        int len)
+        int dict_size;
+};
+static void *squashfs_xz_comp_opts(struct squashfs_sb_info *msblk,
+        void *buff, int len)
 {
-        struct comp_opts *comp_opts = buff;
+        struct disk_comp_opts *comp_opts = buff;
-        struct squashfs_xz *stream;
+        struct comp_opts *opts;
-        int dict_size = msblk->block_size;
+        int err = 0, n;
-        int err, n;
+        opts = kmalloc(sizeof(*opts), GFP_KERNEL);
+        if (opts == NULL) {
+                err = -ENOMEM;
+                goto out2;
+        }
        if (comp_opts) {
                /* check compressor options are the expected length */
                if (len < sizeof(*comp_opts)) {
                        err = -EIO;
-                        goto failed;
+                        goto out;
                }
-                dict_size = le32_to_cpu(comp_opts->dictionary_size);
+                opts->dict_size = le32_to_cpu(comp_opts->dictionary_size);
                /* the dictionary size should be 2^n or 2^n+2^(n+1) */
-                n = ffs(dict_size) - 1;
+                n = ffs(opts->dict_size) - 1;
-                if (dict_size != (1 << n) && dict_size != (1 << n) +
+                if (opts->dict_size != (1 << n) && opts->dict_size != (1 << n) +
                                                (1 << (n + 1))) {
                        err = -EIO;
-                        goto failed;
+                        goto out;
                }
-        }
+        } else
+                /* use defaults */
+                opts->dict_size = max_t(int, msblk->block_size,
+                                                        SQUASHFS_METADATA_SIZE);
+        return opts;
+out:
+        kfree(opts);
+out2:
+        return ERR_PTR(err);
+}
-        dict_size = max_t(int, dict_size, SQUASHFS_METADATA_SIZE);
+static void *squashfs_xz_init(struct squashfs_sb_info *msblk, void *buff)
+{
+        struct comp_opts *comp_opts = buff;
+        struct squashfs_xz *stream;
+        int err;
        stream = kmalloc(sizeof(*stream), GFP_KERNEL);
        if (stream == NULL) {
@@ -77,7 +103,7 @@ static void *squashfs_xz_init(struct squashfs_sb_info *msblk, void *buff,
                goto failed;
        }
-        stream->state = xz_dec_init(XZ_PREALLOC, dict_size);
+        stream->state = xz_dec_init(XZ_PREALLOC, comp_opts->dict_size);
        if (stream->state == NULL) {
                kfree(stream);
                err = -ENOMEM;
@@ -103,42 +129,37 @@ static void squashfs_xz_free(void *strm)
 }
-static int squashfs_xz_uncompress(struct squashfs_sb_info *msblk, void **buffer,
+static int squashfs_xz_uncompress(struct squashfs_sb_info *msblk, void *strm,
-        struct buffer_head **bh, int b, int offset, int length, int srclength,
+        struct buffer_head **bh, int b, int offset, int length,
-        int pages)
+        struct squashfs_page_actor *output)
 {
        enum xz_ret xz_err;
-        int avail, total = 0, k = 0, page = 0;
+        int avail, total = 0, k = 0;
-        struct squashfs_xz *stream = msblk->stream;
+        struct squashfs_xz *stream = strm;
-        mutex_lock(&msblk->read_data_mutex);
        xz_dec_reset(stream->state);
        stream->buf.in_pos = 0;
        stream->buf.in_size = 0;
        stream->buf.out_pos = 0;
        stream->buf.out_size = PAGE_CACHE_SIZE;
-        stream->buf.out = buffer[page++];
+        stream->buf.out = squashfs_first_page(output);
        do {
                if (stream->buf.in_pos == stream->buf.in_size && k < b) {
                        avail = min(length, msblk->devblksize - offset);
                        length -= avail;
-                        wait_on_buffer(bh[k]);
-                        if (!buffer_uptodate(bh[k]))
-                                goto release_mutex;
                        stream->buf.in = bh[k]->b_data + offset;
                        stream->buf.in_size = avail;
                        stream->buf.in_pos = 0;
                        offset = 0;
                }
-                if (stream->buf.out_pos == stream->buf.out_size
+                if (stream->buf.out_pos == stream->buf.out_size) {
-                                                        && page < pages) {
+                        stream->buf.out = squashfs_next_page(output);
-                        stream->buf.out = buffer[page++];
+                        if (stream->buf.out != NULL) {
-                        stream->buf.out_pos = 0;
+                                stream->buf.out_pos = 0;
-                        total += PAGE_CACHE_SIZE;
+                                total += PAGE_CACHE_SIZE;
+                        }
                }
                xz_err = xz_dec_run(stream->state, &stream->buf);
@@ -147,23 +168,14 @@ static int squashfs_xz_uncompress(struct squashfs_sb_info *msblk, void **buffer,
                        put_bh(bh[k++]);
        } while (xz_err == XZ_OK);
-        if (xz_err != XZ_STREAM_END) {
+        squashfs_finish_page(output);
-                ERROR("xz_dec_run error, data probably corrupt\n");
-                goto release_mutex;
-        }
-        if (k < b) {
-                ERROR("xz_uncompress error, input remaining\n");
-                goto release_mutex;
-        }
-        total += stream->buf.out_pos;
+        if (xz_err != XZ_STREAM_END || k < b)
-        mutex_unlock(&msblk->read_data_mutex);
+                goto out;
-        return total;
-release_mutex:
+        return total + stream->buf.out_pos;
-        mutex_unlock(&msblk->read_data_mutex);
+out:
        for (; k < b; k++)
                put_bh(bh[k]);
@@ -172,6 +184,7 @@ release_mutex:
 const struct squashfs_decompressor squashfs_xz_comp_ops = {
        .init = squashfs_xz_init,
+        .comp_opts = squashfs_xz_comp_opts,
        .free = squashfs_xz_free,
        .decompress = squashfs_xz_uncompress,
        .id = XZ_COMPRESSION,
diff --git a/fs/squashfs/zlib_wrapper.c b/fs/squashfs/zlib_wrapper.c
index 55d918fd2d86..8727caba6882 100644
--- a/fs/squashfs/zlib_wrapper.c
+++ b/fs/squashfs/zlib_wrapper.c
@@ -32,8 +32,9 @@
 #include "squashfs_fs_sb.h"
 #include "squashfs.h"
 #include "decompressor.h"
+#include "page_actor.h"
-static void *zlib_init(struct squashfs_sb_info *dummy, void *buff, int len)
+static void *zlib_init(struct squashfs_sb_info *dummy, void *buff)
 {
        z_stream *stream = kmalloc(sizeof(z_stream), GFP_KERNEL);
        if (stream == NULL)
@@ -61,44 +62,37 @@ static void zlib_free(void *strm)
 }
-static int zlib_uncompress(struct squashfs_sb_info *msblk, void **buffer,
+static int zlib_uncompress(struct squashfs_sb_info *msblk, void *strm,
-        struct buffer_head **bh, int b, int offset, int length, int srclength,
+        struct buffer_head **bh, int b, int offset, int length,
-        int pages)
+        struct squashfs_page_actor *output)
 {
-        int zlib_err, zlib_init = 0;
+        int zlib_err, zlib_init = 0, k = 0;
-        int k = 0, page = 0;
+        z_stream *stream = strm;
-        z_stream *stream = msblk->stream;
-        mutex_lock(&msblk->read_data_mutex);
-        stream->avail_out = 0;
+        stream->avail_out = PAGE_CACHE_SIZE;
+        stream->next_out = squashfs_first_page(output);
        stream->avail_in = 0;
        do {
                if (stream->avail_in == 0 && k < b) {
                        int avail = min(length, msblk->devblksize - offset);
                        length -= avail;
-                        wait_on_buffer(bh[k]);
-                        if (!buffer_uptodate(bh[k]))
-                                goto release_mutex;
                        stream->next_in = bh[k]->b_data + offset;
                        stream->avail_in = avail;
                        offset = 0;
                }
-                if (stream->avail_out == 0 && page < pages) {
+                if (stream->avail_out == 0) {
-                        stream->next_out = buffer[page++];
+                        stream->next_out = squashfs_next_page(output);
-                        stream->avail_out = PAGE_CACHE_SIZE;
+                        if (stream->next_out != NULL)
+                                stream->avail_out = PAGE_CACHE_SIZE;
                }
                if (!zlib_init) {
                        zlib_err = zlib_inflateInit(stream);
                        if (zlib_err != Z_OK) {
-                                ERROR("zlib_inflateInit returned unexpected "
+                                squashfs_finish_page(output);
-                                        "result 0x%x, srclength %d\n",
+                                goto out;
-                                        zlib_err, srclength);
-                                goto release_mutex;
                        }
                        zlib_init = 1;
                }
@@ -109,29 +103,21 @@ static int zlib_uncompress(struct squashfs_sb_info *msblk, void **buffer,
                        put_bh(bh[k++]);
        } while (zlib_err == Z_OK);
-        if (zlib_err != Z_STREAM_END) {
+        squashfs_finish_page(output);
-                ERROR("zlib_inflate error, data probably corrupt\n");
-                goto release_mutex;
-        }
-        zlib_err = zlib_inflateEnd(stream);
+        if (zlib_err != Z_STREAM_END)
-        if (zlib_err != Z_OK) {
+                goto out;
-                ERROR("zlib_inflate error, data probably corrupt\n");
-                goto release_mutex;
-        }
-        if (k < b) {
+        zlib_err = zlib_inflateEnd(stream);
-                ERROR("zlib_uncompress error, data remaining\n");
+        if (zlib_err != Z_OK)
-                goto release_mutex;
+                goto out;
-        }
-        length = stream->total_out;
+        if (k < b)
-        mutex_unlock(&msblk->read_data_mutex);
+                goto out;
-        return length;
-release_mutex:
+        return stream->total_out;
-        mutex_unlock(&msblk->read_data_mutex);
+out:
        for (; k < b; k++)
                put_bh(bh[k]);
diff --git a/fs/sysfs/file.c b/fs/sysfs/file.c
index 79b5da2acbe1..b94f93685093 100644
--- a/fs/sysfs/file.c
+++ b/fs/sysfs/file.c
@@ -609,7 +609,7 @@ static int sysfs_open_file(struct inode *inode, struct file *file)
        struct sysfs_dirent *attr_sd = file->f_path.dentry->d_fsdata;
        struct kobject *kobj = attr_sd->s_parent->s_dir.kobj;
        struct sysfs_open_file *of;
-        bool has_read, has_write;
+        bool has_read, has_write, has_mmap;
        int error = -EACCES;
        /* need attr_sd for attr and ops, its parent for kobj */
@@ -621,6 +621,7 @@ static int sysfs_open_file(struct inode *inode, struct file *file)
                has_read = battr->read || battr->mmap;
                has_write = battr->write || battr->mmap;
+                has_mmap = battr->mmap;
        } else {
                const struct sysfs_ops *ops = sysfs_file_ops(attr_sd);
@@ -632,6 +633,7 @@ static int sysfs_open_file(struct inode *inode, struct file *file)
                has_read = ops->show;
                has_write = ops->store;
+                has_mmap = false;
        }
        /* check perms and supported operations */
@@ -649,7 +651,23 @@ static int sysfs_open_file(struct inode *inode, struct file *file)
        if (!of)
                goto err_out;
-        mutex_init(&of->mutex);
+        /*
+         * The following is done to give a different lockdep key to
+         * @of->mutex for files which implement mmap.  This is a rather
+         * crude way to avoid false positive lockdep warning around
+         * mm->mmap_sem - mmap nests @of->mutex under mm->mmap_sem and
+         * reading /sys/block/sda/trace/act_mask grabs sr_mutex, under
+         * which mm->mmap_sem nests, while holding @of->mutex.  As each
+         * open file has a separate mutex, it's okay as long as those don't
+         * happen on the same file.  At this point, we can't easily give
+         * each file a separate locking class.  Let's differentiate on
+         * whether the file has mmap or not for now.
+         */
+        if (has_mmap)
+                mutex_init(&of->mutex);
+        else
+                mutex_init(&of->mutex);
        of->sd = attr_sd;
        of->file = file;
diff --git a/fs/xfs/xfs_bmap.c b/fs/xfs/xfs_bmap.c
index 1c02da8bb7df..3ef11b22e750 100644
--- a/fs/xfs/xfs_bmap.c
+++ b/fs/xfs/xfs_bmap.c
@@ -1137,6 +1137,7 @@ xfs_bmap_add_attrfork(
        int                     committed;      /* xaction was committed */
        int                     logflags;       /* logging flags */
        int                     error;          /* error return value */
+        int                     cancel_flags = 0;
        ASSERT(XFS_IFORK_Q(ip) == 0);
@@ -1147,19 +1148,20 @@ xfs_bmap_add_attrfork(
        if (rsvd)
                tp->t_flags |= XFS_TRANS_RESERVE;
        error = xfs_trans_reserve(tp, &M_RES(mp)->tr_addafork, blks, 0);
-        if (error)
+        if (error) {
-                goto error0;
+                xfs_trans_cancel(tp, 0);
+                return error;
+        }
+        cancel_flags = XFS_TRANS_RELEASE_LOG_RES;
        xfs_ilock(ip, XFS_ILOCK_EXCL);
        error = xfs_trans_reserve_quota_nblks(tp, ip, blks, 0, rsvd ?
                        XFS_QMOPT_RES_REGBLKS | XFS_QMOPT_FORCE_RES :
                        XFS_QMOPT_RES_REGBLKS);
-        if (error) {
+        if (error)
-                xfs_iunlock(ip, XFS_ILOCK_EXCL);
+                goto trans_cancel;
-                xfs_trans_cancel(tp, XFS_TRANS_RELEASE_LOG_RES);
+        cancel_flags |= XFS_TRANS_ABORT;
-                return error;
-        }
        if (XFS_IFORK_Q(ip))
-                goto error1;
+                goto trans_cancel;
        if (ip->i_d.di_aformat != XFS_DINODE_FMT_EXTENTS) {
                /*
                 * For inodes coming from pre-6.2 filesystems.
@@ -1169,7 +1171,7 @@ xfs_bmap_add_attrfork(
        }
        ASSERT(ip->i_d.di_anextents == 0);
-        xfs_trans_ijoin(tp, ip, XFS_ILOCK_EXCL);
+        xfs_trans_ijoin(tp, ip, 0);
        xfs_trans_log_inode(tp, ip, XFS_ILOG_CORE);
        switch (ip->i_d.di_format) {
@@ -1191,7 +1193,7 @@ xfs_bmap_add_attrfork(
        default:
                ASSERT(0);
                error = XFS_ERROR(EINVAL);
-                goto error1;
+                goto trans_cancel;
        }
        ASSERT(ip->i_afp == NULL);
@@ -1219,7 +1221,7 @@ xfs_bmap_add_attrfork(
        if (logflags)
                xfs_trans_log_inode(tp, ip, logflags);
        if (error)
-                goto error2;
+                goto bmap_cancel;
        if (!xfs_sb_version_hasattr(&mp->m_sb) ||
           (!xfs_sb_version_hasattr2(&mp->m_sb) && version == 2)) {
                __int64_t sbfields = 0;
@@ -1242,14 +1244,16 @@ xfs_bmap_add_attrfork(
        error = xfs_bmap_finish(&tp, &flist, &committed);
        if (error)
-                goto error2;
+                goto bmap_cancel;
-        return xfs_trans_commit(tp, XFS_TRANS_RELEASE_LOG_RES);
+        error = xfs_trans_commit(tp, XFS_TRANS_RELEASE_LOG_RES);
-error2:
+        xfs_iunlock(ip, XFS_ILOCK_EXCL);
+        return error;
+bmap_cancel:
        xfs_bmap_cancel(&flist);
-error1:
+trans_cancel:
+        xfs_trans_cancel(tp, cancel_flags);
        xfs_iunlock(ip, XFS_ILOCK_EXCL);
-error0:
-        xfs_trans_cancel(tp, XFS_TRANS_RELEASE_LOG_RES|XFS_TRANS_ABORT);
        return error;
 }
diff --git a/fs/xfs/xfs_discard.c b/fs/xfs/xfs_discard.c
index 8367d6dc18c9..4f11ef011139 100644
--- a/fs/xfs/xfs_discard.c
+++ b/fs/xfs/xfs_discard.c
@@ -157,7 +157,7 @@ xfs_ioc_trim(
        struct xfs_mount                *mp,
        struct fstrim_range __user      *urange)
 {
-        struct request_queue    *q = mp->m_ddev_targp->bt_bdev->bd_disk->queue;
+        struct request_queue    *q = bdev_get_queue(mp->m_ddev_targp->bt_bdev);
        unsigned int            granularity = q->limits.discard_granularity;
        struct fstrim_range     range;
        xfs_daddr_t             start, end, minlen;
@@ -180,7 +180,8 @@ xfs_ioc_trim(
         * matter as trimming blocks is an advisory interface.
         */
        if (range.start >= XFS_FSB_TO_B(mp, mp->m_sb.sb_dblocks) ||
-            range.minlen > XFS_FSB_TO_B(mp, XFS_ALLOC_AG_MAX_USABLE(mp)))
+            range.minlen > XFS_FSB_TO_B(mp, XFS_ALLOC_AG_MAX_USABLE(mp)) ||
+            range.len < mp->m_sb.sb_blocksize)
                return -XFS_ERROR(EINVAL);
        start = BTOBB(range.start);
diff --git a/fs/xfs/xfs_fsops.c b/fs/xfs/xfs_fsops.c
index a6e54b3319bd..02fb943cbf22 100644
--- a/fs/xfs/xfs_fsops.c
+++ b/fs/xfs/xfs_fsops.c
@@ -220,6 +220,8 @@ xfs_growfs_data_private(
         */
        nfree = 0;
        for (agno = nagcount - 1; agno >= oagcount; agno--, new -= agsize) {
+                __be32  *agfl_bno;
                /*
                 * AG freespace header block
                 */
@@ -279,8 +281,10 @@ xfs_growfs_data_private(
                        agfl->agfl_seqno = cpu_to_be32(agno);
                        uuid_copy(&agfl->agfl_uuid, &mp->m_sb.sb_uuid);
                }
+                agfl_bno = XFS_BUF_TO_AGFL_BNO(mp, bp);
                for (bucket = 0; bucket < XFS_AGFL_SIZE(mp); bucket++)
-                        agfl->agfl_bno[bucket] = cpu_to_be32(NULLAGBLOCK);
+                        agfl_bno[bucket] = cpu_to_be32(NULLAGBLOCK);
                error = xfs_bwrite(bp);
                xfs_buf_relse(bp);
diff --git a/fs/xfs/xfs_ioctl.c b/fs/xfs/xfs_ioctl.c
index 4d613401a5e0..33ad9a77791f 100644
--- a/fs/xfs/xfs_ioctl.c
+++ b/fs/xfs/xfs_ioctl.c
@@ -442,7 +442,8 @@ xfs_attrlist_by_handle(
                return -XFS_ERROR(EPERM);
        if (copy_from_user(&al_hreq, arg, sizeof(xfs_fsop_attrlist_handlereq_t)))
                return -XFS_ERROR(EFAULT);
-        if (al_hreq.buflen > XATTR_LIST_MAX)
+        if (al_hreq.buflen < sizeof(struct attrlist) ||
+            al_hreq.buflen > XATTR_LIST_MAX)
                return -XFS_ERROR(EINVAL);
        /*
diff --git a/fs/xfs/xfs_ioctl32.c b/fs/xfs/xfs_ioctl32.c
index e8fb1231db81..a7992f8de9d3 100644
--- a/fs/xfs/xfs_ioctl32.c
+++ b/fs/xfs/xfs_ioctl32.c
@@ -356,7 +356,8 @@ xfs_compat_attrlist_by_handle(
        if (copy_from_user(&al_hreq, arg,
                           sizeof(compat_xfs_fsop_attrlist_handlereq_t)))
                return -XFS_ERROR(EFAULT);
-        if (al_hreq.buflen > XATTR_LIST_MAX)
+        if (al_hreq.buflen < sizeof(struct attrlist) ||
+            al_hreq.buflen > XATTR_LIST_MAX)
                return -XFS_ERROR(EINVAL);
        /*
diff --git a/fs/xfs/xfs_mount.c b/fs/xfs/xfs_mount.c
index da88f167af78..02df7b408a26 100644
--- a/fs/xfs/xfs_mount.c
+++ b/fs/xfs/xfs_mount.c
@@ -41,6 +41,7 @@
 #include "xfs_fsops.h"
 #include "xfs_trace.h"
 #include "xfs_icache.h"
+#include "xfs_dinode.h"
 #ifdef HAVE_PERCPU_SB
@@ -718,8 +719,22 @@ xfs_mountfs(
         * Set the inode cluster size.
         * This may still be overridden by the file system
         * block size if it is larger than the chosen cluster size.
+         *
+         * For v5 filesystems, scale the cluster size with the inode size to
+         * keep a constant ratio of inode per cluster buffer, but only if mkfs
+         * has set the inode alignment value appropriately for larger cluster
+         * sizes.
         */
        mp->m_inode_cluster_size = XFS_INODE_BIG_CLUSTER_SIZE;
+        if (xfs_sb_version_hascrc(&mp->m_sb)) {
+                int     new_size = mp->m_inode_cluster_size;
+                new_size *= mp->m_sb.sb_inodesize / XFS_DINODE_MIN_SIZE;
+                if (mp->m_sb.sb_inoalignmt >= XFS_B_TO_FSBT(mp, new_size))
+                        mp->m_inode_cluster_size = new_size;
+                xfs_info(mp, "Using inode cluster size of %d bytes",
+                         mp->m_inode_cluster_size);
+        }
        /*
         * Set inode alignment fields
diff --git a/fs/xfs/xfs_mount.h b/fs/xfs/xfs_mount.h
index 1d8101a10d8e..a466c5e5826e 100644
--- a/fs/xfs/xfs_mount.h
+++ b/fs/xfs/xfs_mount.h
@@ -112,7 +112,7 @@ typedef struct xfs_mount {
        __uint8_t               m_blkbb_log;    /* blocklog - BBSHIFT */
        __uint8_t               m_agno_log;     /* log #ag's */
        __uint8_t               m_agino_log;    /* #bits for agino in inum */
-        __uint16_t              m_inode_cluster_size;/* min inode buf size */
+        uint                    m_inode_cluster_size;/* min inode buf size */
        uint                    m_blockmask;    /* sb_blocksize-1 */
        uint                    m_blockwsize;   /* sb_blocksize in words */
        uint                    m_blockwmask;   /* blockwsize-1 */
diff --git a/fs/xfs/xfs_trans_inode.c b/fs/xfs/xfs_trans_inode.c
index 1bba7f60d94c..50c3f5614288 100644
--- a/fs/xfs/xfs_trans_inode.c
+++ b/fs/xfs/xfs_trans_inode.c
@@ -111,12 +111,14 @@ xfs_trans_log_inode(
        /*
         * First time we log the inode in a transaction, bump the inode change
-         * counter if it is configured for this to occur.
+         * counter if it is configured for this to occur. We don't use
+         * inode_inc_version() because there is no need for extra locking around
+         * i_version as we already hold the inode locked exclusively for
+         * metadata modification.
         */
        if (!(ip->i_itemp->ili_item.li_desc->lid_flags & XFS_LID_DIRTY) &&
            IS_I_VERSION(VFS_I(ip))) {
-                inode_inc_iversion(VFS_I(ip));
+                ip->i_d.di_changecount = ++VFS_I(ip)->i_version;
-                ip->i_d.di_changecount = VFS_I(ip)->i_version;
                flags |= XFS_ILOG_CORE;
        }
diff --git a/fs/xfs/xfs_trans_resv.c b/fs/xfs/xfs_trans_resv.c
index d53d9f0627a7..2fd59c0dae66 100644
--- a/fs/xfs/xfs_trans_resv.c
+++ b/fs/xfs/xfs_trans_resv.c
@@ -385,8 +385,7 @@ xfs_calc_ifree_reservation(
                xfs_calc_inode_res(mp, 1) +
                xfs_calc_buf_res(2, mp->m_sb.sb_sectsize) +
                xfs_calc_buf_res(1, XFS_FSB_TO_B(mp, 1)) +
-                MAX((__uint16_t)XFS_FSB_TO_B(mp, 1),
+                max_t(uint, XFS_FSB_TO_B(mp, 1), XFS_INODE_CLUSTER_SIZE(mp)) +
-                    XFS_INODE_CLUSTER_SIZE(mp)) +
                xfs_calc_buf_res(1, 0) +
                xfs_calc_buf_res(2 + XFS_IALLOC_BLOCKS(mp) +
                                 mp->m_in_maxlevels, 0) +
author	Ingo Molnar <mingo@kernel.org>	2013-12-17 09:27:08 -0500
committer	Ingo Molnar <mingo@kernel.org>	2013-12-17 09:27:08 -0500
commit	bb799d3b980eb803ca2da4a4eefbd9308f8d988a (patch)
tree	69fbe0cd6d47b23a50f5e1d87bf7489532fae149 /fs
parent	919fc6e34831d1c2b58bfb5ae261dc3facc9b269 (diff)
parent	319e2e3f63c348a9b66db4667efa73178e18b17d (diff)