204 files changed, 3994 insertions, 3403 deletions
diff --git a/fs/Makefile b/fs/Makefile
index 4030cbfbc9af..90c88529892b 100644
--- a/fs/Makefile
+++ b/fs/Makefile
@@ -11,7 +11,7 @@ obj-y :=	open.o read_write.o file_table.o super.o \
                attr.o bad_inode.o file.o filesystems.o namespace.o \
                seq_file.o xattr.o libfs.o fs-writeback.o \
                pnode.o splice.o sync.o utimes.o \
-                stack.o fs_struct.o statfs.o
+                stack.o fs_struct.o statfs.o fs_pin.o
 ifeq ($(CONFIG_BLOCK),y)
 obj-y +=        buffer.o block_dev.o direct-io.o mpage.o
diff --git a/fs/aio.c b/fs/aio.c
index bd7ec2cc2674..ae635872affb 100644
--- a/fs/aio.c
+++ b/fs/aio.c
@@ -192,7 +192,6 @@ static struct file *aio_private_file(struct kioctx *ctx, loff_t nr_pages)
        }
        file->f_flags = O_RDWR;
-        file->private_data = ctx;
        return file;
 }
@@ -202,7 +201,7 @@ static struct dentry *aio_mount(struct file_system_type *fs_type,
        static const struct dentry_operations ops = {
                .d_dname        = simple_dname,
        };
-        return mount_pseudo(fs_type, "aio:", NULL, &ops, 0xa10a10a1);
+        return mount_pseudo(fs_type, "aio:", NULL, &ops, AIO_RING_MAGIC);
 }
 /* aio_setup
@@ -556,8 +555,7 @@ static int ioctx_add_table(struct kioctx *ctx, struct mm_struct *mm)
        struct aio_ring *ring;
        spin_lock(&mm->ioctx_lock);
-        rcu_read_lock();
+        table = rcu_dereference_raw(mm->ioctx_table);
-        table = rcu_dereference(mm->ioctx_table);
        while (1) {
                if (table)
@@ -565,7 +563,6 @@ static int ioctx_add_table(struct kioctx *ctx, struct mm_struct *mm)
                                if (!table->table[i]) {
                                        ctx->id = i;
                                        table->table[i] = ctx;
-                                        rcu_read_unlock();
                                        spin_unlock(&mm->ioctx_lock);
                                        /* While kioctx setup is in progress,
@@ -579,8 +576,6 @@ static int ioctx_add_table(struct kioctx *ctx, struct mm_struct *mm)
                                }
                new_nr = (table ? table->nr : 1) * 4;
-                rcu_read_unlock();
                spin_unlock(&mm->ioctx_lock);
                table = kzalloc(sizeof(*table) + sizeof(struct kioctx *) *
@@ -591,8 +586,7 @@ static int ioctx_add_table(struct kioctx *ctx, struct mm_struct *mm)
                table->nr = new_nr;
                spin_lock(&mm->ioctx_lock);
-                rcu_read_lock();
+                old = rcu_dereference_raw(mm->ioctx_table);
-                old = rcu_dereference(mm->ioctx_table);
                if (!old) {
                        rcu_assign_pointer(mm->ioctx_table, table);
@@ -739,12 +733,9 @@ static int kill_ioctx(struct mm_struct *mm, struct kioctx *ctx,
        spin_lock(&mm->ioctx_lock);
-        rcu_read_lock();
+        table = rcu_dereference_raw(mm->ioctx_table);
-        table = rcu_dereference(mm->ioctx_table);
        WARN_ON(ctx != table->table[ctx->id]);
        table->table[ctx->id] = NULL;
-        rcu_read_unlock();
        spin_unlock(&mm->ioctx_lock);
        /* percpu_ref_kill() will do the necessary call_rcu() */
@@ -793,40 +784,30 @@ EXPORT_SYMBOL(wait_on_sync_kiocb);
 */
 void exit_aio(struct mm_struct *mm)
 {
-        struct kioctx_table *table;
+        struct kioctx_table *table = rcu_dereference_raw(mm->ioctx_table);
-        struct kioctx *ctx;
+        int i;
-        unsigned i = 0;
-        while (1) {
-                rcu_read_lock();
-                table = rcu_dereference(mm->ioctx_table);
-                do {
-                        if (!table || i >= table->nr) {
-                                rcu_read_unlock();
-                                rcu_assign_pointer(mm->ioctx_table, NULL);
-                                if (table)
-                                        kfree(table);
-                                return;
-                        }
-                        ctx = table->table[i++];
+        if (!table)
-                } while (!ctx);
+                return;
-                rcu_read_unlock();
+        for (i = 0; i < table->nr; ++i) {
+                struct kioctx *ctx = table->table[i];
+                if (!ctx)
+                        continue;
                /*
-                 * We don't need to bother with munmap() here -
+                 * We don't need to bother with munmap() here - exit_mmap(mm)
-                 * exit_mmap(mm) is coming and it'll unmap everything.
+                 * is coming and it'll unmap everything. And we simply can't,
-                 * Since aio_free_ring() uses non-zero ->mmap_size
+                 * this is not necessarily our ->mm.
-                 * as indicator that it needs to unmap the area,
+                 * Since kill_ioctx() uses non-zero ->mmap_size as indicator
-                 * just set it to 0; aio_free_ring() is the only
+                 * that it needs to unmap the area, just set it to 0.
-                 * place that uses ->mmap_size, so it's safe.
                 */
                ctx->mmap_size = 0;
                kill_ioctx(mm, ctx, NULL);
        }
+        RCU_INIT_POINTER(mm->ioctx_table, NULL);
+        kfree(table);
 }
 static void put_reqs_available(struct kioctx *ctx, unsigned nr)
@@ -834,10 +815,8 @@ static void put_reqs_available(struct kioctx *ctx, unsigned nr)
        struct kioctx_cpu *kcpu;
        unsigned long flags;
-        preempt_disable();
-        kcpu = this_cpu_ptr(ctx->cpu);
        local_irq_save(flags);
+        kcpu = this_cpu_ptr(ctx->cpu);
        kcpu->reqs_available += nr;
        while (kcpu->reqs_available >= ctx->req_batch * 2) {
@@ -846,7 +825,6 @@ static void put_reqs_available(struct kioctx *ctx, unsigned nr)
        }
        local_irq_restore(flags);
-        preempt_enable();
 }
 static bool get_reqs_available(struct kioctx *ctx)
@@ -855,10 +833,8 @@ static bool get_reqs_available(struct kioctx *ctx)
        bool ret = false;
        unsigned long flags;
-        preempt_disable();
-        kcpu = this_cpu_ptr(ctx->cpu);
        local_irq_save(flags);
+        kcpu = this_cpu_ptr(ctx->cpu);
        if (!kcpu->reqs_available) {
                int old, avail = atomic_read(&ctx->reqs_available);
@@ -878,7 +854,6 @@ static bool get_reqs_available(struct kioctx *ctx)
        kcpu->reqs_available--;
 out:
        local_irq_restore(flags);
-        preempt_enable();
        return ret;
 }
@@ -1047,7 +1022,7 @@ void aio_complete(struct kiocb *iocb, long res, long res2)
 }
 EXPORT_SYMBOL(aio_complete);
-/* aio_read_events
+/* aio_read_events_ring
 *      Pull an event off of the ioctx's event ring.  Returns the number of
 *      events fetched
 */
@@ -1270,12 +1245,12 @@ static ssize_t aio_setup_vectored_rw(struct kiocb *kiocb,
        if (compat)
                ret = compat_rw_copy_check_uvector(rw,
                                (struct compat_iovec __user *)buf,
-                                *nr_segs, 1, *iovec, iovec);
+                                *nr_segs, UIO_FASTIOV, *iovec, iovec);
        else
 #endif
                ret = rw_copy_check_uvector(rw,
                                (struct iovec __user *)buf,
-                                *nr_segs, 1, *iovec, iovec);
+                                *nr_segs, UIO_FASTIOV, *iovec, iovec);
        if (ret < 0)
                return ret;
@@ -1299,9 +1274,8 @@ static ssize_t aio_setup_single_vector(struct kiocb *kiocb,
 }
 /*
- * aio_setup_iocb:
+ * aio_run_iocb:
- *      Performs the initial checks and aio retry method
+ *      Performs the initial checks and io submission.
- *      setup for the kiocb at the time of io submission.
 */
 static ssize_t aio_run_iocb(struct kiocb *req, unsigned opcode,
                            char __user *buf, bool compat)
@@ -1313,7 +1287,7 @@ static ssize_t aio_run_iocb(struct kiocb *req, unsigned opcode,
        fmode_t mode;
        aio_rw_op *rw_op;
        rw_iter_op *iter_op;
-        struct iovec inline_vec, *iovec = &inline_vec;
+        struct iovec inline_vecs[UIO_FASTIOV], *iovec = inline_vecs;
        struct iov_iter iter;
        switch (opcode) {
@@ -1348,7 +1322,7 @@ rw_common:
                if (!ret)
                        ret = rw_verify_area(rw, file, &req->ki_pos, req->ki_nbytes);
                if (ret < 0) {
-                        if (iovec != &inline_vec)
+                        if (iovec != inline_vecs)
                                kfree(iovec);
                        return ret;
                }
@@ -1395,7 +1369,7 @@ rw_common:
                return -EINVAL;
        }
-        if (iovec != &inline_vec)
+        if (iovec != inline_vecs)
                kfree(iovec);
        if (ret != -EIOCBQUEUED) {
diff --git a/fs/bad_inode.c b/fs/bad_inode.c
index 7c93953030fb..afd2b4408adf 100644
--- a/fs/bad_inode.c
+++ b/fs/bad_inode.c
@@ -218,8 +218,9 @@ static int bad_inode_mknod (struct inode *dir, struct dentry *dentry,
        return -EIO;
 }
-static int bad_inode_rename (struct inode *old_dir, struct dentry *old_dentry,
+static int bad_inode_rename2(struct inode *old_dir, struct dentry *old_dentry,
-                struct inode *new_dir, struct dentry *new_dentry)
+                             struct inode *new_dir, struct dentry *new_dentry,
+                             unsigned int flags)
 {
        return -EIO;
 }
@@ -279,7 +280,7 @@ static const struct inode_operations bad_inode_ops =
        .mkdir          = bad_inode_mkdir,
        .rmdir          = bad_inode_rmdir,
        .mknod          = bad_inode_mknod,
-        .rename         = bad_inode_rename,
+        .rename2        = bad_inode_rename2,
        .readlink       = bad_inode_readlink,
        /* follow_link must be no-op, otherwise unmounting this inode
           won't work */
diff --git a/fs/btrfs/inode.c b/fs/btrfs/inode.c
index 3668048e16f8..3183742d6f0d 100644
--- a/fs/btrfs/inode.c
+++ b/fs/btrfs/inode.c
@@ -8476,6 +8476,16 @@ out_notrans:
        return ret;
 }
+static int btrfs_rename2(struct inode *old_dir, struct dentry *old_dentry,
+                         struct inode *new_dir, struct dentry *new_dentry,
+                         unsigned int flags)
+{
+        if (flags & ~RENAME_NOREPLACE)
+                return -EINVAL;
+        return btrfs_rename(old_dir, old_dentry, new_dir, new_dentry);
+}
 static void btrfs_run_delalloc_work(struct btrfs_work *work)
 {
        struct btrfs_delalloc_work *delalloc_work;
@@ -9019,7 +9029,7 @@ static const struct inode_operations btrfs_dir_inode_operations = {
        .link           = btrfs_link,
        .mkdir          = btrfs_mkdir,
        .rmdir          = btrfs_rmdir,
-        .rename         = btrfs_rename,
+        .rename2        = btrfs_rename2,
        .symlink        = btrfs_symlink,
        .setattr        = btrfs_setattr,
        .mknod          = btrfs_mknod,
diff --git a/fs/btrfs/super.c b/fs/btrfs/super.c
index 8e16bca69c56..67b48b9a03e0 100644
--- a/fs/btrfs/super.c
+++ b/fs/btrfs/super.c
@@ -851,7 +851,6 @@ static struct dentry *get_default_root(struct super_block *sb,
        struct btrfs_path *path;
        struct btrfs_key location;
        struct inode *inode;
-        struct dentry *dentry;
        u64 dir_id;
        int new = 0;
@@ -922,13 +921,7 @@ setup_root:
                return dget(sb->s_root);
        }
-        dentry = d_obtain_alias(inode);
+        return d_obtain_root(inode);
-        if (!IS_ERR(dentry)) {
-                spin_lock(&dentry->d_lock);
-                dentry->d_flags &= ~DCACHE_DISCONNECTED;
-                spin_unlock(&dentry->d_lock);
-        }
-        return dentry;
 }
 static int btrfs_fill_super(struct super_block *sb,
diff --git a/fs/ceph/acl.c b/fs/ceph/acl.c
index 469f2e8657e8..cebf2ebefb55 100644
--- a/fs/ceph/acl.c
+++ b/fs/ceph/acl.c
@@ -172,14 +172,24 @@ out:
 int ceph_init_acl(struct dentry *dentry, struct inode *inode, struct inode *dir)
 {
        struct posix_acl *default_acl, *acl;
+        umode_t new_mode = inode->i_mode;
        int error;
-        error = posix_acl_create(dir, &inode->i_mode, &default_acl, &acl);
+        error = posix_acl_create(dir, &new_mode, &default_acl, &acl);
        if (error)
                return error;
-        if (!default_acl && !acl)
+        if (!default_acl && !acl) {
                cache_no_acl(inode);
+                if (new_mode != inode->i_mode) {
+                        struct iattr newattrs = {
+                                .ia_mode = new_mode,
+                                .ia_valid = ATTR_MODE,
+                        };
+                        error = ceph_setattr(dentry, &newattrs);
+                }
+                return error;
+        }
        if (default_acl) {
                error = ceph_set_acl(inode, default_acl, ACL_TYPE_DEFAULT);
diff --git a/fs/ceph/caps.c b/fs/ceph/caps.c
index 1fde164b74b5..6d1cd45dca89 100644
--- a/fs/ceph/caps.c
+++ b/fs/ceph/caps.c
@@ -3277,7 +3277,7 @@ int ceph_encode_inode_release(void **p, struct inode *inode,
                        rel->ino = cpu_to_le64(ceph_ino(inode));
                        rel->cap_id = cpu_to_le64(cap->cap_id);
                        rel->seq = cpu_to_le32(cap->seq);
-                        rel->issue_seq = cpu_to_le32(cap->issue_seq),
+                        rel->issue_seq = cpu_to_le32(cap->issue_seq);
                        rel->mseq = cpu_to_le32(cap->mseq);
                        rel->caps = cpu_to_le32(cap->implemented);
                        rel->wanted = cpu_to_le32(cap->mds_wanted);
diff --git a/fs/ceph/file.c b/fs/ceph/file.c
index 302085100c28..2eb02f80a0ab 100644
--- a/fs/ceph/file.c
+++ b/fs/ceph/file.c
@@ -423,6 +423,9 @@ static ssize_t ceph_sync_read(struct kiocb *iocb, struct iov_iter *i,
        dout("sync_read on file %p %llu~%u %s\n", file, off,
             (unsigned)len,
             (file->f_flags & O_DIRECT) ? "O_DIRECT" : "");
+        if (!len)
+                return 0;
        /*
         * flush any page cache pages in this range.  this
         * will make concurrent normal and sync io slow,
@@ -470,8 +473,11 @@ static ssize_t ceph_sync_read(struct kiocb *iocb, struct iov_iter *i,
                        size_t left = ret;
                        while (left) {
-                                int copy = min_t(size_t, PAGE_SIZE, left);
+                                size_t page_off = off & ~PAGE_MASK;
-                                l = copy_page_to_iter(pages[k++], 0, copy, i);
+                                size_t copy = min_t(size_t,
+                                                    PAGE_SIZE - page_off, left);
+                                l = copy_page_to_iter(pages[k++], page_off,
+                                                      copy, i);
                                off += l;
                                left -= l;
                                if (l < copy)
@@ -531,7 +537,7 @@ static void ceph_sync_write_unsafe(struct ceph_osd_request *req, bool unsafe)
 * objects, rollback on failure, etc.)
 */
 static ssize_t
-ceph_sync_direct_write(struct kiocb *iocb, struct iov_iter *from)
+ceph_sync_direct_write(struct kiocb *iocb, struct iov_iter *from, loff_t pos)
 {
        struct file *file = iocb->ki_filp;
        struct inode *inode = file_inode(file);
@@ -547,7 +553,6 @@ ceph_sync_direct_write(struct kiocb *iocb, struct iov_iter *from)
        int check_caps = 0;
        int ret;
        struct timespec mtime = CURRENT_TIME;
-        loff_t pos = iocb->ki_pos;
        size_t count = iov_iter_count(from);
        if (ceph_snap(file_inode(file)) != CEPH_NOSNAP)
@@ -646,7 +651,8 @@ ceph_sync_direct_write(struct kiocb *iocb, struct iov_iter *from)
 * correct atomic write, we should e.g. take write locks on all
 * objects, rollback on failure, etc.)
 */
-static ssize_t ceph_sync_write(struct kiocb *iocb, struct iov_iter *from)
+static ssize_t
+ceph_sync_write(struct kiocb *iocb, struct iov_iter *from, loff_t pos)
 {
        struct file *file = iocb->ki_filp;
        struct inode *inode = file_inode(file);
@@ -663,7 +669,6 @@ static ssize_t ceph_sync_write(struct kiocb *iocb, struct iov_iter *from)
        int check_caps = 0;
        int ret;
        struct timespec mtime = CURRENT_TIME;
-        loff_t pos = iocb->ki_pos;
        size_t count = iov_iter_count(from);
        if (ceph_snap(file_inode(file)) != CEPH_NOSNAP)
@@ -918,9 +923,9 @@ retry_snap:
                /* we might need to revert back to that point */
                data = *from;
                if (file->f_flags & O_DIRECT)
-                        written = ceph_sync_direct_write(iocb, &data);
+                        written = ceph_sync_direct_write(iocb, &data, pos);
                else
-                        written = ceph_sync_write(iocb, &data);
+                        written = ceph_sync_write(iocb, &data, pos);
                if (written == -EOLDSNAPC) {
                        dout("aio_write %p %llx.%llx %llu~%u"
                                "got EOLDSNAPC, retrying\n",
@@ -1177,6 +1182,9 @@ static long ceph_fallocate(struct file *file, int mode,
        loff_t endoff = 0;
        loff_t size;
+        if (mode & ~(FALLOC_FL_KEEP_SIZE | FALLOC_FL_PUNCH_HOLE))
+                return -EOPNOTSUPP;
        if (!S_ISREG(inode->i_mode))
                return -EOPNOTSUPP;
diff --git a/fs/ceph/mds_client.c b/fs/ceph/mds_client.c
index 92a2548278fc..bad07c09f91e 100644
--- a/fs/ceph/mds_client.c
+++ b/fs/ceph/mds_client.c
@@ -1904,6 +1904,7 @@ static int __prepare_send_request(struct ceph_mds_client *mdsc,
             req->r_tid, ceph_mds_op_name(req->r_op), req->r_attempts);
        if (req->r_got_unsafe) {
+                void *p;
                /*
                 * Replay.  Do not regenerate message (and rebuild
                 * paths, etc.); just use the original message.
@@ -1924,8 +1925,13 @@ static int __prepare_send_request(struct ceph_mds_client *mdsc,
                /* remove cap/dentry releases from message */
                rhead->num_releases = 0;
-                msg->hdr.front_len = cpu_to_le32(req->r_request_release_offset);
-                msg->front.iov_len = req->r_request_release_offset;
+                /* time stamp */
+                p = msg->front.iov_base + req->r_request_release_offset;
+                ceph_encode_copy(&p, &req->r_stamp, sizeof(req->r_stamp));
+                msg->front.iov_len = p - msg->front.iov_base;
+                msg->hdr.front_len = cpu_to_le32(msg->front.iov_len);
                return 0;
        }
@@ -2061,11 +2067,12 @@ static void __wake_requests(struct ceph_mds_client *mdsc,
 static void kick_requests(struct ceph_mds_client *mdsc, int mds)
 {
        struct ceph_mds_request *req;
-        struct rb_node *p;
+        struct rb_node *p = rb_first(&mdsc->request_tree);
        dout("kick_requests mds%d\n", mds);
-        for (p = rb_first(&mdsc->request_tree); p; p = rb_next(p)) {
+        while (p) {
                req = rb_entry(p, struct ceph_mds_request, r_node);
+                p = rb_next(p);
                if (req->r_got_unsafe)
                        continue;
                if (req->r_session &&
@@ -2248,6 +2255,7 @@ static void handle_reply(struct ceph_mds_session *session, struct ceph_msg *msg)
         */
        if (result == -ESTALE) {
                dout("got ESTALE on request %llu", req->r_tid);
+                req->r_resend_mds = -1;
                if (req->r_direct_mode != USE_AUTH_MDS) {
                        dout("not using auth, setting for that now");
                        req->r_direct_mode = USE_AUTH_MDS;
diff --git a/fs/ceph/super.c b/fs/ceph/super.c
index 06150fd745ac..f6e12377335c 100644
--- a/fs/ceph/super.c
+++ b/fs/ceph/super.c
@@ -755,7 +755,7 @@ static struct dentry *open_root_dentry(struct ceph_fs_client *fsc,
                                goto out;
                        }
                } else {
-                        root = d_obtain_alias(inode);
+                        root = d_obtain_root(inode);
                }
                ceph_init_dentry(root);
                dout("open_root_inode success, root dentry is %p\n", root);
diff --git a/fs/ceph/xattr.c b/fs/ceph/xattr.c
index c9c2b887381e..12f58d22e017 100644
--- a/fs/ceph/xattr.c
+++ b/fs/ceph/xattr.c
@@ -592,12 +592,12 @@ start:
                xattr_version = ci->i_xattrs.version;
                spin_unlock(&ci->i_ceph_lock);
-                xattrs = kcalloc(numattr, sizeof(struct ceph_xattr *),
+                xattrs = kcalloc(numattr, sizeof(struct ceph_inode_xattr *),
                                 GFP_NOFS);
                err = -ENOMEM;
                if (!xattrs)
                        goto bad_lock;
-                memset(xattrs, 0, numattr*sizeof(struct ceph_xattr *));
                for (i = 0; i < numattr; i++) {
                        xattrs[i] = kmalloc(sizeof(struct ceph_inode_xattr),
                                            GFP_NOFS);
diff --git a/fs/cifs/cifsfs.c b/fs/cifs/cifsfs.c
index 888398067420..ac4f260155c8 100644
--- a/fs/cifs/cifsfs.c
+++ b/fs/cifs/cifsfs.c
@@ -848,7 +848,7 @@ const struct inode_operations cifs_dir_inode_ops = {
        .link = cifs_hardlink,
        .mkdir = cifs_mkdir,
        .rmdir = cifs_rmdir,
-        .rename = cifs_rename,
+        .rename2 = cifs_rename2,
        .permission = cifs_permission,
 /*      revalidate:cifs_revalidate,   */
        .setattr = cifs_setattr,
diff --git a/fs/cifs/cifsfs.h b/fs/cifs/cifsfs.h
index 560480263336..b0fafa499505 100644
--- a/fs/cifs/cifsfs.h
+++ b/fs/cifs/cifsfs.h
@@ -68,8 +68,8 @@ extern int cifs_hardlink(struct dentry *, struct inode *, struct dentry *);
 extern int cifs_mknod(struct inode *, struct dentry *, umode_t, dev_t);
 extern int cifs_mkdir(struct inode *, struct dentry *, umode_t);
 extern int cifs_rmdir(struct inode *, struct dentry *);
-extern int cifs_rename(struct inode *, struct dentry *, struct inode *,
+extern int cifs_rename2(struct inode *, struct dentry *, struct inode *,
-                       struct dentry *);
+                        struct dentry *, unsigned int);
 extern int cifs_revalidate_file_attr(struct file *filp);
 extern int cifs_revalidate_dentry_attr(struct dentry *);
 extern int cifs_revalidate_file(struct file *filp);
diff --git a/fs/cifs/inode.c b/fs/cifs/inode.c
index 41de3935caa0..426d6c6ad8bf 100644
--- a/fs/cifs/inode.c
+++ b/fs/cifs/inode.c
@@ -1627,8 +1627,9 @@ do_rename_exit:
 }
 int
-cifs_rename(struct inode *source_dir, struct dentry *source_dentry,
+cifs_rename2(struct inode *source_dir, struct dentry *source_dentry,
-            struct inode *target_dir, struct dentry *target_dentry)
+             struct inode *target_dir, struct dentry *target_dentry,
+             unsigned int flags)
 {
        char *from_name = NULL;
        char *to_name = NULL;
@@ -1640,6 +1641,9 @@ cifs_rename(struct inode *source_dir, struct dentry *source_dentry,
        unsigned int xid;
        int rc, tmprc;
+        if (flags & ~RENAME_NOREPLACE)
+                return -EINVAL;
        cifs_sb = CIFS_SB(source_dir->i_sb);
        tlink = cifs_sb_tlink(cifs_sb);
        if (IS_ERR(tlink))
@@ -1667,6 +1671,12 @@ cifs_rename(struct inode *source_dir, struct dentry *source_dentry,
        rc = cifs_do_rename(xid, source_dentry, from_name, target_dentry,
                            to_name);
+        /*
+         * No-replace is the natural behavior for CIFS, so skip unlink hacks.
+         */
+        if (flags & RENAME_NOREPLACE)
+                goto cifs_rename_exit;
        if (rc == -EEXIST && tcon->unix_ext) {
                /*
                 * Are src and dst hardlinks of same inode? We can only tell
diff --git a/fs/dcache.c b/fs/dcache.c
index 06f65857a855..d30ce699ae4b 100644
--- a/fs/dcache.c
+++ b/fs/dcache.c
@@ -731,8 +731,6 @@ EXPORT_SYMBOL(dget_parent);
 /**
 * d_find_alias - grab a hashed alias of inode
 * @inode: inode in question
- * @want_discon:  flag, used by d_splice_alias, to request
- *          that only a DISCONNECTED alias be returned.
 *
 * If inode has a hashed alias, or is a directory and has any alias,
 * acquire the reference to alias and return it. Otherwise return NULL.
@@ -741,10 +739,9 @@ EXPORT_SYMBOL(dget_parent);
 * of a filesystem.
 *
 * If the inode has an IS_ROOT, DCACHE_DISCONNECTED alias, then prefer
- * any other hashed alias over that one unless @want_discon is set,
+ * any other hashed alias over that one.
- * in which case only return an IS_ROOT, DCACHE_DISCONNECTED alias.
 */
-static struct dentry *__d_find_alias(struct inode *inode, int want_discon)
+static struct dentry *__d_find_alias(struct inode *inode)
 {
        struct dentry *alias, *discon_alias;
@@ -756,7 +753,7 @@ again:
                        if (IS_ROOT(alias) &&
                            (alias->d_flags & DCACHE_DISCONNECTED)) {
                                discon_alias = alias;
-                        } else if (!want_discon) {
+                        } else {
                                __dget_dlock(alias);
                                spin_unlock(&alias->d_lock);
                                return alias;
@@ -768,12 +765,9 @@ again:
                alias = discon_alias;
                spin_lock(&alias->d_lock);
                if (S_ISDIR(inode->i_mode) || !d_unhashed(alias)) {
-                        if (IS_ROOT(alias) &&
+                        __dget_dlock(alias);
-                            (alias->d_flags & DCACHE_DISCONNECTED)) {
+                        spin_unlock(&alias->d_lock);
-                                __dget_dlock(alias);
+                        return alias;
-                                spin_unlock(&alias->d_lock);
-                                return alias;
-                        }
                }
                spin_unlock(&alias->d_lock);
                goto again;
@@ -787,7 +781,7 @@ struct dentry *d_find_alias(struct inode *inode)
        if (!hlist_empty(&inode->i_dentry)) {
                spin_lock(&inode->i_lock);
-                de = __d_find_alias(inode, 0);
+                de = __d_find_alias(inode);
                spin_unlock(&inode->i_lock);
        }
        return de;
@@ -1781,25 +1775,7 @@ struct dentry *d_find_any_alias(struct inode *inode)
 }
 EXPORT_SYMBOL(d_find_any_alias);
-/**
+static struct dentry *__d_obtain_alias(struct inode *inode, int disconnected)
- * d_obtain_alias - find or allocate a dentry for a given inode
- * @inode: inode to allocate the dentry for
- *
- * Obtain a dentry for an inode resulting from NFS filehandle conversion or
- * similar open by handle operations.  The returned dentry may be anonymous,
- * or may have a full name (if the inode was already in the cache).
- *
- * When called on a directory inode, we must ensure that the inode only ever
- * has one dentry.  If a dentry is found, that is returned instead of
- * allocating a new one.
- *
- * On successful return, the reference to the inode has been transferred
- * to the dentry.  In case of an error the reference on the inode is released.
- * To make it easier to use in export operations a %NULL or IS_ERR inode may
- * be passed in and will be the error will be propagate to the return value,
- * with a %NULL @inode replaced by ERR_PTR(-ESTALE).
- */
-struct dentry *d_obtain_alias(struct inode *inode)
 {
        static const struct qstr anonstring = QSTR_INIT("/", 1);
        struct dentry *tmp;
@@ -1830,7 +1806,10 @@ struct dentry *d_obtain_alias(struct inode *inode)
        }
        /* attach a disconnected dentry */
-        add_flags = d_flags_for_inode(inode) | DCACHE_DISCONNECTED;
+        add_flags = d_flags_for_inode(inode);
+        if (disconnected)
+                add_flags |= DCACHE_DISCONNECTED;
        spin_lock(&tmp->d_lock);
        tmp->d_inode = inode;
@@ -1851,59 +1830,51 @@ struct dentry *d_obtain_alias(struct inode *inode)
        iput(inode);
        return res;
 }
-EXPORT_SYMBOL(d_obtain_alias);
 /**
- * d_splice_alias - splice a disconnected dentry into the tree if one exists
+ * d_obtain_alias - find or allocate a DISCONNECTED dentry for a given inode
- * @inode:  the inode which may have a disconnected dentry
+ * @inode: inode to allocate the dentry for
- * @dentry: a negative dentry which we want to point to the inode.
- *
- * If inode is a directory and has a 'disconnected' dentry (i.e. IS_ROOT and
- * DCACHE_DISCONNECTED), then d_move that in place of the given dentry
- * and return it, else simply d_add the inode to the dentry and return NULL.
 *
- * This is needed in the lookup routine of any filesystem that is exportable
+ * Obtain a dentry for an inode resulting from NFS filehandle conversion or
- * (via knfsd) so that we can build dcache paths to directories effectively.
+ * similar open by handle operations.  The returned dentry may be anonymous,
+ * or may have a full name (if the inode was already in the cache).
 *
- * If a dentry was found and moved, then it is returned.  Otherwise NULL
+ * When called on a directory inode, we must ensure that the inode only ever
- * is returned.  This matches the expected return value of ->lookup.
+ * has one dentry.  If a dentry is found, that is returned instead of
+ * allocating a new one.
 *
- * Cluster filesystems may call this function with a negative, hashed dentry.
+ * On successful return, the reference to the inode has been transferred
- * In that case, we know that the inode will be a regular file, and also this
+ * to the dentry.  In case of an error the reference on the inode is released.
- * will only occur during atomic_open. So we need to check for the dentry
+ * To make it easier to use in export operations a %NULL or IS_ERR inode may
- * being already hashed only in the final case.
+ * be passed in and the error will be propagated to the return value,
+ * with a %NULL @inode replaced by ERR_PTR(-ESTALE).
 */
-struct dentry *d_splice_alias(struct inode *inode, struct dentry *dentry)
+struct dentry *d_obtain_alias(struct inode *inode)
 {
-        struct dentry *new = NULL;
+        return __d_obtain_alias(inode, 1);
+}
-        if (IS_ERR(inode))
+EXPORT_SYMBOL(d_obtain_alias);
-                return ERR_CAST(inode);
-        if (inode && S_ISDIR(inode->i_mode)) {
+/**
-                spin_lock(&inode->i_lock);
+ * d_obtain_root - find or allocate a dentry for a given inode
-                new = __d_find_alias(inode, 1);
+ * @inode: inode to allocate the dentry for
-                if (new) {
+ *
-                        BUG_ON(!(new->d_flags & DCACHE_DISCONNECTED));
+ * Obtain an IS_ROOT dentry for the root of a filesystem.
-                        spin_unlock(&inode->i_lock);
+ *
-                        security_d_instantiate(new, inode);
+ * We must ensure that directory inodes only ever have one dentry.  If a
-                        d_move(new, dentry);
+ * dentry is found, that is returned instead of allocating a new one.
-                        iput(inode);
+ *
-                } else {
+ * On successful return, the reference to the inode has been transferred
-                        /* already taking inode->i_lock, so d_add() by hand */
+ * to the dentry.  In case of an error the reference on the inode is
-                        __d_instantiate(dentry, inode);
+ * released.  A %NULL or IS_ERR inode may be passed in and will be the
-                        spin_unlock(&inode->i_lock);
+ * error will be propagate to the return value, with a %NULL @inode
-                        security_d_instantiate(dentry, inode);
+ * replaced by ERR_PTR(-ESTALE).
-                        d_rehash(dentry);
+ */
-                }
+struct dentry *d_obtain_root(struct inode *inode)
-        } else {
+{
-                d_instantiate(dentry, inode);
+        return __d_obtain_alias(inode, 0);
-                if (d_unhashed(dentry))
-                        d_rehash(dentry);
-        }
-        return new;
 }
-EXPORT_SYMBOL(d_splice_alias);
+EXPORT_SYMBOL(d_obtain_root);
 /**
 * d_add_ci - lookup or allocate new dentry with case-exact name
@@ -2697,6 +2668,75 @@ static void __d_materialise_dentry(struct dentry *dentry, struct dentry *anon)
 }
 /**
+ * d_splice_alias - splice a disconnected dentry into the tree if one exists
+ * @inode:  the inode which may have a disconnected dentry
+ * @dentry: a negative dentry which we want to point to the inode.
+ *
+ * If inode is a directory and has an IS_ROOT alias, then d_move that in
+ * place of the given dentry and return it, else simply d_add the inode
+ * to the dentry and return NULL.
+ *
+ * If a non-IS_ROOT directory is found, the filesystem is corrupt, and
+ * we should error out: directories can't have multiple aliases.
+ *
+ * This is needed in the lookup routine of any filesystem that is exportable
+ * (via knfsd) so that we can build dcache paths to directories effectively.
+ *
+ * If a dentry was found and moved, then it is returned.  Otherwise NULL
+ * is returned.  This matches the expected return value of ->lookup.
+ *
+ * Cluster filesystems may call this function with a negative, hashed dentry.
+ * In that case, we know that the inode will be a regular file, and also this
+ * will only occur during atomic_open. So we need to check for the dentry
+ * being already hashed only in the final case.
+ */
+struct dentry *d_splice_alias(struct inode *inode, struct dentry *dentry)
+{
+        struct dentry *new = NULL;
+        if (IS_ERR(inode))
+                return ERR_CAST(inode);
+        if (inode && S_ISDIR(inode->i_mode)) {
+                spin_lock(&inode->i_lock);
+                new = __d_find_any_alias(inode);
+                if (new) {
+                        if (!IS_ROOT(new)) {
+                                spin_unlock(&inode->i_lock);
+                                dput(new);
+                                return ERR_PTR(-EIO);
+                        }
+                        if (d_ancestor(new, dentry)) {
+                                spin_unlock(&inode->i_lock);
+                                dput(new);
+                                return ERR_PTR(-EIO);
+                        }
+                        write_seqlock(&rename_lock);
+                        __d_materialise_dentry(dentry, new);
+                        write_sequnlock(&rename_lock);
+                        __d_drop(new);
+                        _d_rehash(new);
+                        spin_unlock(&new->d_lock);
+                        spin_unlock(&inode->i_lock);
+                        security_d_instantiate(new, inode);
+                        iput(inode);
+                } else {
+                        /* already taking inode->i_lock, so d_add() by hand */
+                        __d_instantiate(dentry, inode);
+                        spin_unlock(&inode->i_lock);
+                        security_d_instantiate(dentry, inode);
+                        d_rehash(dentry);
+                }
+        } else {
+                d_instantiate(dentry, inode);
+                if (d_unhashed(dentry))
+                        d_rehash(dentry);
+        }
+        return new;
+}
+EXPORT_SYMBOL(d_splice_alias);
+/**
 * d_materialise_unique - introduce an inode into the tree
 * @dentry: candidate dentry
 * @inode: inode to bind to the dentry, to which aliases may be attached
@@ -2724,7 +2764,7 @@ struct dentry *d_materialise_unique(struct dentry *dentry, struct inode *inode)
                struct dentry *alias;
                /* Does an aliased dentry already exist? */
-                alias = __d_find_alias(inode, 0);
+                alias = __d_find_alias(inode);
                if (alias) {
                        actual = alias;
                        write_seqlock(&rename_lock);
diff --git a/fs/direct-io.c b/fs/direct-io.c
index 17e39b047de5..c3116404ab49 100644
--- a/fs/direct-io.c
+++ b/fs/direct-io.c
@@ -158,7 +158,7 @@ static inline int dio_refill_pages(struct dio *dio, struct dio_submit *sdio)
 {
        ssize_t ret;
-        ret = iov_iter_get_pages(sdio->iter, dio->pages, DIO_PAGES * PAGE_SIZE,
+        ret = iov_iter_get_pages(sdio->iter, dio->pages, DIO_PAGES,
                                &sdio->from);
        if (ret < 0 && sdio->blocks_available && (dio->rw & WRITE)) {
diff --git a/fs/ext2/super.c b/fs/ext2/super.c
index 3750031cfa2f..b88edc05c230 100644
--- a/fs/ext2/super.c
+++ b/fs/ext2/super.c
@@ -161,7 +161,7 @@ static struct kmem_cache * ext2_inode_cachep;
 static struct inode *ext2_alloc_inode(struct super_block *sb)
 {
        struct ext2_inode_info *ei;
-        ei = (struct ext2_inode_info *)kmem_cache_alloc(ext2_inode_cachep, GFP_KERNEL);
+        ei = kmem_cache_alloc(ext2_inode_cachep, GFP_KERNEL);
        if (!ei)
                return NULL;
        ei->i_block_alloc_info = NULL;
diff --git a/fs/ext4/namei.c b/fs/ext4/namei.c
index 3520ab8a6639..b147a67baa0d 100644
--- a/fs/ext4/namei.c
+++ b/fs/ext4/namei.c
@@ -3455,7 +3455,6 @@ const struct inode_operations ext4_dir_inode_operations = {
        .rmdir          = ext4_rmdir,
        .mknod          = ext4_mknod,
        .tmpfile        = ext4_tmpfile,
-        .rename         = ext4_rename,
        .rename2        = ext4_rename2,
        .setattr        = ext4_setattr,
        .setxattr       = generic_setxattr,
diff --git a/fs/fs_pin.c b/fs/fs_pin.c
new file mode 100644
index 000000000000..9368236ca100
--- /dev/null
+++ b/fs/fs_pin.c
@@ -0,0 +1,78 @@
+#include <linux/fs.h>
+#include <linux/slab.h>
+#include <linux/fs_pin.h>
+#include "internal.h"
+#include "mount.h"
+static void pin_free_rcu(struct rcu_head *head)
+{
+        kfree(container_of(head, struct fs_pin, rcu));
+}
+static DEFINE_SPINLOCK(pin_lock);
+void pin_put(struct fs_pin *p)
+{
+        if (atomic_long_dec_and_test(&p->count))
+                call_rcu(&p->rcu, pin_free_rcu);
+}
+void pin_remove(struct fs_pin *pin)
+{
+        spin_lock(&pin_lock);
+        hlist_del(&pin->m_list);
+        hlist_del(&pin->s_list);
+        spin_unlock(&pin_lock);
+}
+void pin_insert(struct fs_pin *pin, struct vfsmount *m)
+{
+        spin_lock(&pin_lock);
+        hlist_add_head(&pin->s_list, &m->mnt_sb->s_pins);
+        hlist_add_head(&pin->m_list, &real_mount(m)->mnt_pins);
+        spin_unlock(&pin_lock);
+}
+void mnt_pin_kill(struct mount *m)
+{
+        while (1) {
+                struct hlist_node *p;
+                struct fs_pin *pin;
+                rcu_read_lock();
+                p = ACCESS_ONCE(m->mnt_pins.first);
+                if (!p) {
+                        rcu_read_unlock();
+                        break;
+                }
+                pin = hlist_entry(p, struct fs_pin, m_list);
+                if (!atomic_long_inc_not_zero(&pin->count)) {
+                        rcu_read_unlock();
+                        cpu_relax();
+                        continue;
+                }
+                rcu_read_unlock();
+                pin->kill(pin);
+        }
+}
+void sb_pin_kill(struct super_block *sb)
+{
+        while (1) {
+                struct hlist_node *p;
+                struct fs_pin *pin;
+                rcu_read_lock();
+                p = ACCESS_ONCE(sb->s_pins.first);
+                if (!p) {
+                        rcu_read_unlock();
+                        break;
+                }
+                pin = hlist_entry(p, struct fs_pin, s_list);
+                if (!atomic_long_inc_not_zero(&pin->count)) {
+                        rcu_read_unlock();
+                        cpu_relax();
+                        continue;
+                }
+                rcu_read_unlock();
+                pin->kill(pin);
+        }
+}
diff --git a/fs/fuse/dir.c b/fs/fuse/dir.c
index 0c6048247a34..de1d84af9f7c 100644
--- a/fs/fuse/dir.c
+++ b/fs/fuse/dir.c
@@ -845,12 +845,6 @@ static int fuse_rename2(struct inode *olddir, struct dentry *oldent,
        return err;
 }
-static int fuse_rename(struct inode *olddir, struct dentry *oldent,
-                       struct inode *newdir, struct dentry *newent)
-{
-        return fuse_rename2(olddir, oldent, newdir, newent, 0);
-}
 static int fuse_link(struct dentry *entry, struct inode *newdir,
                     struct dentry *newent)
 {
@@ -2024,7 +2018,6 @@ static const struct inode_operations fuse_dir_inode_operations = {
        .symlink        = fuse_symlink,
        .unlink         = fuse_unlink,
        .rmdir          = fuse_rmdir,
-        .rename         = fuse_rename,
        .rename2        = fuse_rename2,
        .link           = fuse_link,
        .setattr        = fuse_setattr,
diff --git a/fs/fuse/file.c b/fs/fuse/file.c
index 40ac2628ddcf..912061ac4baf 100644
--- a/fs/fuse/file.c
+++ b/fs/fuse/file.c
@@ -1303,10 +1303,10 @@ static int fuse_get_user_pages(struct fuse_req *req, struct iov_iter *ii,
        while (nbytes < *nbytesp && req->num_pages < req->max_pages) {
                unsigned npages;
                size_t start;
-                unsigned n = req->max_pages - req->num_pages;
                ssize_t ret = iov_iter_get_pages(ii,
                                        &req->pages[req->num_pages],
-                                        n * PAGE_SIZE, &start);
+                                        req->max_pages - req->num_pages,
+                                        &start);
                if (ret < 0)
                        return ret;
diff --git a/fs/hostfs/hostfs.h b/fs/hostfs/hostfs.h
index 9c88da0e855a..4fcd40d6f308 100644
--- a/fs/hostfs/hostfs.h
+++ b/fs/hostfs/hostfs.h
@@ -89,6 +89,7 @@ extern int do_mknod(const char *file, int mode, unsigned int major,
 extern int link_file(const char *from, const char *to);
 extern int hostfs_do_readlink(char *file, char *buf, int size);
 extern int rename_file(char *from, char *to);
+extern int rename2_file(char *from, char *to, unsigned int flags);
 extern int do_statfs(char *root, long *bsize_out, long long *blocks_out,
                     long long *bfree_out, long long *bavail_out,
                     long long *files_out, long long *ffree_out,
diff --git a/fs/hostfs/hostfs_kern.c b/fs/hostfs/hostfs_kern.c
index bb529f3b7f2b..fd62cae0fdcb 100644
--- a/fs/hostfs/hostfs_kern.c
+++ b/fs/hostfs/hostfs_kern.c
@@ -741,21 +741,31 @@ static int hostfs_mknod(struct inode *dir, struct dentry *dentry, umode_t mode,
        return err;
 }
-static int hostfs_rename(struct inode *from_ino, struct dentry *from,
+static int hostfs_rename2(struct inode *old_dir, struct dentry *old_dentry,
-                         struct inode *to_ino, struct dentry *to)
+                          struct inode *new_dir, struct dentry *new_dentry,
+                          unsigned int flags)
 {
-        char *from_name, *to_name;
+        char *old_name, *new_name;
        int err;
-        if ((from_name = dentry_name(from)) == NULL)
+        if (flags & ~(RENAME_NOREPLACE | RENAME_EXCHANGE))
+                return -EINVAL;
+        old_name = dentry_name(old_dentry);
+        if (old_name == NULL)
                return -ENOMEM;
-        if ((to_name = dentry_name(to)) == NULL) {
+        new_name = dentry_name(new_dentry);
-                __putname(from_name);
+        if (new_name == NULL) {
+                __putname(old_name);
                return -ENOMEM;
        }
-        err = rename_file(from_name, to_name);
+        if (!flags)
-        __putname(from_name);
+                err = rename_file(old_name, new_name);
-        __putname(to_name);
+        else
+                err = rename2_file(old_name, new_name, flags);
+        __putname(old_name);
+        __putname(new_name);
        return err;
 }
@@ -867,7 +877,7 @@ static const struct inode_operations hostfs_dir_iops = {
        .mkdir          = hostfs_mkdir,
        .rmdir          = hostfs_rmdir,
        .mknod          = hostfs_mknod,
-        .rename         = hostfs_rename,
+        .rename2        = hostfs_rename2,
        .permission     = hostfs_permission,
        .setattr        = hostfs_setattr,
 };
diff --git a/fs/hostfs/hostfs_user.c b/fs/hostfs/hostfs_user.c
index 67838f3aa20a..9765dab95cbd 100644
--- a/fs/hostfs/hostfs_user.c
+++ b/fs/hostfs/hostfs_user.c
@@ -14,6 +14,7 @@
 #include <sys/time.h>
 #include <sys/types.h>
 #include <sys/vfs.h>
+#include <sys/syscall.h>
 #include "hostfs.h"
 #include <utime.h>
@@ -360,6 +361,33 @@ int rename_file(char *from, char *to)
        return 0;
 }
+int rename2_file(char *from, char *to, unsigned int flags)
+{
+        int err;
+#ifndef SYS_renameat2
+#  ifdef __x86_64__
+#    define SYS_renameat2 316
+#  endif
+#  ifdef __i386__
+#    define SYS_renameat2 353
+#  endif
+#endif
+#ifdef SYS_renameat2
+        err = syscall(SYS_renameat2, AT_FDCWD, from, AT_FDCWD, to, flags);
+        if (err < 0) {
+                if (errno != ENOSYS)
+                        return -errno;
+                else
+                        return -EINVAL;
+        }
+        return 0;
+#else
+        return -EINVAL;
+#endif
+}
 int do_statfs(char *root, long *bsize_out, long long *blocks_out,
              long long *bfree_out, long long *bavail_out,
              long long *files_out, long long *ffree_out,
diff --git a/fs/internal.h b/fs/internal.h
index 465742407466..e325b4f9c799 100644
--- a/fs/internal.h
+++ b/fs/internal.h
@@ -131,7 +131,6 @@ extern long prune_dcache_sb(struct super_block *sb, unsigned long nr_to_scan,
 /*
 * read_write.c
 */
-extern ssize_t __kernel_write(struct file *, const char *, size_t, loff_t *);
 extern int rw_verify_area(int, struct file *, const loff_t *, size_t);
 /*
@@ -144,3 +143,9 @@ extern long do_splice_direct(struct file *in, loff_t *ppos, struct file *out,
 * pipe.c
 */
 extern const struct file_operations pipefifo_fops;
+/*
+ * fs_pin.c
+ */
+extern void sb_pin_kill(struct super_block *sb);
+extern void mnt_pin_kill(struct mount *m);
diff --git a/fs/mount.h b/fs/mount.h
index d55297f2fa05..6740a6215529 100644
--- a/fs/mount.h
+++ b/fs/mount.h
@@ -55,7 +55,7 @@ struct mount {
        int mnt_id;                     /* mount identifier */
        int mnt_group_id;               /* peer group identifier */
        int mnt_expiry_mark;            /* true if marked for expiry */
-        int mnt_pinned;
+        struct hlist_head mnt_pins;
        struct path mnt_ex_mountpoint;
 };
diff --git a/fs/namei.c b/fs/namei.c
index 9eb787e5c167..a996bb48dfab 100644
--- a/fs/namei.c
+++ b/fs/namei.c
@@ -1091,10 +1091,10 @@ int follow_down_one(struct path *path)
 }
 EXPORT_SYMBOL(follow_down_one);
-static inline bool managed_dentry_might_block(struct dentry *dentry)
+static inline int managed_dentry_rcu(struct dentry *dentry)
 {
-        return (dentry->d_flags & DCACHE_MANAGE_TRANSIT &&
+        return (dentry->d_flags & DCACHE_MANAGE_TRANSIT) ?
-                dentry->d_op->d_manage(dentry, true) < 0);
+                dentry->d_op->d_manage(dentry, true) : 0;
 }
 /*
@@ -1110,11 +1110,18 @@ static bool __follow_mount_rcu(struct nameidata *nd, struct path *path,
                 * Don't forget we might have a non-mountpoint managed dentry
                 * that wants to block transit.
                 */
-                if (unlikely(managed_dentry_might_block(path->dentry)))
+                switch (managed_dentry_rcu(path->dentry)) {
+                case -ECHILD:
+                default:
                        return false;
+                case -EISDIR:
+                        return true;
+                case 0:
+                        break;
+                }
                if (!d_mountpoint(path->dentry))
-                        return true;
+                        return !(path->dentry->d_flags & DCACHE_NEED_AUTOMOUNT);
                mounted = __lookup_mnt(path->mnt, path->dentry);
                if (!mounted)
@@ -1130,7 +1137,8 @@ static bool __follow_mount_rcu(struct nameidata *nd, struct path *path,
                 */
                *inode = path->dentry->d_inode;
        }
-        return read_seqretry(&mount_lock, nd->m_seq);
+        return read_seqretry(&mount_lock, nd->m_seq) &&
+                !(path->dentry->d_flags & DCACHE_NEED_AUTOMOUNT);
 }
 static int follow_dotdot_rcu(struct nameidata *nd)
@@ -1402,11 +1410,8 @@ static int lookup_fast(struct nameidata *nd,
                }
                path->mnt = mnt;
                path->dentry = dentry;
-                if (unlikely(!__follow_mount_rcu(nd, path, inode)))
+                if (likely(__follow_mount_rcu(nd, path, inode)))
-                        goto unlazy;
+                        return 0;
-                if (unlikely(path->dentry->d_flags & DCACHE_NEED_AUTOMOUNT))
-                        goto unlazy;
-                return 0;
 unlazy:
                if (unlazy_walk(nd, dentry))
                        return -ECHILD;
@@ -4019,7 +4024,7 @@ SYSCALL_DEFINE2(link, const char __user *, oldname, const char __user *, newname
 * The worst of all namespace operations - renaming directory. "Perverted"
 * doesn't even start to describe it. Somebody in UCB had a heck of a trip...
 * Problems:
- *      a) we can get into loop creation. Check is done in is_subdir().
+ *      a) we can get into loop creation.
 *      b) race potential - two innocent renames can create a loop together.
 *         That's where 4.4 screws up. Current fix: serialization on
 *         sb->s_vfs_rename_mutex. We might be more accurate, but that's another
@@ -4075,7 +4080,7 @@ int vfs_rename(struct inode *old_dir, struct dentry *old_dentry,
        if (error)
                return error;
-        if (!old_dir->i_op->rename)
+        if (!old_dir->i_op->rename && !old_dir->i_op->rename2)
                return -EPERM;
        if (flags && !old_dir->i_op->rename2)
@@ -4134,10 +4139,11 @@ int vfs_rename(struct inode *old_dir, struct dentry *old_dentry,
                if (error)
                        goto out;
        }
-        if (!flags) {
+        if (!old_dir->i_op->rename2) {
                error = old_dir->i_op->rename(old_dir, old_dentry,
                                              new_dir, new_dentry);
        } else {
+                WARN_ON(old_dir->i_op->rename != NULL);
                error = old_dir->i_op->rename2(old_dir, old_dentry,
                                               new_dir, new_dentry, flags);
        }
diff --git a/fs/namespace.c b/fs/namespace.c
index 0acabea58319..a01c7730e9af 100644
--- a/fs/namespace.c
+++ b/fs/namespace.c
@@ -16,7 +16,6 @@
 #include <linux/namei.h>
 #include <linux/security.h>
 #include <linux/idr.h>
-#include <linux/acct.h>         /* acct_auto_close_mnt */
 #include <linux/init.h>         /* init_rootfs */
 #include <linux/fs_struct.h>    /* get_fs_root et.al. */
 #include <linux/fsnotify.h>     /* fsnotify_vfsmount_delete */
@@ -779,6 +778,20 @@ static void attach_mnt(struct mount *mnt,
        list_add_tail(&mnt->mnt_child, &parent->mnt_mounts);
 }
+static void attach_shadowed(struct mount *mnt,
+                        struct mount *parent,
+                        struct mount *shadows)
+{
+        if (shadows) {
+                hlist_add_behind_rcu(&mnt->mnt_hash, &shadows->mnt_hash);
+                list_add(&mnt->mnt_child, &shadows->mnt_child);
+        } else {
+                hlist_add_head_rcu(&mnt->mnt_hash,
+                                m_hash(&parent->mnt, mnt->mnt_mountpoint));
+                list_add_tail(&mnt->mnt_child, &parent->mnt_mounts);
+        }
+}
 /*
 * vfsmount lock must be held for write
 */
@@ -797,12 +810,7 @@ static void commit_tree(struct mount *mnt, struct mount *shadows)
        list_splice(&head, n->list.prev);
-        if (shadows)
+        attach_shadowed(mnt, parent, shadows);
-                hlist_add_behind_rcu(&mnt->mnt_hash, &shadows->mnt_hash);
-        else
-                hlist_add_head_rcu(&mnt->mnt_hash,
-                                m_hash(&parent->mnt, mnt->mnt_mountpoint));
-        list_add_tail(&mnt->mnt_child, &parent->mnt_mounts);
        touch_mnt_namespace(n);
 }
@@ -951,7 +959,6 @@ static struct mount *clone_mnt(struct mount *old, struct dentry *root,
 static void mntput_no_expire(struct mount *mnt)
 {
-put_again:
        rcu_read_lock();
        mnt_add_count(mnt, -1);
        if (likely(mnt->mnt_ns)) { /* shouldn't be the last one */
@@ -964,14 +971,6 @@ put_again:
                unlock_mount_hash();
                return;
        }
-        if (unlikely(mnt->mnt_pinned)) {
-                mnt_add_count(mnt, mnt->mnt_pinned + 1);
-                mnt->mnt_pinned = 0;
-                rcu_read_unlock();
-                unlock_mount_hash();
-                acct_auto_close_mnt(&mnt->mnt);
-                goto put_again;
-        }
        if (unlikely(mnt->mnt.mnt_flags & MNT_DOOMED)) {
                rcu_read_unlock();
                unlock_mount_hash();
@@ -994,6 +993,8 @@ put_again:
         * so mnt_get_writers() below is safe.
         */
        WARN_ON(mnt_get_writers(mnt));
+        if (unlikely(mnt->mnt_pins.first))
+                mnt_pin_kill(mnt);
        fsnotify_vfsmount_delete(&mnt->mnt);
        dput(mnt->mnt.mnt_root);
        deactivate_super(mnt->mnt.mnt_sb);
@@ -1021,25 +1022,15 @@ struct vfsmount *mntget(struct vfsmount *mnt)
 }
 EXPORT_SYMBOL(mntget);
-void mnt_pin(struct vfsmount *mnt)
+struct vfsmount *mnt_clone_internal(struct path *path)
-{
-        lock_mount_hash();
-        real_mount(mnt)->mnt_pinned++;
-        unlock_mount_hash();
-}
-EXPORT_SYMBOL(mnt_pin);
-void mnt_unpin(struct vfsmount *m)
 {
-        struct mount *mnt = real_mount(m);
+        struct mount *p;
-        lock_mount_hash();
+        p = clone_mnt(real_mount(path->mnt), path->dentry, CL_PRIVATE);
-        if (mnt->mnt_pinned) {
+        if (IS_ERR(p))
-                mnt_add_count(mnt, 1);
+                return ERR_CAST(p);
-                mnt->mnt_pinned--;
+        p->mnt.mnt_flags |= MNT_INTERNAL;
-        }
+        return &p->mnt;
-        unlock_mount_hash();
 }
-EXPORT_SYMBOL(mnt_unpin);
 static inline void mangle(struct seq_file *m, const char *s)
 {
@@ -1505,6 +1496,7 @@ struct mount *copy_tree(struct mount *mnt, struct dentry *dentry,
                        continue;
                for (s = r; s; s = next_mnt(s, r)) {
+                        struct mount *t = NULL;
                        if (!(flag & CL_COPY_UNBINDABLE) &&
                            IS_MNT_UNBINDABLE(s)) {
                                s = skip_mnt_tree(s);
@@ -1526,7 +1518,14 @@ struct mount *copy_tree(struct mount *mnt, struct dentry *dentry,
                                goto out;
                        lock_mount_hash();
                        list_add_tail(&q->mnt_list, &res->mnt_list);
-                        attach_mnt(q, parent, p->mnt_mp);
+                        mnt_set_mountpoint(parent, p->mnt_mp, q);
+                        if (!list_empty(&parent->mnt_mounts)) {
+                                t = list_last_entry(&parent->mnt_mounts,
+                                        struct mount, mnt_child);
+                                if (t->mnt_mp != p->mnt_mp)
+                                        t = NULL;
+                        }
+                        attach_shadowed(q, parent, t);
                        unlock_mount_hash();
                }
        }
diff --git a/fs/nfs/blocklayout/blocklayout.c b/fs/nfs/blocklayout/blocklayout.c
index 9b431f44fad9..cbb1797149d5 100644
--- a/fs/nfs/blocklayout/blocklayout.c
+++ b/fs/nfs/blocklayout/blocklayout.c
@@ -210,8 +210,7 @@ static void bl_end_io_read(struct bio *bio, int err)
                        SetPageUptodate(bvec->bv_page);
        if (err) {
-                struct nfs_pgio_data *rdata = par->data;
+                struct nfs_pgio_header *header = par->data;
-                struct nfs_pgio_header *header = rdata->header;
                if (!header->pnfs_error)
                        header->pnfs_error = -EIO;
@@ -224,43 +223,44 @@ static void bl_end_io_read(struct bio *bio, int err)
 static void bl_read_cleanup(struct work_struct *work)
 {
        struct rpc_task *task;
-        struct nfs_pgio_data *rdata;
+        struct nfs_pgio_header *hdr;
        dprintk("%s enter\n", __func__);
        task = container_of(work, struct rpc_task, u.tk_work);
-        rdata = container_of(task, struct nfs_pgio_data, task);
+        hdr = container_of(task, struct nfs_pgio_header, task);
-        pnfs_ld_read_done(rdata);
+        pnfs_ld_read_done(hdr);
 }
 static void
 bl_end_par_io_read(void *data, int unused)
 {
-        struct nfs_pgio_data *rdata = data;
+        struct nfs_pgio_header *hdr = data;
-        rdata->task.tk_status = rdata->header->pnfs_error;
+        hdr->task.tk_status = hdr->pnfs_error;
-        INIT_WORK(&rdata->task.u.tk_work, bl_read_cleanup);
+        INIT_WORK(&hdr->task.u.tk_work, bl_read_cleanup);
-        schedule_work(&rdata->task.u.tk_work);
+        schedule_work(&hdr->task.u.tk_work);
 }
 static enum pnfs_try_status
-bl_read_pagelist(struct nfs_pgio_data *rdata)
+bl_read_pagelist(struct nfs_pgio_header *hdr)
 {
-        struct nfs_pgio_header *header = rdata->header;
+        struct nfs_pgio_header *header = hdr;
        int i, hole;
        struct bio *bio = NULL;
        struct pnfs_block_extent *be = NULL, *cow_read = NULL;
        sector_t isect, extent_length = 0;
        struct parallel_io *par;
-        loff_t f_offset = rdata->args.offset;
+        loff_t f_offset = hdr->args.offset;
-        size_t bytes_left = rdata->args.count;
+        size_t bytes_left = hdr->args.count;
        unsigned int pg_offset, pg_len;
-        struct page **pages = rdata->args.pages;
+        struct page **pages = hdr->args.pages;
-        int pg_index = rdata->args.pgbase >> PAGE_CACHE_SHIFT;
+        int pg_index = hdr->args.pgbase >> PAGE_CACHE_SHIFT;
        const bool is_dio = (header->dreq != NULL);
        dprintk("%s enter nr_pages %u offset %lld count %u\n", __func__,
-               rdata->pages.npages, f_offset, (unsigned int)rdata->args.count);
+                hdr->page_array.npages, f_offset,
+                (unsigned int)hdr->args.count);
-        par = alloc_parallel(rdata);
+        par = alloc_parallel(hdr);
        if (!par)
                goto use_mds;
        par->pnfs_callback = bl_end_par_io_read;
@@ -268,7 +268,7 @@ bl_read_pagelist(struct nfs_pgio_data *rdata)
        isect = (sector_t) (f_offset >> SECTOR_SHIFT);
        /* Code assumes extents are page-aligned */
-        for (i = pg_index; i < rdata->pages.npages; i++) {
+        for (i = pg_index; i < hdr->page_array.npages; i++) {
                if (!extent_length) {
                        /* We've used up the previous extent */
                        bl_put_extent(be);
@@ -317,7 +317,8 @@ bl_read_pagelist(struct nfs_pgio_data *rdata)
                        struct pnfs_block_extent *be_read;
                        be_read = (hole && cow_read) ? cow_read : be;
-                        bio = do_add_page_to_bio(bio, rdata->pages.npages - i,
+                        bio = do_add_page_to_bio(bio,
+                                                 hdr->page_array.npages - i,
                                                 READ,
                                                 isect, pages[i], be_read,
                                                 bl_end_io_read, par,
@@ -332,10 +333,10 @@ bl_read_pagelist(struct nfs_pgio_data *rdata)
                extent_length -= PAGE_CACHE_SECTORS;
        }
        if ((isect << SECTOR_SHIFT) >= header->inode->i_size) {
-                rdata->res.eof = 1;
+                hdr->res.eof = 1;
-                rdata->res.count = header->inode->i_size - rdata->args.offset;
+                hdr->res.count = header->inode->i_size - hdr->args.offset;
        } else {
-                rdata->res.count = (isect << SECTOR_SHIFT) - rdata->args.offset;
+                hdr->res.count = (isect << SECTOR_SHIFT) - hdr->args.offset;
        }
 out:
        bl_put_extent(be);
@@ -390,8 +391,7 @@ static void bl_end_io_write_zero(struct bio *bio, int err)
        }
        if (unlikely(err)) {
-                struct nfs_pgio_data *data = par->data;
+                struct nfs_pgio_header *header = par->data;
-                struct nfs_pgio_header *header = data->header;
                if (!header->pnfs_error)
                        header->pnfs_error = -EIO;
@@ -405,8 +405,7 @@ static void bl_end_io_write(struct bio *bio, int err)
 {
        struct parallel_io *par = bio->bi_private;
        const int uptodate = test_bit(BIO_UPTODATE, &bio->bi_flags);
-        struct nfs_pgio_data *data = par->data;
+        struct nfs_pgio_header *header = par->data;
-        struct nfs_pgio_header *header = data->header;
        if (!uptodate) {
                if (!header->pnfs_error)
@@ -423,32 +422,32 @@ static void bl_end_io_write(struct bio *bio, int err)
 static void bl_write_cleanup(struct work_struct *work)
 {
        struct rpc_task *task;
-        struct nfs_pgio_data *wdata;
+        struct nfs_pgio_header *hdr;
        dprintk("%s enter\n", __func__);
        task = container_of(work, struct rpc_task, u.tk_work);
-        wdata = container_of(task, struct nfs_pgio_data, task);
+        hdr = container_of(task, struct nfs_pgio_header, task);
-        if (likely(!wdata->header->pnfs_error)) {
+        if (likely(!hdr->pnfs_error)) {
                /* Marks for LAYOUTCOMMIT */
-                mark_extents_written(BLK_LSEG2EXT(wdata->header->lseg),
+                mark_extents_written(BLK_LSEG2EXT(hdr->lseg),
-                                     wdata->args.offset, wdata->args.count);
+                                     hdr->args.offset, hdr->args.count);
        }
-        pnfs_ld_write_done(wdata);
+        pnfs_ld_write_done(hdr);
 }
 /* Called when last of bios associated with a bl_write_pagelist call finishes */
 static void bl_end_par_io_write(void *data, int num_se)
 {
-        struct nfs_pgio_data *wdata = data;
+        struct nfs_pgio_header *hdr = data;
-        if (unlikely(wdata->header->pnfs_error)) {
+        if (unlikely(hdr->pnfs_error)) {
-                bl_free_short_extents(&BLK_LSEG2EXT(wdata->header->lseg)->bl_inval,
+                bl_free_short_extents(&BLK_LSEG2EXT(hdr->lseg)->bl_inval,
                                        num_se);
        }
-        wdata->task.tk_status = wdata->header->pnfs_error;
+        hdr->task.tk_status = hdr->pnfs_error;
-        wdata->verf.committed = NFS_FILE_SYNC;
+        hdr->verf.committed = NFS_FILE_SYNC;
-        INIT_WORK(&wdata->task.u.tk_work, bl_write_cleanup);
+        INIT_WORK(&hdr->task.u.tk_work, bl_write_cleanup);
-        schedule_work(&wdata->task.u.tk_work);
+        schedule_work(&hdr->task.u.tk_work);
 }
 /* FIXME STUB - mark intersection of layout and page as bad, so is not
@@ -673,18 +672,17 @@ check_page:
 }
 static enum pnfs_try_status
-bl_write_pagelist(struct nfs_pgio_data *wdata, int sync)
+bl_write_pagelist(struct nfs_pgio_header *header, int sync)
 {
-        struct nfs_pgio_header *header = wdata->header;
        int i, ret, npg_zero, pg_index, last = 0;
        struct bio *bio = NULL;
        struct pnfs_block_extent *be = NULL, *cow_read = NULL;
        sector_t isect, last_isect = 0, extent_length = 0;
        struct parallel_io *par = NULL;
-        loff_t offset = wdata->args.offset;
+        loff_t offset = header->args.offset;
-        size_t count = wdata->args.count;
+        size_t count = header->args.count;
        unsigned int pg_offset, pg_len, saved_len;
-        struct page **pages = wdata->args.pages;
+        struct page **pages = header->args.pages;
        struct page *page;
        pgoff_t index;
        u64 temp;
@@ -699,11 +697,11 @@ bl_write_pagelist(struct nfs_pgio_data *wdata, int sync)
                dprintk("pnfsblock nonblock aligned DIO writes. Resend MDS\n");
                goto out_mds;
        }
-        /* At this point, wdata->pages is a (sequential) list of nfs_pages.
+        /* At this point, header->page_aray is a (sequential) list of nfs_pages.
         * We want to write each, and if there is an error set pnfs_error
         * to have it redone using nfs.
         */
-        par = alloc_parallel(wdata);
+        par = alloc_parallel(header);
        if (!par)
                goto out_mds;
        par->pnfs_callback = bl_end_par_io_write;
@@ -790,8 +788,8 @@ next_page:
        bio = bl_submit_bio(WRITE, bio);
        /* Middle pages */
-        pg_index = wdata->args.pgbase >> PAGE_CACHE_SHIFT;
+        pg_index = header->args.pgbase >> PAGE_CACHE_SHIFT;
-        for (i = pg_index; i < wdata->pages.npages; i++) {
+        for (i = pg_index; i < header->page_array.npages; i++) {
                if (!extent_length) {
                        /* We've used up the previous extent */
                        bl_put_extent(be);
@@ -862,7 +860,8 @@ next_page:
                }
-                bio = do_add_page_to_bio(bio, wdata->pages.npages - i, WRITE,
+                bio = do_add_page_to_bio(bio, header->page_array.npages - i,
+                                         WRITE,
                                         isect, pages[i], be,
                                         bl_end_io_write, par,
                                         pg_offset, pg_len);
@@ -890,7 +889,7 @@ next_page:
        }
 write_done:
-        wdata->res.count = wdata->args.count;
+        header->res.count = header->args.count;
 out:
        bl_put_extent(be);
        bl_put_extent(cow_read);
@@ -1063,7 +1062,7 @@ nfs4_blk_get_deviceinfo(struct nfs_server *server, const struct nfs_fh *fh,
                return ERR_PTR(-ENOMEM);
        }
-        pages = kzalloc(max_pages * sizeof(struct page *), GFP_NOFS);
+        pages = kcalloc(max_pages, sizeof(struct page *), GFP_NOFS);
        if (pages == NULL) {
                kfree(dev);
                return ERR_PTR(-ENOMEM);
diff --git a/fs/nfs/callback.c b/fs/nfs/callback.c
index 073b4cf67ed9..54de482143cc 100644
--- a/fs/nfs/callback.c
+++ b/fs/nfs/callback.c
@@ -428,6 +428,18 @@ check_gss_callback_principal(struct nfs_client *clp, struct svc_rqst *rqstp)
        if (p == NULL)
                return 0;
+        /*
+         * Did we get the acceptor from userland during the SETCLIENID
+         * negotiation?
+         */
+        if (clp->cl_acceptor)
+                return !strcmp(p, clp->cl_acceptor);
+        /*
+         * Otherwise try to verify it using the cl_hostname. Note that this
+         * doesn't work if a non-canonical hostname was used in the devname.
+         */
        /* Expect a GSS_C_NT_HOSTBASED_NAME like "nfs@serverhostname" */
        if (memcmp(p, "nfs@", 4) != 0)
diff --git a/fs/nfs/client.c b/fs/nfs/client.c
index 180d1ec9c32e..1c5ff6d58385 100644
--- a/fs/nfs/client.c
+++ b/fs/nfs/client.c
@@ -110,8 +110,8 @@ struct nfs_subversion *get_nfs_version(unsigned int version)
                mutex_unlock(&nfs_version_mutex);
        }
-        if (!IS_ERR(nfs))
+        if (!IS_ERR(nfs) && !try_module_get(nfs->owner))
-                try_module_get(nfs->owner);
+                return ERR_PTR(-EAGAIN);
        return nfs;
 }
@@ -158,7 +158,8 @@ struct nfs_client *nfs_alloc_client(const struct nfs_client_initdata *cl_init)
                goto error_0;
        clp->cl_nfs_mod = cl_init->nfs_mod;
-        try_module_get(clp->cl_nfs_mod->owner);
+        if (!try_module_get(clp->cl_nfs_mod->owner))
+                goto error_dealloc;
        clp->rpc_ops = clp->cl_nfs_mod->rpc_ops;
@@ -190,6 +191,7 @@ struct nfs_client *nfs_alloc_client(const struct nfs_client_initdata *cl_init)
 error_cleanup:
        put_nfs_version(clp->cl_nfs_mod);
+error_dealloc:
        kfree(clp);
 error_0:
        return ERR_PTR(err);
@@ -252,6 +254,7 @@ void nfs_free_client(struct nfs_client *clp)
        put_net(clp->cl_net);
        put_nfs_version(clp->cl_nfs_mod);
        kfree(clp->cl_hostname);
+        kfree(clp->cl_acceptor);
        kfree(clp);
        dprintk("<-- nfs_free_client()\n");
@@ -482,8 +485,13 @@ nfs_get_client(const struct nfs_client_initdata *cl_init,
        struct nfs_net *nn = net_generic(cl_init->net, nfs_net_id);
        const struct nfs_rpc_ops *rpc_ops = cl_init->nfs_mod->rpc_ops;
+        if (cl_init->hostname == NULL) {
+                WARN_ON(1);
+                return NULL;
+        }
        dprintk("--> nfs_get_client(%s,v%u)\n",
-                cl_init->hostname ?: "", rpc_ops->version);
+                cl_init->hostname, rpc_ops->version);
        /* see if the client already exists */
        do {
@@ -510,7 +518,7 @@ nfs_get_client(const struct nfs_client_initdata *cl_init,
        } while (!IS_ERR(new));
        dprintk("<-- nfs_get_client() Failed to find %s (%ld)\n",
-                cl_init->hostname ?: "", PTR_ERR(new));
+                cl_init->hostname, PTR_ERR(new));
        return new;
 }
 EXPORT_SYMBOL_GPL(nfs_get_client);
diff --git a/fs/nfs/delegation.c b/fs/nfs/delegation.c
index 5d8ccecf5f5c..5853f53db732 100644
--- a/fs/nfs/delegation.c
+++ b/fs/nfs/delegation.c
@@ -41,14 +41,8 @@ void nfs_mark_delegation_referenced(struct nfs_delegation *delegation)
        set_bit(NFS_DELEGATION_REFERENCED, &delegation->flags);
 }
-/**
+static int
- * nfs_have_delegation - check if inode has a delegation
+nfs4_do_check_delegation(struct inode *inode, fmode_t flags, bool mark)
- * @inode: inode to check
- * @flags: delegation types to check for
- *
- * Returns one if inode has the indicated delegation, otherwise zero.
- */
-int nfs4_have_delegation(struct inode *inode, fmode_t flags)
 {
        struct nfs_delegation *delegation;
        int ret = 0;
@@ -58,12 +52,34 @@ int nfs4_have_delegation(struct inode *inode, fmode_t flags)
        delegation = rcu_dereference(NFS_I(inode)->delegation);
        if (delegation != NULL && (delegation->type & flags) == flags &&
            !test_bit(NFS_DELEGATION_RETURNING, &delegation->flags)) {
-                nfs_mark_delegation_referenced(delegation);
+                if (mark)
+                        nfs_mark_delegation_referenced(delegation);
                ret = 1;
        }
        rcu_read_unlock();
        return ret;
 }
+/**
+ * nfs_have_delegation - check if inode has a delegation, mark it
+ * NFS_DELEGATION_REFERENCED if there is one.
+ * @inode: inode to check
+ * @flags: delegation types to check for
+ *
+ * Returns one if inode has the indicated delegation, otherwise zero.
+ */
+int nfs4_have_delegation(struct inode *inode, fmode_t flags)
+{
+        return nfs4_do_check_delegation(inode, flags, true);
+}
+/*
+ * nfs4_check_delegation - check if inode has a delegation, do not mark
+ * NFS_DELEGATION_REFERENCED if it has one.
+ */
+int nfs4_check_delegation(struct inode *inode, fmode_t flags)
+{
+        return nfs4_do_check_delegation(inode, flags, false);
+}
 static int nfs_delegation_claim_locks(struct nfs_open_context *ctx, struct nfs4_state *state, const nfs4_stateid *stateid)
 {
diff --git a/fs/nfs/delegation.h b/fs/nfs/delegation.h
index 9a79c7a99d6d..5c1cce39297f 100644
--- a/fs/nfs/delegation.h
+++ b/fs/nfs/delegation.h
@@ -59,6 +59,7 @@ bool nfs4_copy_delegation_stateid(nfs4_stateid *dst, struct inode *inode, fmode_
 void nfs_mark_delegation_referenced(struct nfs_delegation *delegation);
 int nfs4_have_delegation(struct inode *inode, fmode_t flags);
+int nfs4_check_delegation(struct inode *inode, fmode_t flags);
 #endif
diff --git a/fs/nfs/dir.c b/fs/nfs/dir.c
index 4a3d4ef76127..36d921f0c602 100644
--- a/fs/nfs/dir.c
+++ b/fs/nfs/dir.c
@@ -988,9 +988,13 @@ EXPORT_SYMBOL_GPL(nfs_force_lookup_revalidate);
 * A check for whether or not the parent directory has changed.
 * In the case it has, we assume that the dentries are untrustworthy
 * and may need to be looked up again.
+ * If rcu_walk prevents us from performing a full check, return 0.
 */
-static int nfs_check_verifier(struct inode *dir, struct dentry *dentry)
+static int nfs_check_verifier(struct inode *dir, struct dentry *dentry,
+                              int rcu_walk)
 {
+        int ret;
        if (IS_ROOT(dentry))
                return 1;
        if (NFS_SERVER(dir)->flags & NFS_MOUNT_LOOKUP_CACHE_NONE)
@@ -998,7 +1002,11 @@ static int nfs_check_verifier(struct inode *dir, struct dentry *dentry)
        if (!nfs_verify_change_attribute(dir, dentry->d_time))
                return 0;
        /* Revalidate nfsi->cache_change_attribute before we declare a match */
-        if (nfs_revalidate_inode(NFS_SERVER(dir), dir) < 0)
+        if (rcu_walk)
+                ret = nfs_revalidate_inode_rcu(NFS_SERVER(dir), dir);
+        else
+                ret = nfs_revalidate_inode(NFS_SERVER(dir), dir);
+        if (ret < 0)
                return 0;
        if (!nfs_verify_change_attribute(dir, dentry->d_time))
                return 0;
@@ -1042,6 +1050,8 @@ int nfs_lookup_verify_inode(struct inode *inode, unsigned int flags)
 out:
        return (inode->i_nlink == 0) ? -ENOENT : 0;
 out_force:
+        if (flags & LOOKUP_RCU)
+                return -ECHILD;
        ret = __nfs_revalidate_inode(server, inode);
        if (ret != 0)
                return ret;
@@ -1054,6 +1064,9 @@ out_force:
 *
 * If parent mtime has changed, we revalidate, else we wait for a
 * period corresponding to the parent's attribute cache timeout value.
+ *
+ * If LOOKUP_RCU prevents us from performing a full check, return 1
+ * suggesting a reval is needed.
 */
 static inline
 int nfs_neg_need_reval(struct inode *dir, struct dentry *dentry,
@@ -1064,7 +1077,7 @@ int nfs_neg_need_reval(struct inode *dir, struct dentry *dentry,
                return 0;
        if (NFS_SERVER(dir)->flags & NFS_MOUNT_LOOKUP_CACHE_NONEG)
                return 1;
-        return !nfs_check_verifier(dir, dentry);
+        return !nfs_check_verifier(dir, dentry, flags & LOOKUP_RCU);
 }
 /*
@@ -1088,21 +1101,30 @@ static int nfs_lookup_revalidate(struct dentry *dentry, unsigned int flags)
        struct nfs4_label *label = NULL;
        int error;
-        if (flags & LOOKUP_RCU)
+        if (flags & LOOKUP_RCU) {
-                return -ECHILD;
+                parent = ACCESS_ONCE(dentry->d_parent);
+                dir = ACCESS_ONCE(parent->d_inode);
-        parent = dget_parent(dentry);
+                if (!dir)
-        dir = parent->d_inode;
+                        return -ECHILD;
+        } else {
+                parent = dget_parent(dentry);
+                dir = parent->d_inode;
+        }
        nfs_inc_stats(dir, NFSIOS_DENTRYREVALIDATE);
        inode = dentry->d_inode;
        if (!inode) {
-                if (nfs_neg_need_reval(dir, dentry, flags))
+                if (nfs_neg_need_reval(dir, dentry, flags)) {
+                        if (flags & LOOKUP_RCU)
+                                return -ECHILD;
                        goto out_bad;
+                }
                goto out_valid_noent;
        }
        if (is_bad_inode(inode)) {
+                if (flags & LOOKUP_RCU)
+                        return -ECHILD;
                dfprintk(LOOKUPCACHE, "%s: %pd2 has dud inode\n",
                                __func__, dentry);
                goto out_bad;
@@ -1112,12 +1134,20 @@ static int nfs_lookup_revalidate(struct dentry *dentry, unsigned int flags)
                goto out_set_verifier;
        /* Force a full look up iff the parent directory has changed */
-        if (!nfs_is_exclusive_create(dir, flags) && nfs_check_verifier(dir, dentry)) {
+        if (!nfs_is_exclusive_create(dir, flags) &&
-                if (nfs_lookup_verify_inode(inode, flags))
+            nfs_check_verifier(dir, dentry, flags & LOOKUP_RCU)) {
+                if (nfs_lookup_verify_inode(inode, flags)) {
+                        if (flags & LOOKUP_RCU)
+                                return -ECHILD;
                        goto out_zap_parent;
+                }
                goto out_valid;
        }
+        if (flags & LOOKUP_RCU)
+                return -ECHILD;
        if (NFS_STALE(inode))
                goto out_bad;
@@ -1153,13 +1183,18 @@ out_set_verifier:
        /* Success: notify readdir to use READDIRPLUS */
        nfs_advise_use_readdirplus(dir);
 out_valid_noent:
-        dput(parent);
+        if (flags & LOOKUP_RCU) {
+                if (parent != ACCESS_ONCE(dentry->d_parent))
+                        return -ECHILD;
+        } else
+                dput(parent);
        dfprintk(LOOKUPCACHE, "NFS: %s(%pd2) is valid\n",
                        __func__, dentry);
        return 1;
 out_zap_parent:
        nfs_zap_caches(dir);
 out_bad:
+        WARN_ON(flags & LOOKUP_RCU);
        nfs_free_fattr(fattr);
        nfs_free_fhandle(fhandle);
        nfs4_label_free(label);
@@ -1185,6 +1220,7 @@ out_zap_parent:
                        __func__, dentry);
        return 0;
 out_error:
+        WARN_ON(flags & LOOKUP_RCU);
        nfs_free_fattr(fattr);
        nfs_free_fhandle(fhandle);
        nfs4_label_free(label);
@@ -1529,14 +1565,9 @@ EXPORT_SYMBOL_GPL(nfs_atomic_open);
 static int nfs4_lookup_revalidate(struct dentry *dentry, unsigned int flags)
 {
-        struct dentry *parent = NULL;
        struct inode *inode;
-        struct inode *dir;
        int ret = 0;
-        if (flags & LOOKUP_RCU)
-                return -ECHILD;
        if (!(flags & LOOKUP_OPEN) || (flags & LOOKUP_DIRECTORY))
                goto no_open;
        if (d_mountpoint(dentry))
@@ -1545,34 +1576,47 @@ static int nfs4_lookup_revalidate(struct dentry *dentry, unsigned int flags)
                goto no_open;
        inode = dentry->d_inode;
-        parent = dget_parent(dentry);
-        dir = parent->d_inode;
        /* We can't create new files in nfs_open_revalidate(), so we
         * optimize away revalidation of negative dentries.
         */
        if (inode == NULL) {
+                struct dentry *parent;
+                struct inode *dir;
+                if (flags & LOOKUP_RCU) {
+                        parent = ACCESS_ONCE(dentry->d_parent);
+                        dir = ACCESS_ONCE(parent->d_inode);
+                        if (!dir)
+                                return -ECHILD;
+                } else {
+                        parent = dget_parent(dentry);
+                        dir = parent->d_inode;
+                }
                if (!nfs_neg_need_reval(dir, dentry, flags))
                        ret = 1;
+                else if (flags & LOOKUP_RCU)
+                        ret = -ECHILD;
+                if (!(flags & LOOKUP_RCU))
+                        dput(parent);
+                else if (parent != ACCESS_ONCE(dentry->d_parent))
+                        return -ECHILD;
                goto out;
        }
        /* NFS only supports OPEN on regular files */
        if (!S_ISREG(inode->i_mode))
-                goto no_open_dput;
+                goto no_open;
        /* We cannot do exclusive creation on a positive dentry */
        if (flags & LOOKUP_EXCL)
-                goto no_open_dput;
+                goto no_open;
        /* Let f_op->open() actually open (and revalidate) the file */
        ret = 1;
 out:
-        dput(parent);
        return ret;
-no_open_dput:
-        dput(parent);
 no_open:
        return nfs_lookup_revalidate(dentry, flags);
 }
@@ -2028,10 +2072,14 @@ static DEFINE_SPINLOCK(nfs_access_lru_lock);
 static LIST_HEAD(nfs_access_lru_list);
 static atomic_long_t nfs_access_nr_entries;
+static unsigned long nfs_access_max_cachesize = ULONG_MAX;
+module_param(nfs_access_max_cachesize, ulong, 0644);
+MODULE_PARM_DESC(nfs_access_max_cachesize, "NFS access maximum total cache length");
 static void nfs_access_free_entry(struct nfs_access_entry *entry)
 {
        put_rpccred(entry->cred);
-        kfree(entry);
+        kfree_rcu(entry, rcu_head);
        smp_mb__before_atomic();
        atomic_long_dec(&nfs_access_nr_entries);
        smp_mb__after_atomic();
@@ -2048,19 +2096,14 @@ static void nfs_access_free_list(struct list_head *head)
        }
 }
-unsigned long
+static unsigned long
-nfs_access_cache_scan(struct shrinker *shrink, struct shrink_control *sc)
+nfs_do_access_cache_scan(unsigned int nr_to_scan)
 {
        LIST_HEAD(head);
        struct nfs_inode *nfsi, *next;
        struct nfs_access_entry *cache;
-        int nr_to_scan = sc->nr_to_scan;
-        gfp_t gfp_mask = sc->gfp_mask;
        long freed = 0;
-        if ((gfp_mask & GFP_KERNEL) != GFP_KERNEL)
-                return SHRINK_STOP;
        spin_lock(&nfs_access_lru_lock);
        list_for_each_entry_safe(nfsi, next, &nfs_access_lru_list, access_cache_inode_lru) {
                struct inode *inode;
@@ -2094,11 +2137,39 @@ remove_lru_entry:
 }
 unsigned long
+nfs_access_cache_scan(struct shrinker *shrink, struct shrink_control *sc)
+{
+        int nr_to_scan = sc->nr_to_scan;
+        gfp_t gfp_mask = sc->gfp_mask;
+        if ((gfp_mask & GFP_KERNEL) != GFP_KERNEL)
+                return SHRINK_STOP;
+        return nfs_do_access_cache_scan(nr_to_scan);
+}
+unsigned long
 nfs_access_cache_count(struct shrinker *shrink, struct shrink_control *sc)
 {
        return vfs_pressure_ratio(atomic_long_read(&nfs_access_nr_entries));
 }
+static void
+nfs_access_cache_enforce_limit(void)
+{
+        long nr_entries = atomic_long_read(&nfs_access_nr_entries);
+        unsigned long diff;
+        unsigned int nr_to_scan;
+        if (nr_entries < 0 || nr_entries <= nfs_access_max_cachesize)
+                return;
+        nr_to_scan = 100;
+        diff = nr_entries - nfs_access_max_cachesize;
+        if (diff < nr_to_scan)
+                nr_to_scan = diff;
+        nfs_do_access_cache_scan(nr_to_scan);
+}
 static void __nfs_access_zap_cache(struct nfs_inode *nfsi, struct list_head *head)
 {
        struct rb_root *root_node = &nfsi->access_cache;
@@ -2186,6 +2257,38 @@ out_zap:
        return -ENOENT;
 }
+static int nfs_access_get_cached_rcu(struct inode *inode, struct rpc_cred *cred, struct nfs_access_entry *res)
+{
+        /* Only check the most recently returned cache entry,
+         * but do it without locking.
+         */
+        struct nfs_inode *nfsi = NFS_I(inode);
+        struct nfs_access_entry *cache;
+        int err = -ECHILD;
+        struct list_head *lh;
+        rcu_read_lock();
+        if (nfsi->cache_validity & NFS_INO_INVALID_ACCESS)
+                goto out;
+        lh = rcu_dereference(nfsi->access_cache_entry_lru.prev);
+        cache = list_entry(lh, struct nfs_access_entry, lru);
+        if (lh == &nfsi->access_cache_entry_lru ||
+            cred != cache->cred)
+                cache = NULL;
+        if (cache == NULL)
+                goto out;
+        if (!nfs_have_delegated_attributes(inode) &&
+            !time_in_range_open(jiffies, cache->jiffies, cache->jiffies + nfsi->attrtimeo))
+                goto out;
+        res->jiffies = cache->jiffies;
+        res->cred = cache->cred;
+        res->mask = cache->mask;
+        err = 0;
+out:
+        rcu_read_unlock();
+        return err;
+}
 static void nfs_access_add_rbtree(struct inode *inode, struct nfs_access_entry *set)
 {
        struct nfs_inode *nfsi = NFS_I(inode);
@@ -2229,6 +2332,11 @@ void nfs_access_add_cache(struct inode *inode, struct nfs_access_entry *set)
        cache->cred = get_rpccred(set->cred);
        cache->mask = set->mask;
+        /* The above field assignments must be visible
+         * before this item appears on the lru.  We cannot easily
+         * use rcu_assign_pointer, so just force the memory barrier.
+         */
+        smp_wmb();
        nfs_access_add_rbtree(inode, cache);
        /* Update accounting */
@@ -2244,6 +2352,7 @@ void nfs_access_add_cache(struct inode *inode, struct nfs_access_entry *set)
                                        &nfs_access_lru_list);
                spin_unlock(&nfs_access_lru_lock);
        }
+        nfs_access_cache_enforce_limit();
 }
 EXPORT_SYMBOL_GPL(nfs_access_add_cache);
@@ -2267,10 +2376,16 @@ static int nfs_do_access(struct inode *inode, struct rpc_cred *cred, int mask)
        trace_nfs_access_enter(inode);
-        status = nfs_access_get_cached(inode, cred, &cache);
+        status = nfs_access_get_cached_rcu(inode, cred, &cache);
+        if (status != 0)
+                status = nfs_access_get_cached(inode, cred, &cache);
        if (status == 0)
                goto out_cached;
+        status = -ECHILD;
+        if (mask & MAY_NOT_BLOCK)
+                goto out;
        /* Be clever: ask server to check for all possible rights */
        cache.mask = MAY_EXEC | MAY_WRITE | MAY_READ;
        cache.cred = cred;
@@ -2321,9 +2436,6 @@ int nfs_permission(struct inode *inode, int mask)
        struct rpc_cred *cred;
        int res = 0;
-        if (mask & MAY_NOT_BLOCK)
-                return -ECHILD;
        nfs_inc_stats(inode, NFSIOS_VFSACCESS);
        if ((mask & (MAY_READ | MAY_WRITE | MAY_EXEC)) == 0)
@@ -2350,12 +2462,23 @@ force_lookup:
        if (!NFS_PROTO(inode)->access)
                goto out_notsup;
-        cred = rpc_lookup_cred();
+        /* Always try fast lookups first */
-        if (!IS_ERR(cred)) {
+        rcu_read_lock();
-                res = nfs_do_access(inode, cred, mask);
+        cred = rpc_lookup_cred_nonblock();
-                put_rpccred(cred);
+        if (!IS_ERR(cred))
-        } else
+                res = nfs_do_access(inode, cred, mask|MAY_NOT_BLOCK);
+        else
                res = PTR_ERR(cred);
+        rcu_read_unlock();
+        if (res == -ECHILD && !(mask & MAY_NOT_BLOCK)) {
+                /* Fast lookup failed, try the slow way */
+                cred = rpc_lookup_cred();
+                if (!IS_ERR(cred)) {
+                        res = nfs_do_access(inode, cred, mask);
+                        put_rpccred(cred);
+                } else
+                        res = PTR_ERR(cred);
+        }
 out:
        if (!res && (mask & MAY_EXEC) && !execute_ok(inode))
                res = -EACCES;
@@ -2364,6 +2487,9 @@ out:
                inode->i_sb->s_id, inode->i_ino, mask, res);
        return res;
 out_notsup:
+        if (mask & MAY_NOT_BLOCK)
+                return -ECHILD;
        res = nfs_revalidate_inode(NFS_SERVER(inode), inode);
        if (res == 0)
                res = generic_permission(inode, mask);
diff --git a/fs/nfs/direct.c b/fs/nfs/direct.c
index f11b9eed0de1..65ef6e00deee 100644
--- a/fs/nfs/direct.c
+++ b/fs/nfs/direct.c
@@ -148,8 +148,8 @@ static void nfs_direct_set_hdr_verf(struct nfs_direct_req *dreq,
 {
        struct nfs_writeverf *verfp;
-        verfp = nfs_direct_select_verf(dreq, hdr->data->ds_clp,
+        verfp = nfs_direct_select_verf(dreq, hdr->ds_clp,
-                                      hdr->data->ds_idx);
+                                      hdr->ds_idx);
        WARN_ON_ONCE(verfp->committed >= 0);
        memcpy(verfp, &hdr->verf, sizeof(struct nfs_writeverf));
        WARN_ON_ONCE(verfp->committed < 0);
@@ -169,8 +169,8 @@ static int nfs_direct_set_or_cmp_hdr_verf(struct nfs_direct_req *dreq,
 {
        struct nfs_writeverf *verfp;
-        verfp = nfs_direct_select_verf(dreq, hdr->data->ds_clp,
+        verfp = nfs_direct_select_verf(dreq, hdr->ds_clp,
-                                         hdr->data->ds_idx);
+                                         hdr->ds_idx);
        if (verfp->committed < 0) {
                nfs_direct_set_hdr_verf(dreq, hdr);
                return 0;
@@ -715,7 +715,7 @@ static void nfs_direct_write_completion(struct nfs_pgio_header *hdr)
 {
        struct nfs_direct_req *dreq = hdr->dreq;
        struct nfs_commit_info cinfo;
-        int bit = -1;
+        bool request_commit = false;
        struct nfs_page *req = nfs_list_entry(hdr->pages.next);
        if (test_bit(NFS_IOHDR_REDO, &hdr->flags))
@@ -729,27 +729,20 @@ static void nfs_direct_write_completion(struct nfs_pgio_header *hdr)
                dreq->flags = 0;
                dreq->error = hdr->error;
        }
-        if (dreq->error != 0)
+        if (dreq->error == 0) {
-                bit = NFS_IOHDR_ERROR;
-        else {
                dreq->count += hdr->good_bytes;
-                if (test_bit(NFS_IOHDR_NEED_RESCHED, &hdr->flags)) {
+                if (nfs_write_need_commit(hdr)) {
-                        dreq->flags = NFS_ODIRECT_RESCHED_WRITES;
-                        bit = NFS_IOHDR_NEED_RESCHED;
-                } else if (test_bit(NFS_IOHDR_NEED_COMMIT, &hdr->flags)) {
                        if (dreq->flags == NFS_ODIRECT_RESCHED_WRITES)
-                                bit = NFS_IOHDR_NEED_RESCHED;
+                                request_commit = true;
                        else if (dreq->flags == 0) {
                                nfs_direct_set_hdr_verf(dreq, hdr);
-                                bit = NFS_IOHDR_NEED_COMMIT;
+                                request_commit = true;
                                dreq->flags = NFS_ODIRECT_DO_COMMIT;
                        } else if (dreq->flags == NFS_ODIRECT_DO_COMMIT) {
-                                if (nfs_direct_set_or_cmp_hdr_verf(dreq, hdr)) {
+                                request_commit = true;
+                                if (nfs_direct_set_or_cmp_hdr_verf(dreq, hdr))
                                        dreq->flags =
                                                NFS_ODIRECT_RESCHED_WRITES;
-                                        bit = NFS_IOHDR_NEED_RESCHED;
-                                } else
-                                        bit = NFS_IOHDR_NEED_COMMIT;
                        }
                }
        }
@@ -759,9 +752,7 @@ static void nfs_direct_write_completion(struct nfs_pgio_header *hdr)
                req = nfs_list_entry(hdr->pages.next);
                nfs_list_remove_request(req);
-                switch (bit) {
+                if (request_commit) {
-                case NFS_IOHDR_NEED_RESCHED:
-                case NFS_IOHDR_NEED_COMMIT:
                        kref_get(&req->wb_kref);
                        nfs_mark_request_commit(req, hdr->lseg, &cinfo);
                }
diff --git a/fs/nfs/filelayout/filelayout.c b/fs/nfs/filelayout/filelayout.c
index d2eba1c13b7e..1359c4a27393 100644
--- a/fs/nfs/filelayout/filelayout.c
+++ b/fs/nfs/filelayout/filelayout.c
@@ -84,45 +84,37 @@ filelayout_get_dserver_offset(struct pnfs_layout_segment *lseg, loff_t offset)
        BUG();
 }
-static void filelayout_reset_write(struct nfs_pgio_data *data)
+static void filelayout_reset_write(struct nfs_pgio_header *hdr)
 {
-        struct nfs_pgio_header *hdr = data->header;
+        struct rpc_task *task = &hdr->task;
-        struct rpc_task *task = &data->task;
        if (!test_and_set_bit(NFS_IOHDR_REDO, &hdr->flags)) {
                dprintk("%s Reset task %5u for i/o through MDS "
                        "(req %s/%llu, %u bytes @ offset %llu)\n", __func__,
-                        data->task.tk_pid,
+                        hdr->task.tk_pid,
                        hdr->inode->i_sb->s_id,
                        (unsigned long long)NFS_FILEID(hdr->inode),
-                        data->args.count,
+                        hdr->args.count,
-                        (unsigned long long)data->args.offset);
+                        (unsigned long long)hdr->args.offset);
-                task->tk_status = pnfs_write_done_resend_to_mds(hdr->inode,
+                task->tk_status = pnfs_write_done_resend_to_mds(hdr);
-                                                        &hdr->pages,
-                                                        hdr->completion_ops,
-                                                        hdr->dreq);
        }
 }
-static void filelayout_reset_read(struct nfs_pgio_data *data)
+static void filelayout_reset_read(struct nfs_pgio_header *hdr)
 {
-        struct nfs_pgio_header *hdr = data->header;
+        struct rpc_task *task = &hdr->task;
-        struct rpc_task *task = &data->task;
        if (!test_and_set_bit(NFS_IOHDR_REDO, &hdr->flags)) {
                dprintk("%s Reset task %5u for i/o through MDS "
                        "(req %s/%llu, %u bytes @ offset %llu)\n", __func__,
-                        data->task.tk_pid,
+                        hdr->task.tk_pid,
                        hdr->inode->i_sb->s_id,
                        (unsigned long long)NFS_FILEID(hdr->inode),
-                        data->args.count,
+                        hdr->args.count,
-                        (unsigned long long)data->args.offset);
+                        (unsigned long long)hdr->args.offset);
-                task->tk_status = pnfs_read_done_resend_to_mds(hdr->inode,
+                task->tk_status = pnfs_read_done_resend_to_mds(hdr);
-                                                        &hdr->pages,
-                                                        hdr->completion_ops,
-                                                        hdr->dreq);
        }
 }
@@ -243,18 +235,17 @@ wait_on_recovery:
 /* NFS_PROTO call done callback routines */
 static int filelayout_read_done_cb(struct rpc_task *task,
-                                struct nfs_pgio_data *data)
+                                struct nfs_pgio_header *hdr)
 {
-        struct nfs_pgio_header *hdr = data->header;
        int err;
-        trace_nfs4_pnfs_read(data, task->tk_status);
+        trace_nfs4_pnfs_read(hdr, task->tk_status);
-        err = filelayout_async_handle_error(task, data->args.context->state,
+        err = filelayout_async_handle_error(task, hdr->args.context->state,
-                                            data->ds_clp, hdr->lseg);
+                                            hdr->ds_clp, hdr->lseg);
        switch (err) {
        case -NFS4ERR_RESET_TO_MDS:
-                filelayout_reset_read(data);
+                filelayout_reset_read(hdr);
                return task->tk_status;
        case -EAGAIN:
                rpc_restart_call_prepare(task);
@@ -270,15 +261,14 @@ static int filelayout_read_done_cb(struct rpc_task *task,
 * rfc5661 is not clear about which credential should be used.
 */
 static void
-filelayout_set_layoutcommit(struct nfs_pgio_data *wdata)
+filelayout_set_layoutcommit(struct nfs_pgio_header *hdr)
 {
-        struct nfs_pgio_header *hdr = wdata->header;
        if (FILELAYOUT_LSEG(hdr->lseg)->commit_through_mds ||
-            wdata->res.verf->committed == NFS_FILE_SYNC)
+            hdr->res.verf->committed == NFS_FILE_SYNC)
                return;
-        pnfs_set_layoutcommit(wdata);
+        pnfs_set_layoutcommit(hdr);
        dprintk("%s inode %lu pls_end_pos %lu\n", __func__, hdr->inode->i_ino,
                (unsigned long) NFS_I(hdr->inode)->layout->plh_lwb);
 }
@@ -305,83 +295,82 @@ filelayout_reset_to_mds(struct pnfs_layout_segment *lseg)
 */
 static void filelayout_read_prepare(struct rpc_task *task, void *data)
 {
-        struct nfs_pgio_data *rdata = data;
+        struct nfs_pgio_header *hdr = data;
-        if (unlikely(test_bit(NFS_CONTEXT_BAD, &rdata->args.context->flags))) {
+        if (unlikely(test_bit(NFS_CONTEXT_BAD, &hdr->args.context->flags))) {
                rpc_exit(task, -EIO);
                return;
        }
-        if (filelayout_reset_to_mds(rdata->header->lseg)) {
+        if (filelayout_reset_to_mds(hdr->lseg)) {
                dprintk("%s task %u reset io to MDS\n", __func__, task->tk_pid);
-                filelayout_reset_read(rdata);
+                filelayout_reset_read(hdr);
                rpc_exit(task, 0);
                return;
        }
-        rdata->pgio_done_cb = filelayout_read_done_cb;
+        hdr->pgio_done_cb = filelayout_read_done_cb;
-        if (nfs41_setup_sequence(rdata->ds_clp->cl_session,
+        if (nfs41_setup_sequence(hdr->ds_clp->cl_session,
-                        &rdata->args.seq_args,
+                        &hdr->args.seq_args,
-                        &rdata->res.seq_res,
+                        &hdr->res.seq_res,
                        task))
                return;
-        if (nfs4_set_rw_stateid(&rdata->args.stateid, rdata->args.context,
+        if (nfs4_set_rw_stateid(&hdr->args.stateid, hdr->args.context,
-                        rdata->args.lock_context, FMODE_READ) == -EIO)
+                        hdr->args.lock_context, FMODE_READ) == -EIO)
                rpc_exit(task, -EIO); /* lost lock, terminate I/O */
 }
 static void filelayout_read_call_done(struct rpc_task *task, void *data)
 {
-        struct nfs_pgio_data *rdata = data;
+        struct nfs_pgio_header *hdr = data;
        dprintk("--> %s task->tk_status %d\n", __func__, task->tk_status);
-        if (test_bit(NFS_IOHDR_REDO, &rdata->header->flags) &&
+        if (test_bit(NFS_IOHDR_REDO, &hdr->flags) &&
            task->tk_status == 0) {
-                nfs41_sequence_done(task, &rdata->res.seq_res);
+                nfs41_sequence_done(task, &hdr->res.seq_res);
                return;
        }
        /* Note this may cause RPC to be resent */
-        rdata->header->mds_ops->rpc_call_done(task, data);
+        hdr->mds_ops->rpc_call_done(task, data);
 }
 static void filelayout_read_count_stats(struct rpc_task *task, void *data)
 {
-        struct nfs_pgio_data *rdata = data;
+        struct nfs_pgio_header *hdr = data;
-        rpc_count_iostats(task, NFS_SERVER(rdata->header->inode)->client->cl_metrics);
+        rpc_count_iostats(task, NFS_SERVER(hdr->inode)->client->cl_metrics);
 }
 static void filelayout_read_release(void *data)
 {
-        struct nfs_pgio_data *rdata = data;
+        struct nfs_pgio_header *hdr = data;
-        struct pnfs_layout_hdr *lo = rdata->header->lseg->pls_layout;
+        struct pnfs_layout_hdr *lo = hdr->lseg->pls_layout;
        filelayout_fenceme(lo->plh_inode, lo);
-        nfs_put_client(rdata->ds_clp);
+        nfs_put_client(hdr->ds_clp);
-        rdata->header->mds_ops->rpc_release(data);
+        hdr->mds_ops->rpc_release(data);
 }
 static int filelayout_write_done_cb(struct rpc_task *task,
-                                struct nfs_pgio_data *data)
+                                struct nfs_pgio_header *hdr)
 {
-        struct nfs_pgio_header *hdr = data->header;
        int err;
-        trace_nfs4_pnfs_write(data, task->tk_status);
+        trace_nfs4_pnfs_write(hdr, task->tk_status);
-        err = filelayout_async_handle_error(task, data->args.context->state,
+        err = filelayout_async_handle_error(task, hdr->args.context->state,
-                                            data->ds_clp, hdr->lseg);
+                                            hdr->ds_clp, hdr->lseg);
        switch (err) {
        case -NFS4ERR_RESET_TO_MDS:
-                filelayout_reset_write(data);
+                filelayout_reset_write(hdr);
                return task->tk_status;
        case -EAGAIN:
                rpc_restart_call_prepare(task);
                return -EAGAIN;
        }
-        filelayout_set_layoutcommit(data);
+        filelayout_set_layoutcommit(hdr);
        return 0;
 }
@@ -419,57 +408,57 @@ static int filelayout_commit_done_cb(struct rpc_task *task,
 static void filelayout_write_prepare(struct rpc_task *task, void *data)
 {
-        struct nfs_pgio_data *wdata = data;
+        struct nfs_pgio_header *hdr = data;
-        if (unlikely(test_bit(NFS_CONTEXT_BAD, &wdata->args.context->flags))) {
+        if (unlikely(test_bit(NFS_CONTEXT_BAD, &hdr->args.context->flags))) {
                rpc_exit(task, -EIO);
                return;
        }
-        if (filelayout_reset_to_mds(wdata->header->lseg)) {
+        if (filelayout_reset_to_mds(hdr->lseg)) {
                dprintk("%s task %u reset io to MDS\n", __func__, task->tk_pid);
-                filelayout_reset_write(wdata);
+                filelayout_reset_write(hdr);
                rpc_exit(task, 0);
                return;
        }
-        if (nfs41_setup_sequence(wdata->ds_clp->cl_session,
+        if (nfs41_setup_sequence(hdr->ds_clp->cl_session,
-                        &wdata->args.seq_args,
+                        &hdr->args.seq_args,
-                        &wdata->res.seq_res,
+                        &hdr->res.seq_res,
                        task))
                return;
-        if (nfs4_set_rw_stateid(&wdata->args.stateid, wdata->args.context,
+        if (nfs4_set_rw_stateid(&hdr->args.stateid, hdr->args.context,
-                        wdata->args.lock_context, FMODE_WRITE) == -EIO)
+                        hdr->args.lock_context, FMODE_WRITE) == -EIO)
                rpc_exit(task, -EIO); /* lost lock, terminate I/O */
 }
 static void filelayout_write_call_done(struct rpc_task *task, void *data)
 {
-        struct nfs_pgio_data *wdata = data;
+        struct nfs_pgio_header *hdr = data;
-        if (test_bit(NFS_IOHDR_REDO, &wdata->header->flags) &&
+        if (test_bit(NFS_IOHDR_REDO, &hdr->flags) &&
            task->tk_status == 0) {
-                nfs41_sequence_done(task, &wdata->res.seq_res);
+                nfs41_sequence_done(task, &hdr->res.seq_res);
                return;
        }
        /* Note this may cause RPC to be resent */
-        wdata->header->mds_ops->rpc_call_done(task, data);
+        hdr->mds_ops->rpc_call_done(task, data);
 }
 static void filelayout_write_count_stats(struct rpc_task *task, void *data)
 {
-        struct nfs_pgio_data *wdata = data;
+        struct nfs_pgio_header *hdr = data;
-        rpc_count_iostats(task, NFS_SERVER(wdata->header->inode)->client->cl_metrics);
+        rpc_count_iostats(task, NFS_SERVER(hdr->inode)->client->cl_metrics);
 }
 static void filelayout_write_release(void *data)
 {
-        struct nfs_pgio_data *wdata = data;
+        struct nfs_pgio_header *hdr = data;
-        struct pnfs_layout_hdr *lo = wdata->header->lseg->pls_layout;
+        struct pnfs_layout_hdr *lo = hdr->lseg->pls_layout;
        filelayout_fenceme(lo->plh_inode, lo);
-        nfs_put_client(wdata->ds_clp);
+        nfs_put_client(hdr->ds_clp);
-        wdata->header->mds_ops->rpc_release(data);
+        hdr->mds_ops->rpc_release(data);
 }
 static void filelayout_commit_prepare(struct rpc_task *task, void *data)
@@ -529,19 +518,18 @@ static const struct rpc_call_ops filelayout_commit_call_ops = {
 };
 static enum pnfs_try_status
-filelayout_read_pagelist(struct nfs_pgio_data *data)
+filelayout_read_pagelist(struct nfs_pgio_header *hdr)
 {
-        struct nfs_pgio_header *hdr = data->header;
        struct pnfs_layout_segment *lseg = hdr->lseg;
        struct nfs4_pnfs_ds *ds;
        struct rpc_clnt *ds_clnt;
-        loff_t offset = data->args.offset;
+        loff_t offset = hdr->args.offset;
        u32 j, idx;
        struct nfs_fh *fh;
        dprintk("--> %s ino %lu pgbase %u req %Zu@%llu\n",
                __func__, hdr->inode->i_ino,
-                data->args.pgbase, (size_t)data->args.count, offset);
+                hdr->args.pgbase, (size_t)hdr->args.count, offset);
        /* Retrieve the correct rpc_client for the byte range */
        j = nfs4_fl_calc_j_index(lseg, offset);
@@ -559,30 +547,29 @@ filelayout_read_pagelist(struct nfs_pgio_data *data)
        /* No multipath support. Use first DS */
        atomic_inc(&ds->ds_clp->cl_count);
-        data->ds_clp = ds->ds_clp;
+        hdr->ds_clp = ds->ds_clp;
-        data->ds_idx = idx;
+        hdr->ds_idx = idx;
        fh = nfs4_fl_select_ds_fh(lseg, j);
        if (fh)
-                data->args.fh = fh;
+                hdr->args.fh = fh;
-        data->args.offset = filelayout_get_dserver_offset(lseg, offset);
+        hdr->args.offset = filelayout_get_dserver_offset(lseg, offset);
-        data->mds_offset = offset;
+        hdr->mds_offset = offset;
        /* Perform an asynchronous read to ds */
-        nfs_initiate_pgio(ds_clnt, data,
+        nfs_initiate_pgio(ds_clnt, hdr,
                            &filelayout_read_call_ops, 0, RPC_TASK_SOFTCONN);
        return PNFS_ATTEMPTED;
 }
 /* Perform async writes. */
 static enum pnfs_try_status
-filelayout_write_pagelist(struct nfs_pgio_data *data, int sync)
+filelayout_write_pagelist(struct nfs_pgio_header *hdr, int sync)
 {
-        struct nfs_pgio_header *hdr = data->header;
        struct pnfs_layout_segment *lseg = hdr->lseg;
        struct nfs4_pnfs_ds *ds;
        struct rpc_clnt *ds_clnt;
-        loff_t offset = data->args.offset;
+        loff_t offset = hdr->args.offset;
        u32 j, idx;
        struct nfs_fh *fh;
@@ -598,21 +585,20 @@ filelayout_write_pagelist(struct nfs_pgio_data *data, int sync)
                return PNFS_NOT_ATTEMPTED;
        dprintk("%s ino %lu sync %d req %Zu@%llu DS: %s cl_count %d\n",
-                __func__, hdr->inode->i_ino, sync, (size_t) data->args.count,
+                __func__, hdr->inode->i_ino, sync, (size_t) hdr->args.count,
                offset, ds->ds_remotestr, atomic_read(&ds->ds_clp->cl_count));
-        data->pgio_done_cb = filelayout_write_done_cb;
+        hdr->pgio_done_cb = filelayout_write_done_cb;
        atomic_inc(&ds->ds_clp->cl_count);
-        data->ds_clp = ds->ds_clp;
+        hdr->ds_clp = ds->ds_clp;
-        data->ds_idx = idx;
+        hdr->ds_idx = idx;
        fh = nfs4_fl_select_ds_fh(lseg, j);
        if (fh)
-                data->args.fh = fh;
+                hdr->args.fh = fh;
+        hdr->args.offset = filelayout_get_dserver_offset(lseg, offset);
-        data->args.offset = filelayout_get_dserver_offset(lseg, offset);
        /* Perform an asynchronous write */
-        nfs_initiate_pgio(ds_clnt, data,
+        nfs_initiate_pgio(ds_clnt, hdr,
                                    &filelayout_write_call_ops, sync,
                                    RPC_TASK_SOFTCONN);
        return PNFS_ATTEMPTED;
@@ -1023,6 +1009,7 @@ static u32 select_bucket_index(struct nfs4_filelayout_segment *fl, u32 j)
 /* The generic layer is about to remove the req from the commit list.
 * If this will make the bucket empty, it will need to put the lseg reference.
+ * Note this is must be called holding the inode (/cinfo) lock
 */
 static void
 filelayout_clear_request_commit(struct nfs_page *req,
@@ -1030,7 +1017,6 @@ filelayout_clear_request_commit(struct nfs_page *req,
 {
        struct pnfs_layout_segment *freeme = NULL;
-        spin_lock(cinfo->lock);
        if (!test_and_clear_bit(PG_COMMIT_TO_DS, &req->wb_flags))
                goto out;
        cinfo->ds->nwritten--;
@@ -1045,22 +1031,25 @@ filelayout_clear_request_commit(struct nfs_page *req,
        }
 out:
        nfs_request_remove_commit_list(req, cinfo);
-        spin_unlock(cinfo->lock);
+        pnfs_put_lseg_async(freeme);
-        pnfs_put_lseg(freeme);
 }
-static struct list_head *
+static void
-filelayout_choose_commit_list(struct nfs_page *req,
+filelayout_mark_request_commit(struct nfs_page *req,
-                              struct pnfs_layout_segment *lseg,
+                               struct pnfs_layout_segment *lseg,
-                              struct nfs_commit_info *cinfo)
+                               struct nfs_commit_info *cinfo)
 {
        struct nfs4_filelayout_segment *fl = FILELAYOUT_LSEG(lseg);
        u32 i, j;
        struct list_head *list;
        struct pnfs_commit_bucket *buckets;
-        if (fl->commit_through_mds)
+        if (fl->commit_through_mds) {
-                return &cinfo->mds->list;
+                list = &cinfo->mds->list;
+                spin_lock(cinfo->lock);
+                goto mds_commit;
+        }
        /* Note that we are calling nfs4_fl_calc_j_index on each page
         * that ends up being committed to a data server.  An attractive
@@ -1084,19 +1073,22 @@ filelayout_choose_commit_list(struct nfs_page *req,
        }
        set_bit(PG_COMMIT_TO_DS, &req->wb_flags);
        cinfo->ds->nwritten++;
-        spin_unlock(cinfo->lock);
-        return list;
-}
-static void
+mds_commit:
-filelayout_mark_request_commit(struct nfs_page *req,
+        /* nfs_request_add_commit_list(). We need to add req to list without
-                               struct pnfs_layout_segment *lseg,
+         * dropping cinfo lock.
-                               struct nfs_commit_info *cinfo)
+         */
-{
+        set_bit(PG_CLEAN, &(req)->wb_flags);
-        struct list_head *list;
+        nfs_list_add_request(req, list);
+        cinfo->mds->ncommit++;
-        list = filelayout_choose_commit_list(req, lseg, cinfo);
+        spin_unlock(cinfo->lock);
-        nfs_request_add_commit_list(req, list, cinfo);
+        if (!cinfo->dreq) {
+                inc_zone_page_state(req->wb_page, NR_UNSTABLE_NFS);
+                inc_bdi_stat(page_file_mapping(req->wb_page)->backing_dev_info,
+                             BDI_RECLAIMABLE);
+                __mark_inode_dirty(req->wb_context->dentry->d_inode,
+                                   I_DIRTY_DATASYNC);
+        }
 }
 static u32 calc_ds_index_from_commit(struct pnfs_layout_segment *lseg, u32 i)
@@ -1244,15 +1236,63 @@ restart:
        spin_unlock(cinfo->lock);
 }
+/* filelayout_search_commit_reqs - Search lists in @cinfo for the head reqest
+ *                                 for @page
+ * @cinfo - commit info for current inode
+ * @page - page to search for matching head request
+ *
+ * Returns a the head request if one is found, otherwise returns NULL.
+ */
+static struct nfs_page *
+filelayout_search_commit_reqs(struct nfs_commit_info *cinfo, struct page *page)
+{
+        struct nfs_page *freq, *t;
+        struct pnfs_commit_bucket *b;
+        int i;
+        /* Linearly search the commit lists for each bucket until a matching
+         * request is found */
+        for (i = 0, b = cinfo->ds->buckets; i < cinfo->ds->nbuckets; i++, b++) {
+                list_for_each_entry_safe(freq, t, &b->written, wb_list) {
+                        if (freq->wb_page == page)
+                                return freq->wb_head;
+                }
+                list_for_each_entry_safe(freq, t, &b->committing, wb_list) {
+                        if (freq->wb_page == page)
+                                return freq->wb_head;
+                }
+        }
+        return NULL;
+}
+static void filelayout_retry_commit(struct nfs_commit_info *cinfo, int idx)
+{
+        struct pnfs_ds_commit_info *fl_cinfo = cinfo->ds;
+        struct pnfs_commit_bucket *bucket = fl_cinfo->buckets;
+        struct pnfs_layout_segment *freeme;
+        int i;
+        for (i = idx; i < fl_cinfo->nbuckets; i++, bucket++) {
+                if (list_empty(&bucket->committing))
+                        continue;
+                nfs_retry_commit(&bucket->committing, bucket->clseg, cinfo);
+                spin_lock(cinfo->lock);
+                freeme = bucket->clseg;
+                bucket->clseg = NULL;
+                spin_unlock(cinfo->lock);
+                pnfs_put_lseg(freeme);
+        }
+}
 static unsigned int
 alloc_ds_commits(struct nfs_commit_info *cinfo, struct list_head *list)
 {
        struct pnfs_ds_commit_info *fl_cinfo;
        struct pnfs_commit_bucket *bucket;
        struct nfs_commit_data *data;
-        int i, j;
+        int i;
        unsigned int nreq = 0;
-        struct pnfs_layout_segment *freeme;
        fl_cinfo = cinfo->ds;
        bucket = fl_cinfo->buckets;
@@ -1272,16 +1312,7 @@ alloc_ds_commits(struct nfs_commit_info *cinfo, struct list_head *list)
        }
        /* Clean up on error */
-        for (j = i; j < fl_cinfo->nbuckets; j++, bucket++) {
+        filelayout_retry_commit(cinfo, i);
-                if (list_empty(&bucket->committing))
-                        continue;
-                nfs_retry_commit(&bucket->committing, bucket->clseg, cinfo);
-                spin_lock(cinfo->lock);
-                freeme = bucket->clseg;
-                bucket->clseg = NULL;
-                spin_unlock(cinfo->lock);
-                pnfs_put_lseg(freeme);
-        }
        /* Caller will clean up entries put on list */
        return nreq;
 }
@@ -1301,8 +1332,12 @@ filelayout_commit_pagelist(struct inode *inode, struct list_head *mds_pages,
                        data->lseg = NULL;
                        list_add(&data->pages, &list);
                        nreq++;
-                } else
+                } else {
                        nfs_retry_commit(mds_pages, NULL, cinfo);
+                        filelayout_retry_commit(cinfo, 0);
+                        cinfo->completion_ops->error_cleanup(NFS_I(inode));
+                        return -ENOMEM;
+                }
        }
        nreq += alloc_ds_commits(cinfo, &list);
@@ -1380,6 +1415,7 @@ static struct pnfs_layoutdriver_type filelayout_type = {
        .clear_request_commit   = filelayout_clear_request_commit,
        .scan_commit_lists      = filelayout_scan_commit_lists,
        .recover_commit_reqs    = filelayout_recover_commit_reqs,
+        .search_commit_reqs     = filelayout_search_commit_reqs,
        .commit_pagelist        = filelayout_commit_pagelist,
        .read_pagelist          = filelayout_read_pagelist,
        .write_pagelist         = filelayout_write_pagelist,
diff --git a/fs/nfs/filelayout/filelayoutdev.c b/fs/nfs/filelayout/filelayoutdev.c
index e2a0361e24c6..8540516f4d71 100644
--- a/fs/nfs/filelayout/filelayoutdev.c
+++ b/fs/nfs/filelayout/filelayoutdev.c
@@ -695,7 +695,7 @@ filelayout_get_device_info(struct inode *inode,
        if (pdev == NULL)
                return NULL;
-        pages = kzalloc(max_pages * sizeof(struct page *), gfp_flags);
+        pages = kcalloc(max_pages, sizeof(struct page *), gfp_flags);
        if (pages == NULL) {
                kfree(pdev);
                return NULL;
diff --git a/fs/nfs/getroot.c b/fs/nfs/getroot.c
index b94f80420a58..880618a8b048 100644
--- a/fs/nfs/getroot.c
+++ b/fs/nfs/getroot.c
@@ -112,7 +112,7 @@ struct dentry *nfs_get_root(struct super_block *sb, struct nfs_fh *mntfh,
         * if the dentry tree reaches them; however if the dentry already
         * exists, we'll pick it up at this point and use it as the root
         */
-        ret = d_obtain_alias(inode);
+        ret = d_obtain_root(inode);
        if (IS_ERR(ret)) {
                dprintk("nfs_get_root: get root dentry failed\n");
                goto out;
diff --git a/fs/nfs/inode.c b/fs/nfs/inode.c
index 68921b01b792..577a36f0a510 100644
--- a/fs/nfs/inode.c
+++ b/fs/nfs/inode.c
@@ -1002,6 +1002,15 @@ int nfs_revalidate_inode(struct nfs_server *server, struct inode *inode)
 }
 EXPORT_SYMBOL_GPL(nfs_revalidate_inode);
+int nfs_revalidate_inode_rcu(struct nfs_server *server, struct inode *inode)
+{
+        if (!(NFS_I(inode)->cache_validity &
+                        (NFS_INO_INVALID_ATTR|NFS_INO_INVALID_LABEL))
+                        && !nfs_attribute_cache_expired(inode))
+                return NFS_STALE(inode) ? -ESTALE : 0;
+        return -ECHILD;
+}
 static int nfs_invalidate_mapping(struct inode *inode, struct address_space *mapping)
 {
        struct nfs_inode *nfsi = NFS_I(inode);
diff --git a/fs/nfs/internal.h b/fs/nfs/internal.h
index e2a45ae5014e..9056622d2230 100644
--- a/fs/nfs/internal.h
+++ b/fs/nfs/internal.h
@@ -247,11 +247,11 @@ void nfs_set_pgio_error(struct nfs_pgio_header *hdr, int error, loff_t pos);
 int nfs_iocounter_wait(struct nfs_io_counter *c);
 extern const struct nfs_pageio_ops nfs_pgio_rw_ops;
-struct nfs_rw_header *nfs_rw_header_alloc(const struct nfs_rw_ops *);
+struct nfs_pgio_header *nfs_pgio_header_alloc(const struct nfs_rw_ops *);
-void nfs_rw_header_free(struct nfs_pgio_header *);
+void nfs_pgio_header_free(struct nfs_pgio_header *);
-void nfs_pgio_data_release(struct nfs_pgio_data *);
+void nfs_pgio_data_destroy(struct nfs_pgio_header *);
 int nfs_generic_pgio(struct nfs_pageio_descriptor *, struct nfs_pgio_header *);
-int nfs_initiate_pgio(struct rpc_clnt *, struct nfs_pgio_data *,
+int nfs_initiate_pgio(struct rpc_clnt *, struct nfs_pgio_header *,
                      const struct rpc_call_ops *, int, int);
 void nfs_free_request(struct nfs_page *req);
@@ -451,6 +451,7 @@ int nfs_scan_commit(struct inode *inode, struct list_head *dst,
 void nfs_mark_request_commit(struct nfs_page *req,
                             struct pnfs_layout_segment *lseg,
                             struct nfs_commit_info *cinfo);
+int nfs_write_need_commit(struct nfs_pgio_header *);
 int nfs_generic_commit_list(struct inode *inode, struct list_head *head,
                            int how, struct nfs_commit_info *cinfo);
 void nfs_retry_commit(struct list_head *page_list,
@@ -491,7 +492,7 @@ static inline void nfs_inode_dio_wait(struct inode *inode)
 extern ssize_t nfs_dreq_bytes_left(struct nfs_direct_req *dreq);
 /* nfs4proc.c */
-extern void __nfs4_read_done_cb(struct nfs_pgio_data *);
+extern void __nfs4_read_done_cb(struct nfs_pgio_header *);
 extern struct nfs_client *nfs4_init_client(struct nfs_client *clp,
                            const struct rpc_timeout *timeparms,
                            const char *ip_addr);
diff --git a/fs/nfs/nfs3acl.c b/fs/nfs/nfs3acl.c
index 8f854dde4150..d0fec260132a 100644
--- a/fs/nfs/nfs3acl.c
+++ b/fs/nfs/nfs3acl.c
@@ -256,7 +256,7 @@ nfs3_list_one_acl(struct inode *inode, int type, const char *name, void *data,
        char *p = data + *result;
        acl = get_acl(inode, type);
-        if (!acl)
+        if (IS_ERR_OR_NULL(acl))
                return 0;
        posix_acl_release(acl);
diff --git a/fs/nfs/nfs3proc.c b/fs/nfs/nfs3proc.c
index f0afa291fd58..809670eba52a 100644
--- a/fs/nfs/nfs3proc.c
+++ b/fs/nfs/nfs3proc.c
@@ -795,41 +795,44 @@ nfs3_proc_pathconf(struct nfs_server *server, struct nfs_fh *fhandle,
        return status;
 }
-static int nfs3_read_done(struct rpc_task *task, struct nfs_pgio_data *data)
+static int nfs3_read_done(struct rpc_task *task, struct nfs_pgio_header *hdr)
 {
-        struct inode *inode = data->header->inode;
+        struct inode *inode = hdr->inode;
        if (nfs3_async_handle_jukebox(task, inode))
                return -EAGAIN;
        nfs_invalidate_atime(inode);
-        nfs_refresh_inode(inode, &data->fattr);
+        nfs_refresh_inode(inode, &hdr->fattr);
        return 0;
 }
-static void nfs3_proc_read_setup(struct nfs_pgio_data *data, struct rpc_message *msg)
+static void nfs3_proc_read_setup(struct nfs_pgio_header *hdr,
+                                 struct rpc_message *msg)
 {
        msg->rpc_proc = &nfs3_procedures[NFS3PROC_READ];
 }
-static int nfs3_proc_pgio_rpc_prepare(struct rpc_task *task, struct nfs_pgio_data *data)
+static int nfs3_proc_pgio_rpc_prepare(struct rpc_task *task,
+                                      struct nfs_pgio_header *hdr)
 {
        rpc_call_start(task);
        return 0;
 }
-static int nfs3_write_done(struct rpc_task *task, struct nfs_pgio_data *data)
+static int nfs3_write_done(struct rpc_task *task, struct nfs_pgio_header *hdr)
 {
-        struct inode *inode = data->header->inode;
+        struct inode *inode = hdr->inode;
        if (nfs3_async_handle_jukebox(task, inode))
                return -EAGAIN;
        if (task->tk_status >= 0)
-                nfs_post_op_update_inode_force_wcc(inode, data->res.fattr);
+                nfs_post_op_update_inode_force_wcc(inode, hdr->res.fattr);
        return 0;
 }
-static void nfs3_proc_write_setup(struct nfs_pgio_data *data, struct rpc_message *msg)
+static void nfs3_proc_write_setup(struct nfs_pgio_header *hdr,
+                                  struct rpc_message *msg)
 {
        msg->rpc_proc = &nfs3_procedures[NFS3PROC_WRITE];
 }
diff --git a/fs/nfs/nfs4_fs.h b/fs/nfs/nfs4_fs.h
index ba2affa51941..92193eddb41d 100644
--- a/fs/nfs/nfs4_fs.h
+++ b/fs/nfs/nfs4_fs.h
@@ -54,7 +54,7 @@ struct nfs4_minor_version_ops {
                        const nfs4_stateid *);
        int     (*find_root_sec)(struct nfs_server *, struct nfs_fh *,
                        struct nfs_fsinfo *);
-        int     (*free_lock_state)(struct nfs_server *,
+        void    (*free_lock_state)(struct nfs_server *,
                        struct nfs4_lock_state *);
        const struct rpc_call_ops *call_sync_ops;
        const struct nfs4_state_recovery_ops *reboot_recovery_ops;
@@ -129,27 +129,17 @@ enum {
 * LOCK: one nfs4_state (LOCK) to hold the lock stateid nfs4_state(OPEN)
 */
-struct nfs4_lock_owner {
-        unsigned int lo_type;
-#define NFS4_ANY_LOCK_TYPE      (0U)
-#define NFS4_FLOCK_LOCK_TYPE    (1U << 0)
-#define NFS4_POSIX_LOCK_TYPE    (1U << 1)
-        union {
-                fl_owner_t posix_owner;
-                pid_t flock_owner;
-        } lo_u;
-};
 struct nfs4_lock_state {
-        struct list_head        ls_locks;       /* Other lock stateids */
+        struct list_head                ls_locks;   /* Other lock stateids */
-        struct nfs4_state *     ls_state;       /* Pointer to open state */
+        struct nfs4_state *             ls_state;   /* Pointer to open state */
 #define NFS_LOCK_INITIALIZED 0
 #define NFS_LOCK_LOST        1
-        unsigned long           ls_flags;
+        unsigned long                   ls_flags;
        struct nfs_seqid_counter        ls_seqid;
-        nfs4_stateid            ls_stateid;
+        nfs4_stateid                    ls_stateid;
-        atomic_t                ls_count;
+        atomic_t                        ls_count;
-        struct nfs4_lock_owner  ls_owner;
+        fl_owner_t                      ls_owner;
+        struct work_struct              ls_release;
 };
 /* bits for nfs4_state->flags */
@@ -337,11 +327,11 @@ nfs4_state_protect(struct nfs_client *clp, unsigned long sp4_mode,
 */
 static inline void
 nfs4_state_protect_write(struct nfs_client *clp, struct rpc_clnt **clntp,
-                         struct rpc_message *msg, struct nfs_pgio_data *wdata)
+                         struct rpc_message *msg, struct nfs_pgio_header *hdr)
 {
        if (_nfs4_state_protect(clp, NFS_SP4_MACH_CRED_WRITE, clntp, msg) &&
            !test_bit(NFS_SP4_MACH_CRED_COMMIT, &clp->cl_sp4_flags))
-                wdata->args.stable = NFS_FILE_SYNC;
+                hdr->args.stable = NFS_FILE_SYNC;
 }
 #else /* CONFIG_NFS_v4_1 */
 static inline struct nfs4_session *nfs4_get_session(const struct nfs_server *server)
@@ -369,7 +359,7 @@ nfs4_state_protect(struct nfs_client *clp, unsigned long sp4_flags,
 static inline void
 nfs4_state_protect_write(struct nfs_client *clp, struct rpc_clnt **clntp,
-                         struct rpc_message *msg, struct nfs_pgio_data *wdata)
+                         struct rpc_message *msg, struct nfs_pgio_header *hdr)
 {
 }
 #endif /* CONFIG_NFS_V4_1 */
diff --git a/fs/nfs/nfs4client.c b/fs/nfs/nfs4client.c
index aa9ef4876046..53e435a95260 100644
--- a/fs/nfs/nfs4client.c
+++ b/fs/nfs/nfs4client.c
@@ -855,6 +855,11 @@ struct nfs_client *nfs4_set_ds_client(struct nfs_client* mds_clp,
        };
        struct rpc_timeout ds_timeout;
        struct nfs_client *clp;
+        char buf[INET6_ADDRSTRLEN + 1];
+        if (rpc_ntop(ds_addr, buf, sizeof(buf)) <= 0)
+                return ERR_PTR(-EINVAL);
+        cl_init.hostname = buf;
        /*
         * Set an authflavor equual to the MDS value. Use the MDS nfs_client
diff --git a/fs/nfs/nfs4proc.c b/fs/nfs/nfs4proc.c
index 4bf3d97cc5a0..75ae8d22f067 100644
--- a/fs/nfs/nfs4proc.c
+++ b/fs/nfs/nfs4proc.c
@@ -1952,6 +1952,14 @@ static int _nfs4_recover_proc_open(struct nfs4_opendata *data)
        return status;
 }
+/*
+ * Additional permission checks in order to distinguish between an
+ * open for read, and an open for execute. This works around the
+ * fact that NFSv4 OPEN treats read and execute permissions as being
+ * the same.
+ * Note that in the non-execute case, we want to turn off permission
+ * checking if we just created a new file (POSIX open() semantics).
+ */
 static int nfs4_opendata_access(struct rpc_cred *cred,
                                struct nfs4_opendata *opendata,
                                struct nfs4_state *state, fmode_t fmode,
@@ -1966,14 +1974,14 @@ static int nfs4_opendata_access(struct rpc_cred *cred,
                return 0;
        mask = 0;
-        /* don't check MAY_WRITE - a newly created file may not have
+        /*
-         * write mode bits, but POSIX allows the creating process to write.
+         * Use openflags to check for exec, because fmode won't
-         * use openflags to check for exec, because fmode won't
+         * always have FMODE_EXEC set when file open for exec.
-         * always have FMODE_EXEC set when file open for exec. */
+         */
        if (openflags & __FMODE_EXEC) {
                /* ONLY check for exec rights */
                mask = MAY_EXEC;
-        } else if (fmode & FMODE_READ)
+        } else if ((fmode & FMODE_READ) && !opendata->file_created)
                mask = MAY_READ;
        cache.cred = cred;
@@ -2216,8 +2224,15 @@ static int _nfs4_open_and_get_state(struct nfs4_opendata *opendata,
        seq = raw_seqcount_begin(&sp->so_reclaim_seqcount);
        ret = _nfs4_proc_open(opendata);
-        if (ret != 0)
+        if (ret != 0) {
+                if (ret == -ENOENT) {
+                        d_drop(opendata->dentry);
+                        d_add(opendata->dentry, NULL);
+                        nfs_set_verifier(opendata->dentry,
+                                         nfs_save_change_attribute(opendata->dir->d_inode));
+                }
                goto out;
+        }
        state = nfs4_opendata_to_nfs4_state(opendata);
        ret = PTR_ERR(state);
@@ -2647,6 +2662,48 @@ static const struct rpc_call_ops nfs4_close_ops = {
        .rpc_release = nfs4_free_closedata,
 };
+static bool nfs4_state_has_opener(struct nfs4_state *state)
+{
+        /* first check existing openers */
+        if (test_bit(NFS_O_RDONLY_STATE, &state->flags) != 0 &&
+            state->n_rdonly != 0)
+                return true;
+        if (test_bit(NFS_O_WRONLY_STATE, &state->flags) != 0 &&
+            state->n_wronly != 0)
+                return true;
+        if (test_bit(NFS_O_RDWR_STATE, &state->flags) != 0 &&
+            state->n_rdwr != 0)
+                return true;
+        return false;
+}
+static bool nfs4_roc(struct inode *inode)
+{
+        struct nfs_inode *nfsi = NFS_I(inode);
+        struct nfs_open_context *ctx;
+        struct nfs4_state *state;
+        spin_lock(&inode->i_lock);
+        list_for_each_entry(ctx, &nfsi->open_files, list) {
+                state = ctx->state;
+                if (state == NULL)
+                        continue;
+                if (nfs4_state_has_opener(state)) {
+                        spin_unlock(&inode->i_lock);
+                        return false;
+                }
+        }
+        spin_unlock(&inode->i_lock);
+        if (nfs4_check_delegation(inode, FMODE_READ))
+                return false;
+        return pnfs_roc(inode);
+}
 /* 
 * It is possible for data to be read/written from a mem-mapped file 
 * after the sys_close call (which hits the vfs layer as a flush).
@@ -2697,7 +2754,7 @@ int nfs4_do_close(struct nfs4_state *state, gfp_t gfp_mask, int wait)
        calldata->res.fattr = &calldata->fattr;
        calldata->res.seqid = calldata->arg.seqid;
        calldata->res.server = server;
-        calldata->roc = pnfs_roc(state->inode);
+        calldata->roc = nfs4_roc(state->inode);
        nfs_sb_active(calldata->inode->i_sb);
        msg.rpc_argp = &calldata->arg;
@@ -4033,24 +4090,25 @@ static bool nfs4_error_stateid_expired(int err)
        return false;
 }
-void __nfs4_read_done_cb(struct nfs_pgio_data *data)
+void __nfs4_read_done_cb(struct nfs_pgio_header *hdr)
 {
-        nfs_invalidate_atime(data->header->inode);
+        nfs_invalidate_atime(hdr->inode);
 }
-static int nfs4_read_done_cb(struct rpc_task *task, struct nfs_pgio_data *data)
+static int nfs4_read_done_cb(struct rpc_task *task, struct nfs_pgio_header *hdr)
 {
-        struct nfs_server *server = NFS_SERVER(data->header->inode);
+        struct nfs_server *server = NFS_SERVER(hdr->inode);
-        trace_nfs4_read(data, task->tk_status);
+        trace_nfs4_read(hdr, task->tk_status);
-        if (nfs4_async_handle_error(task, server, data->args.context->state) == -EAGAIN) {
+        if (nfs4_async_handle_error(task, server,
+                                    hdr->args.context->state) == -EAGAIN) {
                rpc_restart_call_prepare(task);
                return -EAGAIN;
        }
-        __nfs4_read_done_cb(data);
+        __nfs4_read_done_cb(hdr);
        if (task->tk_status > 0)
-                renew_lease(server, data->timestamp);
+                renew_lease(server, hdr->timestamp);
        return 0;
 }
@@ -4068,54 +4126,59 @@ static bool nfs4_read_stateid_changed(struct rpc_task *task,
        return true;
 }
-static int nfs4_read_done(struct rpc_task *task, struct nfs_pgio_data *data)
+static int nfs4_read_done(struct rpc_task *task, struct nfs_pgio_header *hdr)
 {
        dprintk("--> %s\n", __func__);
-        if (!nfs4_sequence_done(task, &data->res.seq_res))
+        if (!nfs4_sequence_done(task, &hdr->res.seq_res))
                return -EAGAIN;
-        if (nfs4_read_stateid_changed(task, &data->args))
+        if (nfs4_read_stateid_changed(task, &hdr->args))
                return -EAGAIN;
-        return data->pgio_done_cb ? data->pgio_done_cb(task, data) :
+        return hdr->pgio_done_cb ? hdr->pgio_done_cb(task, hdr) :
-                                    nfs4_read_done_cb(task, data);
+                                    nfs4_read_done_cb(task, hdr);
 }
-static void nfs4_proc_read_setup(struct nfs_pgio_data *data, struct rpc_message *msg)
+static void nfs4_proc_read_setup(struct nfs_pgio_header *hdr,
+                                 struct rpc_message *msg)
 {
-        data->timestamp   = jiffies;
+        hdr->timestamp   = jiffies;
-        data->pgio_done_cb = nfs4_read_done_cb;
+        hdr->pgio_done_cb = nfs4_read_done_cb;
        msg->rpc_proc = &nfs4_procedures[NFSPROC4_CLNT_READ];
-        nfs4_init_sequence(&data->args.seq_args, &data->res.seq_res, 0);
+        nfs4_init_sequence(&hdr->args.seq_args, &hdr->res.seq_res, 0);
 }
-static int nfs4_proc_pgio_rpc_prepare(struct rpc_task *task, struct nfs_pgio_data *data)
+static int nfs4_proc_pgio_rpc_prepare(struct rpc_task *task,
+                                      struct nfs_pgio_header *hdr)
 {
-        if (nfs4_setup_sequence(NFS_SERVER(data->header->inode),
+        if (nfs4_setup_sequence(NFS_SERVER(hdr->inode),
-                        &data->args.seq_args,
+                        &hdr->args.seq_args,
-                        &data->res.seq_res,
+                        &hdr->res.seq_res,
                        task))
                return 0;
-        if (nfs4_set_rw_stateid(&data->args.stateid, data->args.context,
+        if (nfs4_set_rw_stateid(&hdr->args.stateid, hdr->args.context,
-                                data->args.lock_context, data->header->rw_ops->rw_mode) == -EIO)
+                                hdr->args.lock_context,
+                                hdr->rw_ops->rw_mode) == -EIO)
                return -EIO;
-        if (unlikely(test_bit(NFS_CONTEXT_BAD, &data->args.context->flags)))
+        if (unlikely(test_bit(NFS_CONTEXT_BAD, &hdr->args.context->flags)))
                return -EIO;
        return 0;
 }
-static int nfs4_write_done_cb(struct rpc_task *task, struct nfs_pgio_data *data)
+static int nfs4_write_done_cb(struct rpc_task *task,
+                              struct nfs_pgio_header *hdr)
 {
-        struct inode *inode = data->header->inode;
+        struct inode *inode = hdr->inode;
        
-        trace_nfs4_write(data, task->tk_status);
+        trace_nfs4_write(hdr, task->tk_status);
-        if (nfs4_async_handle_error(task, NFS_SERVER(inode), data->args.context->state) == -EAGAIN) {
+        if (nfs4_async_handle_error(task, NFS_SERVER(inode),
+                                    hdr->args.context->state) == -EAGAIN) {
                rpc_restart_call_prepare(task);
                return -EAGAIN;
        }
        if (task->tk_status >= 0) {
-                renew_lease(NFS_SERVER(inode), data->timestamp);
+                renew_lease(NFS_SERVER(inode), hdr->timestamp);
-                nfs_post_op_update_inode_force_wcc(inode, &data->fattr);
+                nfs_post_op_update_inode_force_wcc(inode, &hdr->fattr);
        }
        return 0;
 }
@@ -4134,23 +4197,21 @@ static bool nfs4_write_stateid_changed(struct rpc_task *task,
        return true;
 }
-static int nfs4_write_done(struct rpc_task *task, struct nfs_pgio_data *data)
+static int nfs4_write_done(struct rpc_task *task, struct nfs_pgio_header *hdr)
 {
-        if (!nfs4_sequence_done(task, &data->res.seq_res))
+        if (!nfs4_sequence_done(task, &hdr->res.seq_res))
                return -EAGAIN;
-        if (nfs4_write_stateid_changed(task, &data->args))
+        if (nfs4_write_stateid_changed(task, &hdr->args))
                return -EAGAIN;
-        return data->pgio_done_cb ? data->pgio_done_cb(task, data) :
+        return hdr->pgio_done_cb ? hdr->pgio_done_cb(task, hdr) :
-                nfs4_write_done_cb(task, data);
+                nfs4_write_done_cb(task, hdr);
 }
 static
-bool nfs4_write_need_cache_consistency_data(const struct nfs_pgio_data *data)
+bool nfs4_write_need_cache_consistency_data(struct nfs_pgio_header *hdr)
 {
-        const struct nfs_pgio_header *hdr = data->header;
        /* Don't request attributes for pNFS or O_DIRECT writes */
-        if (data->ds_clp != NULL || hdr->dreq != NULL)
+        if (hdr->ds_clp != NULL || hdr->dreq != NULL)
                return false;
        /* Otherwise, request attributes if and only if we don't hold
         * a delegation
@@ -4158,23 +4219,24 @@ bool nfs4_write_need_cache_consistency_data(const struct nfs_pgio_data *data)
        return nfs4_have_delegation(hdr->inode, FMODE_READ) == 0;
 }
-static void nfs4_proc_write_setup(struct nfs_pgio_data *data, struct rpc_message *msg)
+static void nfs4_proc_write_setup(struct nfs_pgio_header *hdr,
+                                  struct rpc_message *msg)
 {
-        struct nfs_server *server = NFS_SERVER(data->header->inode);
+        struct nfs_server *server = NFS_SERVER(hdr->inode);
-        if (!nfs4_write_need_cache_consistency_data(data)) {
+        if (!nfs4_write_need_cache_consistency_data(hdr)) {
-                data->args.bitmask = NULL;
+                hdr->args.bitmask = NULL;
-                data->res.fattr = NULL;
+                hdr->res.fattr = NULL;
        } else
-                data->args.bitmask = server->cache_consistency_bitmask;
+                hdr->args.bitmask = server->cache_consistency_bitmask;
-        if (!data->pgio_done_cb)
+        if (!hdr->pgio_done_cb)
-                data->pgio_done_cb = nfs4_write_done_cb;
+                hdr->pgio_done_cb = nfs4_write_done_cb;
-        data->res.server = server;
+        hdr->res.server = server;
-        data->timestamp   = jiffies;
+        hdr->timestamp   = jiffies;
        msg->rpc_proc = &nfs4_procedures[NFSPROC4_CLNT_WRITE];
-        nfs4_init_sequence(&data->args.seq_args, &data->res.seq_res, 1);
+        nfs4_init_sequence(&hdr->args.seq_args, &hdr->res.seq_res, 1);
 }
 static void nfs4_proc_commit_rpc_prepare(struct rpc_task *task, struct nfs_commit_data *data)
@@ -4881,6 +4943,18 @@ nfs4_init_callback_netid(const struct nfs_client *clp, char *buf, size_t len)
                return scnprintf(buf, len, "tcp");
 }
+static void nfs4_setclientid_done(struct rpc_task *task, void *calldata)
+{
+        struct nfs4_setclientid *sc = calldata;
+        if (task->tk_status == 0)
+                sc->sc_cred = get_rpccred(task->tk_rqstp->rq_cred);
+}
+static const struct rpc_call_ops nfs4_setclientid_ops = {
+        .rpc_call_done = nfs4_setclientid_done,
+};
 /**
 * nfs4_proc_setclientid - Negotiate client ID
 * @clp: state data structure
@@ -4907,6 +4981,14 @@ int nfs4_proc_setclientid(struct nfs_client *clp, u32 program,
                .rpc_resp = res,
                .rpc_cred = cred,
        };
+        struct rpc_task *task;
+        struct rpc_task_setup task_setup_data = {
+                .rpc_client = clp->cl_rpcclient,
+                .rpc_message = &msg,
+                .callback_ops = &nfs4_setclientid_ops,
+                .callback_data = &setclientid,
+                .flags = RPC_TASK_TIMEOUT,
+        };
        int status;
        /* nfs_client_id4 */
@@ -4933,7 +5015,18 @@ int nfs4_proc_setclientid(struct nfs_client *clp, u32 program,
        dprintk("NFS call  setclientid auth=%s, '%.*s'\n",
                clp->cl_rpcclient->cl_auth->au_ops->au_name,
                setclientid.sc_name_len, setclientid.sc_name);
-        status = rpc_call_sync(clp->cl_rpcclient, &msg, RPC_TASK_TIMEOUT);
+        task = rpc_run_task(&task_setup_data);
+        if (IS_ERR(task)) {
+                status = PTR_ERR(task);
+                goto out;
+        }
+        status = task->tk_status;
+        if (setclientid.sc_cred) {
+                clp->cl_acceptor = rpcauth_stringify_acceptor(setclientid.sc_cred);
+                put_rpccred(setclientid.sc_cred);
+        }
+        rpc_put_task(task);
+out:
        trace_nfs4_setclientid(clp, status);
        dprintk("NFS reply setclientid: %d\n", status);
        return status;
@@ -4975,6 +5068,9 @@ struct nfs4_delegreturndata {
        unsigned long timestamp;
        struct nfs_fattr fattr;
        int rpc_status;
+        struct inode *inode;
+        bool roc;
+        u32 roc_barrier;
 };
 static void nfs4_delegreturn_done(struct rpc_task *task, void *calldata)
@@ -4988,7 +5084,6 @@ static void nfs4_delegreturn_done(struct rpc_task *task, void *calldata)
        switch (task->tk_status) {
        case 0:
                renew_lease(data->res.server, data->timestamp);
-                break;
        case -NFS4ERR_ADMIN_REVOKED:
        case -NFS4ERR_DELEG_REVOKED:
        case -NFS4ERR_BAD_STATEID:
@@ -4996,6 +5091,8 @@ static void nfs4_delegreturn_done(struct rpc_task *task, void *calldata)
        case -NFS4ERR_STALE_STATEID:
        case -NFS4ERR_EXPIRED:
                task->tk_status = 0;
+                if (data->roc)
+                        pnfs_roc_set_barrier(data->inode, data->roc_barrier);
                break;
        default:
                if (nfs4_async_handle_error(task, data->res.server, NULL) ==
@@ -5009,6 +5106,10 @@ static void nfs4_delegreturn_done(struct rpc_task *task, void *calldata)
 static void nfs4_delegreturn_release(void *calldata)
 {
+        struct nfs4_delegreturndata *data = calldata;
+        if (data->roc)
+                pnfs_roc_release(data->inode);
        kfree(calldata);
 }
@@ -5018,6 +5119,10 @@ static void nfs4_delegreturn_prepare(struct rpc_task *task, void *data)
        d_data = (struct nfs4_delegreturndata *)data;
+        if (d_data->roc &&
+            pnfs_roc_drain(d_data->inode, &d_data->roc_barrier, task))
+                return;
        nfs4_setup_sequence(d_data->res.server,
                        &d_data->args.seq_args,
                        &d_data->res.seq_res,
@@ -5061,6 +5166,9 @@ static int _nfs4_proc_delegreturn(struct inode *inode, struct rpc_cred *cred, co
        nfs_fattr_init(data->res.fattr);
        data->timestamp = jiffies;
        data->rpc_status = 0;
+        data->inode = inode;
+        data->roc = list_empty(&NFS_I(inode)->open_files) ?
+                    pnfs_roc(inode) : false;
        task_setup_data.callback_data = data;
        msg.rpc_argp = &data->args;
@@ -5834,8 +5942,10 @@ struct nfs_release_lockowner_data {
 static void nfs4_release_lockowner_prepare(struct rpc_task *task, void *calldata)
 {
        struct nfs_release_lockowner_data *data = calldata;
-        nfs40_setup_sequence(data->server,
+        struct nfs_server *server = data->server;
-                                &data->args.seq_args, &data->res.seq_res, task);
+        nfs40_setup_sequence(server, &data->args.seq_args,
+                                &data->res.seq_res, task);
+        data->args.lock_owner.clientid = server->nfs_client->cl_clientid;
        data->timestamp = jiffies;
 }
@@ -5852,6 +5962,8 @@ static void nfs4_release_lockowner_done(struct rpc_task *task, void *calldata)
                break;
        case -NFS4ERR_STALE_CLIENTID:
        case -NFS4ERR_EXPIRED:
+                nfs4_schedule_lease_recovery(server->nfs_client);
+                break;
        case -NFS4ERR_LEASE_MOVED:
        case -NFS4ERR_DELAY:
                if (nfs4_async_handle_error(task, server, NULL) == -EAGAIN)
@@ -5872,7 +5984,8 @@ static const struct rpc_call_ops nfs4_release_lockowner_ops = {
        .rpc_release = nfs4_release_lockowner_release,
 };
-static int nfs4_release_lockowner(struct nfs_server *server, struct nfs4_lock_state *lsp)
+static void
+nfs4_release_lockowner(struct nfs_server *server, struct nfs4_lock_state *lsp)
 {
        struct nfs_release_lockowner_data *data;
        struct rpc_message msg = {
@@ -5880,11 +5993,11 @@ static int nfs4_release_lockowner(struct nfs_server *server, struct nfs4_lock_st
        };
        if (server->nfs_client->cl_mvops->minor_version != 0)
-                return -EINVAL;
+                return;
        data = kmalloc(sizeof(*data), GFP_NOFS);
        if (!data)
-                return -ENOMEM;
+                return;
        data->lsp = lsp;
        data->server = server;
        data->args.lock_owner.clientid = server->nfs_client->cl_clientid;
@@ -5895,7 +6008,6 @@ static int nfs4_release_lockowner(struct nfs_server *server, struct nfs4_lock_st
        msg.rpc_resp = &data->res;
        nfs4_init_sequence(&data->args.seq_args, &data->res.seq_res, 0);
        rpc_call_async(server->client, &msg, 0, &nfs4_release_lockowner_ops, data);
-        return 0;
 }
 #define XATTR_NAME_NFSV4_ACL "system.nfs4_acl"
@@ -8182,7 +8294,8 @@ static int nfs41_free_stateid(struct nfs_server *server,
        return ret;
 }
-static int nfs41_free_lock_state(struct nfs_server *server, struct nfs4_lock_state *lsp)
+static void
+nfs41_free_lock_state(struct nfs_server *server, struct nfs4_lock_state *lsp)
 {
        struct rpc_task *task;
        struct rpc_cred *cred = lsp->ls_state->owner->so_cred;
@@ -8190,9 +8303,8 @@ static int nfs41_free_lock_state(struct nfs_server *server, struct nfs4_lock_sta
        task = _nfs41_free_stateid(server, &lsp->ls_stateid, cred, false);
        nfs4_free_lock_state(server, lsp);
        if (IS_ERR(task))
-                return PTR_ERR(task);
+                return;
        rpc_put_task(task);
-        return 0;
 }
 static bool nfs41_match_stateid(const nfs4_stateid *s1,
diff --git a/fs/nfs/nfs4state.c b/fs/nfs/nfs4state.c
index 42f121182167..a043f618cd5a 100644
--- a/fs/nfs/nfs4state.c
+++ b/fs/nfs/nfs4state.c
@@ -787,33 +787,36 @@ void nfs4_close_sync(struct nfs4_state *state, fmode_t fmode)
 * that is compatible with current->files
 */
 static struct nfs4_lock_state *
-__nfs4_find_lock_state(struct nfs4_state *state, fl_owner_t fl_owner, pid_t fl_pid, unsigned int type)
+__nfs4_find_lock_state(struct nfs4_state *state, fl_owner_t fl_owner)
 {
        struct nfs4_lock_state *pos;
        list_for_each_entry(pos, &state->lock_states, ls_locks) {
-                if (type != NFS4_ANY_LOCK_TYPE && pos->ls_owner.lo_type != type)
+                if (pos->ls_owner != fl_owner)
                        continue;
-                switch (pos->ls_owner.lo_type) {
-                case NFS4_POSIX_LOCK_TYPE:
-                        if (pos->ls_owner.lo_u.posix_owner != fl_owner)
-                                continue;
-                        break;
-                case NFS4_FLOCK_LOCK_TYPE:
-                        if (pos->ls_owner.lo_u.flock_owner != fl_pid)
-                                continue;
-                }
                atomic_inc(&pos->ls_count);
                return pos;
        }
        return NULL;
 }
+static void
+free_lock_state_work(struct work_struct *work)
+{
+        struct nfs4_lock_state *lsp = container_of(work,
+                                        struct nfs4_lock_state, ls_release);
+        struct nfs4_state *state = lsp->ls_state;
+        struct nfs_server *server = state->owner->so_server;
+        struct nfs_client *clp = server->nfs_client;
+        clp->cl_mvops->free_lock_state(server, lsp);
+}
 /*
 * Return a compatible lock_state. If no initialized lock_state structure
 * exists, return an uninitialized one.
 *
 */
-static struct nfs4_lock_state *nfs4_alloc_lock_state(struct nfs4_state *state, fl_owner_t fl_owner, pid_t fl_pid, unsigned int type)
+static struct nfs4_lock_state *nfs4_alloc_lock_state(struct nfs4_state *state, fl_owner_t fl_owner)
 {
        struct nfs4_lock_state *lsp;
        struct nfs_server *server = state->owner->so_server;
@@ -824,21 +827,12 @@ static struct nfs4_lock_state *nfs4_alloc_lock_state(struct nfs4_state *state, f
        nfs4_init_seqid_counter(&lsp->ls_seqid);
        atomic_set(&lsp->ls_count, 1);
        lsp->ls_state = state;
-        lsp->ls_owner.lo_type = type;
+        lsp->ls_owner = fl_owner;
-        switch (lsp->ls_owner.lo_type) {
-        case NFS4_FLOCK_LOCK_TYPE:
-                lsp->ls_owner.lo_u.flock_owner = fl_pid;
-                break;
-        case NFS4_POSIX_LOCK_TYPE:
-                lsp->ls_owner.lo_u.posix_owner = fl_owner;
-                break;
-        default:
-                goto out_free;
-        }
        lsp->ls_seqid.owner_id = ida_simple_get(&server->lockowner_id, 0, 0, GFP_NOFS);
        if (lsp->ls_seqid.owner_id < 0)
                goto out_free;
        INIT_LIST_HEAD(&lsp->ls_locks);
+        INIT_WORK(&lsp->ls_release, free_lock_state_work);
        return lsp;
 out_free:
        kfree(lsp);
@@ -857,13 +851,13 @@ void nfs4_free_lock_state(struct nfs_server *server, struct nfs4_lock_state *lsp
 * exists, return an uninitialized one.
 *
 */
-static struct nfs4_lock_state *nfs4_get_lock_state(struct nfs4_state *state, fl_owner_t owner, pid_t pid, unsigned int type)
+static struct nfs4_lock_state *nfs4_get_lock_state(struct nfs4_state *state, fl_owner_t owner)
 {
        struct nfs4_lock_state *lsp, *new = NULL;
        
        for(;;) {
                spin_lock(&state->state_lock);
-                lsp = __nfs4_find_lock_state(state, owner, pid, type);
+                lsp = __nfs4_find_lock_state(state, owner);
                if (lsp != NULL)
                        break;
                if (new != NULL) {
@@ -874,7 +868,7 @@ static struct nfs4_lock_state *nfs4_get_lock_state(struct nfs4_state *state, fl_
                        break;
                }
                spin_unlock(&state->state_lock);
-                new = nfs4_alloc_lock_state(state, owner, pid, type);
+                new = nfs4_alloc_lock_state(state, owner);
                if (new == NULL)
                        return NULL;
        }
@@ -902,13 +896,12 @@ void nfs4_put_lock_state(struct nfs4_lock_state *lsp)
        if (list_empty(&state->lock_states))
                clear_bit(LK_STATE_IN_USE, &state->flags);
        spin_unlock(&state->state_lock);
-        server = state->owner->so_server;
+        if (test_bit(NFS_LOCK_INITIALIZED, &lsp->ls_flags))
-        if (test_bit(NFS_LOCK_INITIALIZED, &lsp->ls_flags)) {
+                queue_work(nfsiod_workqueue, &lsp->ls_release);
-                struct nfs_client *clp = server->nfs_client;
+        else {
+                server = state->owner->so_server;
-                clp->cl_mvops->free_lock_state(server, lsp);
-        } else
                nfs4_free_lock_state(server, lsp);
+        }
 }
 static void nfs4_fl_copy_lock(struct file_lock *dst, struct file_lock *src)
@@ -935,13 +928,7 @@ int nfs4_set_lock_state(struct nfs4_state *state, struct file_lock *fl)
        if (fl->fl_ops != NULL)
                return 0;
-        if (fl->fl_flags & FL_POSIX)
+        lsp = nfs4_get_lock_state(state, fl->fl_owner);
-                lsp = nfs4_get_lock_state(state, fl->fl_owner, 0, NFS4_POSIX_LOCK_TYPE);
-        else if (fl->fl_flags & FL_FLOCK)
-                lsp = nfs4_get_lock_state(state, NULL, fl->fl_pid,
-                                NFS4_FLOCK_LOCK_TYPE);
-        else
-                return -EINVAL;
        if (lsp == NULL)
                return -ENOMEM;
        fl->fl_u.nfs4_fl.owner = lsp;
@@ -955,7 +942,6 @@ static int nfs4_copy_lock_stateid(nfs4_stateid *dst,
 {
        struct nfs4_lock_state *lsp;
        fl_owner_t fl_owner;
-        pid_t fl_pid;
        int ret = -ENOENT;
@@ -966,9 +952,8 @@ static int nfs4_copy_lock_stateid(nfs4_stateid *dst,
                goto out;
        fl_owner = lockowner->l_owner;
-        fl_pid = lockowner->l_pid;
        spin_lock(&state->state_lock);
-        lsp = __nfs4_find_lock_state(state, fl_owner, fl_pid, NFS4_ANY_LOCK_TYPE);
+        lsp = __nfs4_find_lock_state(state, fl_owner);
        if (lsp && test_bit(NFS_LOCK_LOST, &lsp->ls_flags))
                ret = -EIO;
        else if (lsp != NULL && test_bit(NFS_LOCK_INITIALIZED, &lsp->ls_flags) != 0) {
diff --git a/fs/nfs/nfs4trace.h b/fs/nfs/nfs4trace.h
index 0a744f3a86f6..1c32adbe728d 100644
--- a/fs/nfs/nfs4trace.h
+++ b/fs/nfs/nfs4trace.h
@@ -932,11 +932,11 @@ DEFINE_NFS4_IDMAP_EVENT(nfs4_map_gid_to_group);
 DECLARE_EVENT_CLASS(nfs4_read_event,
                TP_PROTO(
-                        const struct nfs_pgio_data *data,
+                        const struct nfs_pgio_header *hdr,
                        int error
                ),
-                TP_ARGS(data, error),
+                TP_ARGS(hdr, error),
                TP_STRUCT__entry(
                        __field(dev_t, dev)
@@ -948,12 +948,12 @@ DECLARE_EVENT_CLASS(nfs4_read_event,
                ),
                TP_fast_assign(
-                        const struct inode *inode = data->header->inode;
+                        const struct inode *inode = hdr->inode;
                        __entry->dev = inode->i_sb->s_dev;
                        __entry->fileid = NFS_FILEID(inode);
                        __entry->fhandle = nfs_fhandle_hash(NFS_FH(inode));
-                        __entry->offset = data->args.offset;
+                        __entry->offset = hdr->args.offset;
-                        __entry->count = data->args.count;
+                        __entry->count = hdr->args.count;
                        __entry->error = error;
                ),
@@ -972,10 +972,10 @@ DECLARE_EVENT_CLASS(nfs4_read_event,
 #define DEFINE_NFS4_READ_EVENT(name) \
        DEFINE_EVENT(nfs4_read_event, name, \
                        TP_PROTO( \
-                                const struct nfs_pgio_data *data, \
+                                const struct nfs_pgio_header *hdr, \
                                int error \
                        ), \
-                        TP_ARGS(data, error))
+                        TP_ARGS(hdr, error))
 DEFINE_NFS4_READ_EVENT(nfs4_read);
 #ifdef CONFIG_NFS_V4_1
 DEFINE_NFS4_READ_EVENT(nfs4_pnfs_read);
@@ -983,11 +983,11 @@ DEFINE_NFS4_READ_EVENT(nfs4_pnfs_read);
 DECLARE_EVENT_CLASS(nfs4_write_event,
                TP_PROTO(
-                        const struct nfs_pgio_data *data,
+                        const struct nfs_pgio_header *hdr,
                        int error
                ),
-                TP_ARGS(data, error),
+                TP_ARGS(hdr, error),
                TP_STRUCT__entry(
                        __field(dev_t, dev)
@@ -999,12 +999,12 @@ DECLARE_EVENT_CLASS(nfs4_write_event,
                ),
                TP_fast_assign(
-                        const struct inode *inode = data->header->inode;
+                        const struct inode *inode = hdr->inode;
                        __entry->dev = inode->i_sb->s_dev;
                        __entry->fileid = NFS_FILEID(inode);
                        __entry->fhandle = nfs_fhandle_hash(NFS_FH(inode));
-                        __entry->offset = data->args.offset;
+                        __entry->offset = hdr->args.offset;
-                        __entry->count = data->args.count;
+                        __entry->count = hdr->args.count;
                        __entry->error = error;
                ),
@@ -1024,10 +1024,10 @@ DECLARE_EVENT_CLASS(nfs4_write_event,
 #define DEFINE_NFS4_WRITE_EVENT(name) \
        DEFINE_EVENT(nfs4_write_event, name, \
                        TP_PROTO( \
-                                const struct nfs_pgio_data *data, \
+                                const struct nfs_pgio_header *hdr, \
                                int error \
                        ), \
-                        TP_ARGS(data, error))
+                        TP_ARGS(hdr, error))
 DEFINE_NFS4_WRITE_EVENT(nfs4_write);
 #ifdef CONFIG_NFS_V4_1
 DEFINE_NFS4_WRITE_EVENT(nfs4_pnfs_write);
diff --git a/fs/nfs/nfs4xdr.c b/fs/nfs/nfs4xdr.c
index 939ae606cfa4..e13b59d8d9aa 100644
--- a/fs/nfs/nfs4xdr.c
+++ b/fs/nfs/nfs4xdr.c
@@ -7092,7 +7092,7 @@ static int nfs4_xdr_dec_reclaim_complete(struct rpc_rqst *rqstp,
        if (!status)
                status = decode_sequence(xdr, &res->seq_res, rqstp);
        if (!status)
-                status = decode_reclaim_complete(xdr, (void *)NULL);
+                status = decode_reclaim_complete(xdr, NULL);
        return status;
 }
diff --git a/fs/nfs/objlayout/objio_osd.c b/fs/nfs/objlayout/objio_osd.c
index 611320753db2..ae05278b3761 100644
--- a/fs/nfs/objlayout/objio_osd.c
+++ b/fs/nfs/objlayout/objio_osd.c
@@ -439,22 +439,21 @@ static void _read_done(struct ore_io_state *ios, void *private)
        objlayout_read_done(&objios->oir, status, objios->sync);
 }
-int objio_read_pagelist(struct nfs_pgio_data *rdata)
+int objio_read_pagelist(struct nfs_pgio_header *hdr)
 {
-        struct nfs_pgio_header *hdr = rdata->header;
        struct objio_state *objios;
        int ret;
        ret = objio_alloc_io_state(NFS_I(hdr->inode)->layout, true,
-                        hdr->lseg, rdata->args.pages, rdata->args.pgbase,
+                        hdr->lseg, hdr->args.pages, hdr->args.pgbase,
-                        rdata->args.offset, rdata->args.count, rdata,
+                        hdr->args.offset, hdr->args.count, hdr,
                        GFP_KERNEL, &objios);
        if (unlikely(ret))
                return ret;
        objios->ios->done = _read_done;
        dprintk("%s: offset=0x%llx length=0x%x\n", __func__,
-                rdata->args.offset, rdata->args.count);
+                hdr->args.offset, hdr->args.count);
        ret = ore_read(objios->ios);
        if (unlikely(ret))
                objio_free_result(&objios->oir);
@@ -487,11 +486,11 @@ static void _write_done(struct ore_io_state *ios, void *private)
 static struct page *__r4w_get_page(void *priv, u64 offset, bool *uptodate)
 {
        struct objio_state *objios = priv;
-        struct nfs_pgio_data *wdata = objios->oir.rpcdata;
+        struct nfs_pgio_header *hdr = objios->oir.rpcdata;
-        struct address_space *mapping = wdata->header->inode->i_mapping;
+        struct address_space *mapping = hdr->inode->i_mapping;
        pgoff_t index = offset / PAGE_SIZE;
        struct page *page;
-        loff_t i_size = i_size_read(wdata->header->inode);
+        loff_t i_size = i_size_read(hdr->inode);
        if (offset >= i_size) {
                *uptodate = true;
@@ -531,15 +530,14 @@ static const struct _ore_r4w_op _r4w_op = {
        .put_page = &__r4w_put_page,
 };
-int objio_write_pagelist(struct nfs_pgio_data *wdata, int how)
+int objio_write_pagelist(struct nfs_pgio_header *hdr, int how)
 {
-        struct nfs_pgio_header *hdr = wdata->header;
        struct objio_state *objios;
        int ret;
        ret = objio_alloc_io_state(NFS_I(hdr->inode)->layout, false,
-                        hdr->lseg, wdata->args.pages, wdata->args.pgbase,
+                        hdr->lseg, hdr->args.pages, hdr->args.pgbase,
-                        wdata->args.offset, wdata->args.count, wdata, GFP_NOFS,
+                        hdr->args.offset, hdr->args.count, hdr, GFP_NOFS,
                        &objios);
        if (unlikely(ret))
                return ret;
@@ -551,7 +549,7 @@ int objio_write_pagelist(struct nfs_pgio_data *wdata, int how)
                objios->ios->done = _write_done;
        dprintk("%s: offset=0x%llx length=0x%x\n", __func__,
-                wdata->args.offset, wdata->args.count);
+                hdr->args.offset, hdr->args.count);
        ret = ore_write(objios->ios);
        if (unlikely(ret)) {
                objio_free_result(&objios->oir);
diff --git a/fs/nfs/objlayout/objlayout.c b/fs/nfs/objlayout/objlayout.c
index 765d3f54e986..697a16d11fac 100644
--- a/fs/nfs/objlayout/objlayout.c
+++ b/fs/nfs/objlayout/objlayout.c
@@ -229,36 +229,36 @@ objlayout_io_set_result(struct objlayout_io_res *oir, unsigned index,
 static void _rpc_read_complete(struct work_struct *work)
 {
        struct rpc_task *task;
-        struct nfs_pgio_data *rdata;
+        struct nfs_pgio_header *hdr;
        dprintk("%s enter\n", __func__);
        task = container_of(work, struct rpc_task, u.tk_work);
-        rdata = container_of(task, struct nfs_pgio_data, task);
+        hdr = container_of(task, struct nfs_pgio_header, task);
-        pnfs_ld_read_done(rdata);
+        pnfs_ld_read_done(hdr);
 }
 void
 objlayout_read_done(struct objlayout_io_res *oir, ssize_t status, bool sync)
 {
-        struct nfs_pgio_data *rdata = oir->rpcdata;
+        struct nfs_pgio_header *hdr = oir->rpcdata;
-        oir->status = rdata->task.tk_status = status;
+        oir->status = hdr->task.tk_status = status;
        if (status >= 0)
-                rdata->res.count = status;
+                hdr->res.count = status;
        else
-                rdata->header->pnfs_error = status;
+                hdr->pnfs_error = status;
        objlayout_iodone(oir);
        /* must not use oir after this point */
        dprintk("%s: Return status=%zd eof=%d sync=%d\n", __func__,
-                status, rdata->res.eof, sync);
+                status, hdr->res.eof, sync);
        if (sync)
-                pnfs_ld_read_done(rdata);
+                pnfs_ld_read_done(hdr);
        else {
-                INIT_WORK(&rdata->task.u.tk_work, _rpc_read_complete);
+                INIT_WORK(&hdr->task.u.tk_work, _rpc_read_complete);
-                schedule_work(&rdata->task.u.tk_work);
+                schedule_work(&hdr->task.u.tk_work);
        }
 }
@@ -266,12 +266,11 @@ objlayout_read_done(struct objlayout_io_res *oir, ssize_t status, bool sync)
 * Perform sync or async reads.
 */
 enum pnfs_try_status
-objlayout_read_pagelist(struct nfs_pgio_data *rdata)
+objlayout_read_pagelist(struct nfs_pgio_header *hdr)
 {
-        struct nfs_pgio_header *hdr = rdata->header;
        struct inode *inode = hdr->inode;
-        loff_t offset = rdata->args.offset;
+        loff_t offset = hdr->args.offset;
-        size_t count = rdata->args.count;
+        size_t count = hdr->args.count;
        int err;
        loff_t eof;
@@ -279,23 +278,23 @@ objlayout_read_pagelist(struct nfs_pgio_data *rdata)
        if (unlikely(offset + count > eof)) {
                if (offset >= eof) {
                        err = 0;
-                        rdata->res.count = 0;
+                        hdr->res.count = 0;
-                        rdata->res.eof = 1;
+                        hdr->res.eof = 1;
                        /*FIXME: do we need to call pnfs_ld_read_done() */
                        goto out;
                }
                count = eof - offset;
        }
-        rdata->res.eof = (offset + count) >= eof;
+        hdr->res.eof = (offset + count) >= eof;
-        _fix_verify_io_params(hdr->lseg, &rdata->args.pages,
+        _fix_verify_io_params(hdr->lseg, &hdr->args.pages,
-                              &rdata->args.pgbase,
+                              &hdr->args.pgbase,
-                              rdata->args.offset, rdata->args.count);
+                              hdr->args.offset, hdr->args.count);
        dprintk("%s: inode(%lx) offset 0x%llx count 0x%Zx eof=%d\n",
-                __func__, inode->i_ino, offset, count, rdata->res.eof);
+                __func__, inode->i_ino, offset, count, hdr->res.eof);
-        err = objio_read_pagelist(rdata);
+        err = objio_read_pagelist(hdr);
 out:
        if (unlikely(err)) {
                hdr->pnfs_error = err;
@@ -312,38 +311,38 @@ objlayout_read_pagelist(struct nfs_pgio_data *rdata)
 static void _rpc_write_complete(struct work_struct *work)
 {
        struct rpc_task *task;
-        struct nfs_pgio_data *wdata;
+        struct nfs_pgio_header *hdr;
        dprintk("%s enter\n", __func__);
        task = container_of(work, struct rpc_task, u.tk_work);
-        wdata = container_of(task, struct nfs_pgio_data, task);
+        hdr = container_of(task, struct nfs_pgio_header, task);
-        pnfs_ld_write_done(wdata);
+        pnfs_ld_write_done(hdr);
 }
 void
 objlayout_write_done(struct objlayout_io_res *oir, ssize_t status, bool sync)
 {
-        struct nfs_pgio_data *wdata = oir->rpcdata;
+        struct nfs_pgio_header *hdr = oir->rpcdata;
-        oir->status = wdata->task.tk_status = status;
+        oir->status = hdr->task.tk_status = status;
        if (status >= 0) {
-                wdata->res.count = status;
+                hdr->res.count = status;
-                wdata->verf.committed = oir->committed;
+                hdr->verf.committed = oir->committed;
        } else {
-                wdata->header->pnfs_error = status;
+                hdr->pnfs_error = status;
        }
        objlayout_iodone(oir);
        /* must not use oir after this point */
        dprintk("%s: Return status %zd committed %d sync=%d\n", __func__,
-                status, wdata->verf.committed, sync);
+                status, hdr->verf.committed, sync);
        if (sync)
-                pnfs_ld_write_done(wdata);
+                pnfs_ld_write_done(hdr);
        else {
-                INIT_WORK(&wdata->task.u.tk_work, _rpc_write_complete);
+                INIT_WORK(&hdr->task.u.tk_work, _rpc_write_complete);
-                schedule_work(&wdata->task.u.tk_work);
+                schedule_work(&hdr->task.u.tk_work);
        }
 }
@@ -351,17 +350,15 @@ objlayout_write_done(struct objlayout_io_res *oir, ssize_t status, bool sync)
 * Perform sync or async writes.
 */
 enum pnfs_try_status
-objlayout_write_pagelist(struct nfs_pgio_data *wdata,
+objlayout_write_pagelist(struct nfs_pgio_header *hdr, int how)
-                         int how)
 {
-        struct nfs_pgio_header *hdr = wdata->header;
        int err;
-        _fix_verify_io_params(hdr->lseg, &wdata->args.pages,
+        _fix_verify_io_params(hdr->lseg, &hdr->args.pages,
-                              &wdata->args.pgbase,
+                              &hdr->args.pgbase,
-                              wdata->args.offset, wdata->args.count);
+                              hdr->args.offset, hdr->args.count);
-        err = objio_write_pagelist(wdata, how);
+        err = objio_write_pagelist(hdr, how);
        if (unlikely(err)) {
                hdr->pnfs_error = err;
                dprintk("%s: Returned Error %d\n", __func__, err);
diff --git a/fs/nfs/objlayout/objlayout.h b/fs/nfs/objlayout/objlayout.h
index 01e041029a6c..fd13f1d2f136 100644
--- a/fs/nfs/objlayout/objlayout.h
+++ b/fs/nfs/objlayout/objlayout.h
@@ -119,8 +119,8 @@ extern void objio_free_lseg(struct pnfs_layout_segment *lseg);
 */
 extern void objio_free_result(struct objlayout_io_res *oir);
-extern int objio_read_pagelist(struct nfs_pgio_data *rdata);
+extern int objio_read_pagelist(struct nfs_pgio_header *rdata);
-extern int objio_write_pagelist(struct nfs_pgio_data *wdata, int how);
+extern int objio_write_pagelist(struct nfs_pgio_header *wdata, int how);
 /*
 * callback API
@@ -168,10 +168,10 @@ extern struct pnfs_layout_segment *objlayout_alloc_lseg(
 extern void objlayout_free_lseg(struct pnfs_layout_segment *);
 extern enum pnfs_try_status objlayout_read_pagelist(
-        struct nfs_pgio_data *);
+        struct nfs_pgio_header *);
 extern enum pnfs_try_status objlayout_write_pagelist(
-        struct nfs_pgio_data *,
+        struct nfs_pgio_header *,
        int how);
 extern void objlayout_encode_layoutcommit(
diff --git a/fs/nfs/pagelist.c b/fs/nfs/pagelist.c
index 0be5050638f7..ba491926df5f 100644
--- a/fs/nfs/pagelist.c
+++ b/fs/nfs/pagelist.c
@@ -141,16 +141,24 @@ nfs_iocounter_wait(struct nfs_io_counter *c)
 * @req - request in group that is to be locked
 *
 * this lock must be held if modifying the page group list
+ *
+ * returns result from wait_on_bit_lock: 0 on success, < 0 on error
 */
-void
+int
-nfs_page_group_lock(struct nfs_page *req)
+nfs_page_group_lock(struct nfs_page *req, bool wait)
 {
        struct nfs_page *head = req->wb_head;
+        int ret;
        WARN_ON_ONCE(head != head->wb_head);
-        wait_on_bit_lock(&head->wb_flags, PG_HEADLOCK,
+        do {
+                ret = wait_on_bit_lock(&head->wb_flags, PG_HEADLOCK,
                        TASK_UNINTERRUPTIBLE);
+        } while (wait && ret != 0);
+        WARN_ON_ONCE(ret > 0);
+        return ret;
 }
 /*
@@ -211,7 +219,7 @@ bool nfs_page_group_sync_on_bit(struct nfs_page *req, unsigned int bit)
 {
        bool ret;
-        nfs_page_group_lock(req);
+        nfs_page_group_lock(req, true);
        ret = nfs_page_group_sync_on_bit_locked(req, bit);
        nfs_page_group_unlock(req);
@@ -454,123 +462,72 @@ size_t nfs_generic_pg_test(struct nfs_pageio_descriptor *desc,
 }
 EXPORT_SYMBOL_GPL(nfs_generic_pg_test);
-static inline struct nfs_rw_header *NFS_RW_HEADER(struct nfs_pgio_header *hdr)
+struct nfs_pgio_header *nfs_pgio_header_alloc(const struct nfs_rw_ops *ops)
-{
-        return container_of(hdr, struct nfs_rw_header, header);
-}
-/**
- * nfs_rw_header_alloc - Allocate a header for a read or write
- * @ops: Read or write function vector
- */
-struct nfs_rw_header *nfs_rw_header_alloc(const struct nfs_rw_ops *ops)
 {
-        struct nfs_rw_header *header = ops->rw_alloc_header();
+        struct nfs_pgio_header *hdr = ops->rw_alloc_header();
-        if (header) {
-                struct nfs_pgio_header *hdr = &header->header;
+        if (hdr) {
                INIT_LIST_HEAD(&hdr->pages);
                spin_lock_init(&hdr->lock);
-                atomic_set(&hdr->refcnt, 0);
                hdr->rw_ops = ops;
        }
-        return header;
+        return hdr;
 }
-EXPORT_SYMBOL_GPL(nfs_rw_header_alloc);
+EXPORT_SYMBOL_GPL(nfs_pgio_header_alloc);
 /*
- * nfs_rw_header_free - Free a read or write header
+ * nfs_pgio_header_free - Free a read or write header
 * @hdr: The header to free
 */
-void nfs_rw_header_free(struct nfs_pgio_header *hdr)
+void nfs_pgio_header_free(struct nfs_pgio_header *hdr)
 {
-        hdr->rw_ops->rw_free_header(NFS_RW_HEADER(hdr));
+        hdr->rw_ops->rw_free_header(hdr);
 }
-EXPORT_SYMBOL_GPL(nfs_rw_header_free);
+EXPORT_SYMBOL_GPL(nfs_pgio_header_free);
 /**
- * nfs_pgio_data_alloc - Allocate pageio data
+ * nfs_pgio_data_destroy - make @hdr suitable for reuse
- * @hdr: The header making a request
+ *
- * @pagecount: Number of pages to create
+ * Frees memory and releases refs from nfs_generic_pgio, so that it may
- */
+ * be called again.
-static struct nfs_pgio_data *nfs_pgio_data_alloc(struct nfs_pgio_header *hdr,
+ *
-                                                 unsigned int pagecount)
+ * @hdr: A header that has had nfs_generic_pgio called
-{
-        struct nfs_pgio_data *data, *prealloc;
-        prealloc = &NFS_RW_HEADER(hdr)->rpc_data;
-        if (prealloc->header == NULL)
-                data = prealloc;
-        else
-                data = kzalloc(sizeof(*data), GFP_KERNEL);
-        if (!data)
-                goto out;
-        if (nfs_pgarray_set(&data->pages, pagecount)) {
-                data->header = hdr;
-                atomic_inc(&hdr->refcnt);
-        } else {
-                if (data != prealloc)
-                        kfree(data);
-                data = NULL;
-        }
-out:
-        return data;
-}
-/**
- * nfs_pgio_data_release - Properly free pageio data
- * @data: The data to release
 */
-void nfs_pgio_data_release(struct nfs_pgio_data *data)
+void nfs_pgio_data_destroy(struct nfs_pgio_header *hdr)
 {
-        struct nfs_pgio_header *hdr = data->header;
+        put_nfs_open_context(hdr->args.context);
-        struct nfs_rw_header *pageio_header = NFS_RW_HEADER(hdr);
+        if (hdr->page_array.pagevec != hdr->page_array.page_array)
+                kfree(hdr->page_array.pagevec);
-        put_nfs_open_context(data->args.context);
-        if (data->pages.pagevec != data->pages.page_array)
-                kfree(data->pages.pagevec);
-        if (data == &pageio_header->rpc_data) {
-                data->header = NULL;
-                data = NULL;
-        }
-        if (atomic_dec_and_test(&hdr->refcnt))
-                hdr->completion_ops->completion(hdr);
-        /* Note: we only free the rpc_task after callbacks are done.
-         * See the comment in rpc_free_task() for why
-         */
-        kfree(data);
 }
-EXPORT_SYMBOL_GPL(nfs_pgio_data_release);
+EXPORT_SYMBOL_GPL(nfs_pgio_data_destroy);
 /**
 * nfs_pgio_rpcsetup - Set up arguments for a pageio call
- * @data: The pageio data
+ * @hdr: The pageio hdr
 * @count: Number of bytes to read
 * @offset: Initial offset
 * @how: How to commit data (writes only)
 * @cinfo: Commit information for the call (writes only)
 */
-static void nfs_pgio_rpcsetup(struct nfs_pgio_data *data,
+static void nfs_pgio_rpcsetup(struct nfs_pgio_header *hdr,
                              unsigned int count, unsigned int offset,
                              int how, struct nfs_commit_info *cinfo)
 {
-        struct nfs_page *req = data->header->req;
+        struct nfs_page *req = hdr->req;
        /* Set up the RPC argument and reply structs
-         * NB: take care not to mess about with data->commit et al. */
+         * NB: take care not to mess about with hdr->commit et al. */
-        data->args.fh     = NFS_FH(data->header->inode);
+        hdr->args.fh     = NFS_FH(hdr->inode);
-        data->args.offset = req_offset(req) + offset;
+        hdr->args.offset = req_offset(req) + offset;
        /* pnfs_set_layoutcommit needs this */
-        data->mds_offset = data->args.offset;
+        hdr->mds_offset = hdr->args.offset;
-        data->args.pgbase = req->wb_pgbase + offset;
+        hdr->args.pgbase = req->wb_pgbase + offset;
-        data->args.pages  = data->pages.pagevec;
+        hdr->args.pages  = hdr->page_array.pagevec;
-        data->args.count  = count;
+        hdr->args.count  = count;
-        data->args.context = get_nfs_open_context(req->wb_context);
+        hdr->args.context = get_nfs_open_context(req->wb_context);
-        data->args.lock_context = req->wb_lock_context;
+        hdr->args.lock_context = req->wb_lock_context;
-        data->args.stable  = NFS_UNSTABLE;
+        hdr->args.stable  = NFS_UNSTABLE;
        switch (how & (FLUSH_STABLE | FLUSH_COND_STABLE)) {
        case 0:
                break;
@@ -578,59 +535,59 @@ static void nfs_pgio_rpcsetup(struct nfs_pgio_data *data,
                if (nfs_reqs_to_commit(cinfo))
                        break;
        default:
-                data->args.stable = NFS_FILE_SYNC;
+                hdr->args.stable = NFS_FILE_SYNC;
        }
-        data->res.fattr   = &data->fattr;
+        hdr->res.fattr   = &hdr->fattr;
-        data->res.count   = count;
+        hdr->res.count   = count;
-        data->res.eof     = 0;
+        hdr->res.eof     = 0;
-        data->res.verf    = &data->verf;
+        hdr->res.verf    = &hdr->verf;
-        nfs_fattr_init(&data->fattr);
+        nfs_fattr_init(&hdr->fattr);
 }
 /**
- * nfs_pgio_prepare - Prepare pageio data to go over the wire
+ * nfs_pgio_prepare - Prepare pageio hdr to go over the wire
 * @task: The current task
- * @calldata: pageio data to prepare
+ * @calldata: pageio header to prepare
 */
 static void nfs_pgio_prepare(struct rpc_task *task, void *calldata)
 {
-        struct nfs_pgio_data *data = calldata;
+        struct nfs_pgio_header *hdr = calldata;
        int err;
-        err = NFS_PROTO(data->header->inode)->pgio_rpc_prepare(task, data);
+        err = NFS_PROTO(hdr->inode)->pgio_rpc_prepare(task, hdr);
        if (err)
                rpc_exit(task, err);
 }
-int nfs_initiate_pgio(struct rpc_clnt *clnt, struct nfs_pgio_data *data,
+int nfs_initiate_pgio(struct rpc_clnt *clnt, struct nfs_pgio_header *hdr,
                      const struct rpc_call_ops *call_ops, int how, int flags)
 {
        struct rpc_task *task;
        struct rpc_message msg = {
-                .rpc_argp = &data->args,
+                .rpc_argp = &hdr->args,
-                .rpc_resp = &data->res,
+                .rpc_resp = &hdr->res,
-                .rpc_cred = data->header->cred,
+                .rpc_cred = hdr->cred,
        };
        struct rpc_task_setup task_setup_data = {
                .rpc_client = clnt,
-                .task = &data->task,
+                .task = &hdr->task,
                .rpc_message = &msg,
                .callback_ops = call_ops,
-                .callback_data = data,
+                .callback_data = hdr,
                .workqueue = nfsiod_workqueue,
                .flags = RPC_TASK_ASYNC | flags,
        };
        int ret = 0;
-        data->header->rw_ops->rw_initiate(data, &msg, &task_setup_data, how);
+        hdr->rw_ops->rw_initiate(hdr, &msg, &task_setup_data, how);
        dprintk("NFS: %5u initiated pgio call "
                "(req %s/%llu, %u bytes @ offset %llu)\n",
-                data->task.tk_pid,
+                hdr->task.tk_pid,
-                data->header->inode->i_sb->s_id,
+                hdr->inode->i_sb->s_id,
-                (unsigned long long)NFS_FILEID(data->header->inode),
+                (unsigned long long)NFS_FILEID(hdr->inode),
-                data->args.count,
+                hdr->args.count,
-                (unsigned long long)data->args.offset);
+                (unsigned long long)hdr->args.offset);
        task = rpc_run_task(&task_setup_data);
        if (IS_ERR(task)) {
@@ -657,22 +614,23 @@ static int nfs_pgio_error(struct nfs_pageio_descriptor *desc,
                          struct nfs_pgio_header *hdr)
 {
        set_bit(NFS_IOHDR_REDO, &hdr->flags);
-        nfs_pgio_data_release(hdr->data);
+        nfs_pgio_data_destroy(hdr);
-        hdr->data = NULL;
+        hdr->completion_ops->completion(hdr);
        desc->pg_completion_ops->error_cleanup(&desc->pg_list);
        return -ENOMEM;
 }
 /**
 * nfs_pgio_release - Release pageio data
- * @calldata: The pageio data to release
+ * @calldata: The pageio header to release
 */
 static void nfs_pgio_release(void *calldata)
 {
-        struct nfs_pgio_data *data = calldata;
+        struct nfs_pgio_header *hdr = calldata;
-        if (data->header->rw_ops->rw_release)
+        if (hdr->rw_ops->rw_release)
-                data->header->rw_ops->rw_release(data);
+                hdr->rw_ops->rw_release(hdr);
-        nfs_pgio_data_release(data);
+        nfs_pgio_data_destroy(hdr);
+        hdr->completion_ops->completion(hdr);
 }
 /**
@@ -713,22 +671,22 @@ EXPORT_SYMBOL_GPL(nfs_pageio_init);
 /**
 * nfs_pgio_result - Basic pageio error handling
 * @task: The task that ran
- * @calldata: Pageio data to check
+ * @calldata: Pageio header to check
 */
 static void nfs_pgio_result(struct rpc_task *task, void *calldata)
 {
-        struct nfs_pgio_data *data = calldata;
+        struct nfs_pgio_header *hdr = calldata;
-        struct inode *inode = data->header->inode;
+        struct inode *inode = hdr->inode;
        dprintk("NFS: %s: %5u, (status %d)\n", __func__,
                task->tk_pid, task->tk_status);
-        if (data->header->rw_ops->rw_done(task, data, inode) != 0)
+        if (hdr->rw_ops->rw_done(task, hdr, inode) != 0)
                return;
        if (task->tk_status < 0)
-                nfs_set_pgio_error(data->header, task->tk_status, data->args.offset);
+                nfs_set_pgio_error(hdr, task->tk_status, hdr->args.offset);
        else
-                data->header->rw_ops->rw_result(task, data);
+                hdr->rw_ops->rw_result(task, hdr);
 }
 /*
@@ -744,17 +702,16 @@ int nfs_generic_pgio(struct nfs_pageio_descriptor *desc,
 {
        struct nfs_page         *req;
        struct page             **pages;
-        struct nfs_pgio_data    *data;
        struct list_head *head = &desc->pg_list;
        struct nfs_commit_info cinfo;
+        unsigned int pagecount;
-        data = nfs_pgio_data_alloc(hdr, nfs_page_array_len(desc->pg_base,
+        pagecount = nfs_page_array_len(desc->pg_base, desc->pg_count);
-                                                           desc->pg_count));
+        if (!nfs_pgarray_set(&hdr->page_array, pagecount))
-        if (!data)
                return nfs_pgio_error(desc, hdr);
        nfs_init_cinfo(&cinfo, desc->pg_inode, desc->pg_dreq);
-        pages = data->pages.pagevec;
+        pages = hdr->page_array.pagevec;
        while (!list_empty(head)) {
                req = nfs_list_entry(head->next);
                nfs_list_remove_request(req);
@@ -767,8 +724,7 @@ int nfs_generic_pgio(struct nfs_pageio_descriptor *desc,
                desc->pg_ioflags &= ~FLUSH_COND_STABLE;
        /* Set up the argument struct */
-        nfs_pgio_rpcsetup(data, desc->pg_count, 0, desc->pg_ioflags, &cinfo);
+        nfs_pgio_rpcsetup(hdr, desc->pg_count, 0, desc->pg_ioflags, &cinfo);
-        hdr->data = data;
        desc->pg_rpc_callops = &nfs_pgio_common_ops;
        return 0;
 }
@@ -776,25 +732,20 @@ EXPORT_SYMBOL_GPL(nfs_generic_pgio);
 static int nfs_generic_pg_pgios(struct nfs_pageio_descriptor *desc)
 {
-        struct nfs_rw_header *rw_hdr;
        struct nfs_pgio_header *hdr;
        int ret;
-        rw_hdr = nfs_rw_header_alloc(desc->pg_rw_ops);
+        hdr = nfs_pgio_header_alloc(desc->pg_rw_ops);
-        if (!rw_hdr) {
+        if (!hdr) {
                desc->pg_completion_ops->error_cleanup(&desc->pg_list);
                return -ENOMEM;
        }
-        hdr = &rw_hdr->header;
+        nfs_pgheader_init(desc, hdr, nfs_pgio_header_free);
-        nfs_pgheader_init(desc, hdr, nfs_rw_header_free);
-        atomic_inc(&hdr->refcnt);
        ret = nfs_generic_pgio(desc, hdr);
        if (ret == 0)
                ret = nfs_initiate_pgio(NFS_CLIENT(hdr->inode),
-                                        hdr->data, desc->pg_rpc_callops,
+                                        hdr, desc->pg_rpc_callops,
                                        desc->pg_ioflags, 0);
-        if (atomic_dec_and_test(&hdr->refcnt))
-                hdr->completion_ops->completion(hdr);
        return ret;
 }
@@ -907,8 +858,13 @@ static int __nfs_pageio_add_request(struct nfs_pageio_descriptor *desc,
        struct nfs_page *subreq;
        unsigned int bytes_left = 0;
        unsigned int offset, pgbase;
+        int ret;
-        nfs_page_group_lock(req);
+        ret = nfs_page_group_lock(req, false);
+        if (ret < 0) {
+                desc->pg_error = ret;
+                return 0;
+        }
        subreq = req;
        bytes_left = subreq->wb_bytes;
@@ -930,7 +886,11 @@ static int __nfs_pageio_add_request(struct nfs_pageio_descriptor *desc,
                        if (desc->pg_recoalesce)
                                return 0;
                        /* retry add_request for this subreq */
-                        nfs_page_group_lock(req);
+                        ret = nfs_page_group_lock(req, false);
+                        if (ret < 0) {
+                                desc->pg_error = ret;
+                                return 0;
+                        }
                        continue;
                }
@@ -1005,7 +965,38 @@ int nfs_pageio_add_request(struct nfs_pageio_descriptor *desc,
        } while (ret);
        return ret;
 }
-EXPORT_SYMBOL_GPL(nfs_pageio_add_request);
+/*
+ * nfs_pageio_resend - Transfer requests to new descriptor and resend
+ * @hdr - the pgio header to move request from
+ * @desc - the pageio descriptor to add requests to
+ *
+ * Try to move each request (nfs_page) from @hdr to @desc then attempt
+ * to send them.
+ *
+ * Returns 0 on success and < 0 on error.
+ */
+int nfs_pageio_resend(struct nfs_pageio_descriptor *desc,
+                      struct nfs_pgio_header *hdr)
+{
+        LIST_HEAD(failed);
+        desc->pg_dreq = hdr->dreq;
+        while (!list_empty(&hdr->pages)) {
+                struct nfs_page *req = nfs_list_entry(hdr->pages.next);
+                nfs_list_remove_request(req);
+                if (!nfs_pageio_add_request(desc, req))
+                        nfs_list_add_request(req, &failed);
+        }
+        nfs_pageio_complete(desc);
+        if (!list_empty(&failed)) {
+                list_move(&failed, &hdr->pages);
+                return -EIO;
+        }
+        return 0;
+}
+EXPORT_SYMBOL_GPL(nfs_pageio_resend);
 /**
 * nfs_pageio_complete - Complete I/O on an nfs_pageio_descriptor
@@ -1021,7 +1012,6 @@ void nfs_pageio_complete(struct nfs_pageio_descriptor *desc)
                        break;
        }
 }
-EXPORT_SYMBOL_GPL(nfs_pageio_complete);
 /**
 * nfs_pageio_cond_complete - Conditional I/O completion
diff --git a/fs/nfs/pnfs.c b/fs/nfs/pnfs.c
index a8914b335617..a3851debf8a2 100644
--- a/fs/nfs/pnfs.c
+++ b/fs/nfs/pnfs.c
@@ -361,6 +361,23 @@ pnfs_put_lseg(struct pnfs_layout_segment *lseg)
 }
 EXPORT_SYMBOL_GPL(pnfs_put_lseg);
+static void pnfs_put_lseg_async_work(struct work_struct *work)
+{
+        struct pnfs_layout_segment *lseg;
+        lseg = container_of(work, struct pnfs_layout_segment, pls_work);
+        pnfs_put_lseg(lseg);
+}
+void
+pnfs_put_lseg_async(struct pnfs_layout_segment *lseg)
+{
+        INIT_WORK(&lseg->pls_work, pnfs_put_lseg_async_work);
+        schedule_work(&lseg->pls_work);
+}
+EXPORT_SYMBOL_GPL(pnfs_put_lseg_async);
 static u64
 end_offset(u64 start, u64 len)
 {
@@ -1470,41 +1487,19 @@ pnfs_generic_pg_test(struct nfs_pageio_descriptor *pgio, struct nfs_page *prev,
 }
 EXPORT_SYMBOL_GPL(pnfs_generic_pg_test);
-int pnfs_write_done_resend_to_mds(struct inode *inode,
+int pnfs_write_done_resend_to_mds(struct nfs_pgio_header *hdr)
-                                struct list_head *head,
-                                const struct nfs_pgio_completion_ops *compl_ops,
-                                struct nfs_direct_req *dreq)
 {
        struct nfs_pageio_descriptor pgio;
-        LIST_HEAD(failed);
        /* Resend all requests through the MDS */
-        nfs_pageio_init_write(&pgio, inode, FLUSH_STABLE, true, compl_ops);
+        nfs_pageio_init_write(&pgio, hdr->inode, FLUSH_STABLE, true,
-        pgio.pg_dreq = dreq;
+                              hdr->completion_ops);
-        while (!list_empty(head)) {
+        return nfs_pageio_resend(&pgio, hdr);
-                struct nfs_page *req = nfs_list_entry(head->next);
-                nfs_list_remove_request(req);
-                if (!nfs_pageio_add_request(&pgio, req))
-                        nfs_list_add_request(req, &failed);
-        }
-        nfs_pageio_complete(&pgio);
-        if (!list_empty(&failed)) {
-                /* For some reason our attempt to resend pages. Mark the
-                 * overall send request as having failed, and let
-                 * nfs_writeback_release_full deal with the error.
-                 */
-                list_move(&failed, head);
-                return -EIO;
-        }
-        return 0;
 }
 EXPORT_SYMBOL_GPL(pnfs_write_done_resend_to_mds);
-static void pnfs_ld_handle_write_error(struct nfs_pgio_data *data)
+static void pnfs_ld_handle_write_error(struct nfs_pgio_header *hdr)
 {
-        struct nfs_pgio_header *hdr = data->header;
        dprintk("pnfs write error = %d\n", hdr->pnfs_error);
        if (NFS_SERVER(hdr->inode)->pnfs_curr_ld->flags &
@@ -1512,50 +1507,42 @@ static void pnfs_ld_handle_write_error(struct nfs_pgio_data *data)
                pnfs_return_layout(hdr->inode);
        }
        if (!test_and_set_bit(NFS_IOHDR_REDO, &hdr->flags))
-                data->task.tk_status = pnfs_write_done_resend_to_mds(hdr->inode,
+                hdr->task.tk_status = pnfs_write_done_resend_to_mds(hdr);
-                                                        &hdr->pages,
-                                                        hdr->completion_ops,
-                                                        hdr->dreq);
 }
 /*
 * Called by non rpc-based layout drivers
 */
-void pnfs_ld_write_done(struct nfs_pgio_data *data)
+void pnfs_ld_write_done(struct nfs_pgio_header *hdr)
 {
-        struct nfs_pgio_header *hdr = data->header;
+        trace_nfs4_pnfs_write(hdr, hdr->pnfs_error);
-        trace_nfs4_pnfs_write(data, hdr->pnfs_error);
        if (!hdr->pnfs_error) {
-                pnfs_set_layoutcommit(data);
+                pnfs_set_layoutcommit(hdr);
-                hdr->mds_ops->rpc_call_done(&data->task, data);
+                hdr->mds_ops->rpc_call_done(&hdr->task, hdr);
        } else
-                pnfs_ld_handle_write_error(data);
+                pnfs_ld_handle_write_error(hdr);
-        hdr->mds_ops->rpc_release(data);
+        hdr->mds_ops->rpc_release(hdr);
 }
 EXPORT_SYMBOL_GPL(pnfs_ld_write_done);
 static void
 pnfs_write_through_mds(struct nfs_pageio_descriptor *desc,
-                struct nfs_pgio_data *data)
+                struct nfs_pgio_header *hdr)
 {
-        struct nfs_pgio_header *hdr = data->header;
        if (!test_and_set_bit(NFS_IOHDR_REDO, &hdr->flags)) {
                list_splice_tail_init(&hdr->pages, &desc->pg_list);
                nfs_pageio_reset_write_mds(desc);
                desc->pg_recoalesce = 1;
        }
-        nfs_pgio_data_release(data);
+        nfs_pgio_data_destroy(hdr);
 }
 static enum pnfs_try_status
-pnfs_try_to_write_data(struct nfs_pgio_data *wdata,
+pnfs_try_to_write_data(struct nfs_pgio_header *hdr,
                        const struct rpc_call_ops *call_ops,
                        struct pnfs_layout_segment *lseg,
                        int how)
 {
-        struct nfs_pgio_header *hdr = wdata->header;
        struct inode *inode = hdr->inode;
        enum pnfs_try_status trypnfs;
        struct nfs_server *nfss = NFS_SERVER(inode);
@@ -1563,8 +1550,8 @@ pnfs_try_to_write_data(struct nfs_pgio_data *wdata,
        hdr->mds_ops = call_ops;
        dprintk("%s: Writing ino:%lu %u@%llu (how %d)\n", __func__,
-                inode->i_ino, wdata->args.count, wdata->args.offset, how);
+                inode->i_ino, hdr->args.count, hdr->args.offset, how);
-        trypnfs = nfss->pnfs_curr_ld->write_pagelist(wdata, how);
+        trypnfs = nfss->pnfs_curr_ld->write_pagelist(hdr, how);
        if (trypnfs != PNFS_NOT_ATTEMPTED)
                nfs_inc_stats(inode, NFSIOS_PNFS_WRITE);
        dprintk("%s End (trypnfs:%d)\n", __func__, trypnfs);
@@ -1575,139 +1562,105 @@ static void
 pnfs_do_write(struct nfs_pageio_descriptor *desc,
              struct nfs_pgio_header *hdr, int how)
 {
-        struct nfs_pgio_data *data = hdr->data;
        const struct rpc_call_ops *call_ops = desc->pg_rpc_callops;
        struct pnfs_layout_segment *lseg = desc->pg_lseg;
        enum pnfs_try_status trypnfs;
        desc->pg_lseg = NULL;
-        trypnfs = pnfs_try_to_write_data(data, call_ops, lseg, how);
+        trypnfs = pnfs_try_to_write_data(hdr, call_ops, lseg, how);
        if (trypnfs == PNFS_NOT_ATTEMPTED)
-                pnfs_write_through_mds(desc, data);
+                pnfs_write_through_mds(desc, hdr);
        pnfs_put_lseg(lseg);
 }
 static void pnfs_writehdr_free(struct nfs_pgio_header *hdr)
 {
        pnfs_put_lseg(hdr->lseg);
-        nfs_rw_header_free(hdr);
+        nfs_pgio_header_free(hdr);
 }
 EXPORT_SYMBOL_GPL(pnfs_writehdr_free);
 int
 pnfs_generic_pg_writepages(struct nfs_pageio_descriptor *desc)
 {
-        struct nfs_rw_header *whdr;
        struct nfs_pgio_header *hdr;
        int ret;
-        whdr = nfs_rw_header_alloc(desc->pg_rw_ops);
+        hdr = nfs_pgio_header_alloc(desc->pg_rw_ops);
-        if (!whdr) {
+        if (!hdr) {
                desc->pg_completion_ops->error_cleanup(&desc->pg_list);
                pnfs_put_lseg(desc->pg_lseg);
                desc->pg_lseg = NULL;
                return -ENOMEM;
        }
-        hdr = &whdr->header;
        nfs_pgheader_init(desc, hdr, pnfs_writehdr_free);
        hdr->lseg = pnfs_get_lseg(desc->pg_lseg);
-        atomic_inc(&hdr->refcnt);
        ret = nfs_generic_pgio(desc, hdr);
        if (ret != 0) {
                pnfs_put_lseg(desc->pg_lseg);
                desc->pg_lseg = NULL;
        } else
                pnfs_do_write(desc, hdr, desc->pg_ioflags);
-        if (atomic_dec_and_test(&hdr->refcnt))
-                hdr->completion_ops->completion(hdr);
        return ret;
 }
 EXPORT_SYMBOL_GPL(pnfs_generic_pg_writepages);
-int pnfs_read_done_resend_to_mds(struct inode *inode,
+int pnfs_read_done_resend_to_mds(struct nfs_pgio_header *hdr)
-                                struct list_head *head,
-                                const struct nfs_pgio_completion_ops *compl_ops,
-                                struct nfs_direct_req *dreq)
 {
        struct nfs_pageio_descriptor pgio;
-        LIST_HEAD(failed);
        /* Resend all requests through the MDS */
-        nfs_pageio_init_read(&pgio, inode, true, compl_ops);
+        nfs_pageio_init_read(&pgio, hdr->inode, true, hdr->completion_ops);
-        pgio.pg_dreq = dreq;
+        return nfs_pageio_resend(&pgio, hdr);
-        while (!list_empty(head)) {
-                struct nfs_page *req = nfs_list_entry(head->next);
-                nfs_list_remove_request(req);
-                if (!nfs_pageio_add_request(&pgio, req))
-                        nfs_list_add_request(req, &failed);
-        }
-        nfs_pageio_complete(&pgio);
-        if (!list_empty(&failed)) {
-                list_move(&failed, head);
-                return -EIO;
-        }
-        return 0;
 }
 EXPORT_SYMBOL_GPL(pnfs_read_done_resend_to_mds);
-static void pnfs_ld_handle_read_error(struct nfs_pgio_data *data)
+static void pnfs_ld_handle_read_error(struct nfs_pgio_header *hdr)
 {
-        struct nfs_pgio_header *hdr = data->header;
        dprintk("pnfs read error = %d\n", hdr->pnfs_error);
        if (NFS_SERVER(hdr->inode)->pnfs_curr_ld->flags &
            PNFS_LAYOUTRET_ON_ERROR) {
                pnfs_return_layout(hdr->inode);
        }
        if (!test_and_set_bit(NFS_IOHDR_REDO, &hdr->flags))
-                data->task.tk_status = pnfs_read_done_resend_to_mds(hdr->inode,
+                hdr->task.tk_status = pnfs_read_done_resend_to_mds(hdr);
-                                                        &hdr->pages,
-                                                        hdr->completion_ops,
-                                                        hdr->dreq);
 }
 /*
 * Called by non rpc-based layout drivers
 */
-void pnfs_ld_read_done(struct nfs_pgio_data *data)
+void pnfs_ld_read_done(struct nfs_pgio_header *hdr)
 {
-        struct nfs_pgio_header *hdr = data->header;
+        trace_nfs4_pnfs_read(hdr, hdr->pnfs_error);
-        trace_nfs4_pnfs_read(data, hdr->pnfs_error);
        if (likely(!hdr->pnfs_error)) {
-                __nfs4_read_done_cb(data);
+                __nfs4_read_done_cb(hdr);
-                hdr->mds_ops->rpc_call_done(&data->task, data);
+                hdr->mds_ops->rpc_call_done(&hdr->task, hdr);
        } else
-                pnfs_ld_handle_read_error(data);
+                pnfs_ld_handle_read_error(hdr);
-        hdr->mds_ops->rpc_release(data);
+        hdr->mds_ops->rpc_release(hdr);
 }
 EXPORT_SYMBOL_GPL(pnfs_ld_read_done);
 static void
 pnfs_read_through_mds(struct nfs_pageio_descriptor *desc,
-                struct nfs_pgio_data *data)
+                struct nfs_pgio_header *hdr)
 {
-        struct nfs_pgio_header *hdr = data->header;
        if (!test_and_set_bit(NFS_IOHDR_REDO, &hdr->flags)) {
                list_splice_tail_init(&hdr->pages, &desc->pg_list);
                nfs_pageio_reset_read_mds(desc);
                desc->pg_recoalesce = 1;
        }
-        nfs_pgio_data_release(data);
+        nfs_pgio_data_destroy(hdr);
 }
 /*
 * Call the appropriate parallel I/O subsystem read function.
 */
 static enum pnfs_try_status
-pnfs_try_to_read_data(struct nfs_pgio_data *rdata,
+pnfs_try_to_read_data(struct nfs_pgio_header *hdr,
                       const struct rpc_call_ops *call_ops,
                       struct pnfs_layout_segment *lseg)
 {
-        struct nfs_pgio_header *hdr = rdata->header;
        struct inode *inode = hdr->inode;
        struct nfs_server *nfss = NFS_SERVER(inode);
        enum pnfs_try_status trypnfs;
@@ -1715,9 +1668,9 @@ pnfs_try_to_read_data(struct nfs_pgio_data *rdata,
        hdr->mds_ops = call_ops;
        dprintk("%s: Reading ino:%lu %u@%llu\n",
-                __func__, inode->i_ino, rdata->args.count, rdata->args.offset);
+                __func__, inode->i_ino, hdr->args.count, hdr->args.offset);
-        trypnfs = nfss->pnfs_curr_ld->read_pagelist(rdata);
+        trypnfs = nfss->pnfs_curr_ld->read_pagelist(hdr);
        if (trypnfs != PNFS_NOT_ATTEMPTED)
                nfs_inc_stats(inode, NFSIOS_PNFS_READ);
        dprintk("%s End (trypnfs:%d)\n", __func__, trypnfs);
@@ -1727,52 +1680,46 @@ pnfs_try_to_read_data(struct nfs_pgio_data *rdata,
 static void
 pnfs_do_read(struct nfs_pageio_descriptor *desc, struct nfs_pgio_header *hdr)
 {
-        struct nfs_pgio_data *data = hdr->data;
        const struct rpc_call_ops *call_ops = desc->pg_rpc_callops;
        struct pnfs_layout_segment *lseg = desc->pg_lseg;
        enum pnfs_try_status trypnfs;
        desc->pg_lseg = NULL;
-        trypnfs = pnfs_try_to_read_data(data, call_ops, lseg);
+        trypnfs = pnfs_try_to_read_data(hdr, call_ops, lseg);
        if (trypnfs == PNFS_NOT_ATTEMPTED)
-                pnfs_read_through_mds(desc, data);
+                pnfs_read_through_mds(desc, hdr);
        pnfs_put_lseg(lseg);
 }
 static void pnfs_readhdr_free(struct nfs_pgio_header *hdr)
 {
        pnfs_put_lseg(hdr->lseg);
-        nfs_rw_header_free(hdr);
+        nfs_pgio_header_free(hdr);
 }
 EXPORT_SYMBOL_GPL(pnfs_readhdr_free);
 int
 pnfs_generic_pg_readpages(struct nfs_pageio_descriptor *desc)
 {
-        struct nfs_rw_header *rhdr;
        struct nfs_pgio_header *hdr;
        int ret;
-        rhdr = nfs_rw_header_alloc(desc->pg_rw_ops);
+        hdr = nfs_pgio_header_alloc(desc->pg_rw_ops);
-        if (!rhdr) {
+        if (!hdr) {
                desc->pg_completion_ops->error_cleanup(&desc->pg_list);
                ret = -ENOMEM;
                pnfs_put_lseg(desc->pg_lseg);
                desc->pg_lseg = NULL;
                return ret;
        }
-        hdr = &rhdr->header;
        nfs_pgheader_init(desc, hdr, pnfs_readhdr_free);
        hdr->lseg = pnfs_get_lseg(desc->pg_lseg);
-        atomic_inc(&hdr->refcnt);
        ret = nfs_generic_pgio(desc, hdr);
        if (ret != 0) {
                pnfs_put_lseg(desc->pg_lseg);
                desc->pg_lseg = NULL;
        } else
                pnfs_do_read(desc, hdr);
-        if (atomic_dec_and_test(&hdr->refcnt))
-                hdr->completion_ops->completion(hdr);
        return ret;
 }
 EXPORT_SYMBOL_GPL(pnfs_generic_pg_readpages);
@@ -1820,12 +1767,11 @@ void pnfs_set_lo_fail(struct pnfs_layout_segment *lseg)
 EXPORT_SYMBOL_GPL(pnfs_set_lo_fail);
 void
-pnfs_set_layoutcommit(struct nfs_pgio_data *wdata)
+pnfs_set_layoutcommit(struct nfs_pgio_header *hdr)
 {
-        struct nfs_pgio_header *hdr = wdata->header;
        struct inode *inode = hdr->inode;
        struct nfs_inode *nfsi = NFS_I(inode);
-        loff_t end_pos = wdata->mds_offset + wdata->res.count;
+        loff_t end_pos = hdr->mds_offset + hdr->res.count;
        bool mark_as_dirty = false;
        spin_lock(&inode->i_lock);
diff --git a/fs/nfs/pnfs.h b/fs/nfs/pnfs.h
index 4fb309a2b4c4..aca3dff5dae6 100644
--- a/fs/nfs/pnfs.h
+++ b/fs/nfs/pnfs.h
@@ -32,6 +32,7 @@
 #include <linux/nfs_fs.h>
 #include <linux/nfs_page.h>
+#include <linux/workqueue.h>
 enum {
        NFS_LSEG_VALID = 0,     /* cleared when lseg is recalled/returned */
@@ -46,6 +47,7 @@ struct pnfs_layout_segment {
        atomic_t pls_refcount;
        unsigned long pls_flags;
        struct pnfs_layout_hdr *pls_layout;
+        struct work_struct pls_work;
 };
 enum pnfs_try_status {
@@ -104,6 +106,8 @@ struct pnfs_layoutdriver_type {
                                  int max);
        void (*recover_commit_reqs) (struct list_head *list,
                                     struct nfs_commit_info *cinfo);
+        struct nfs_page * (*search_commit_reqs)(struct nfs_commit_info *cinfo,
+                                                struct page *page);
        int (*commit_pagelist)(struct inode *inode,
                               struct list_head *mds_pages,
                               int how,
@@ -113,8 +117,8 @@ struct pnfs_layoutdriver_type {
         * Return PNFS_ATTEMPTED to indicate the layout code has attempted
         * I/O, else return PNFS_NOT_ATTEMPTED to fall back to normal NFS
         */
-        enum pnfs_try_status (*read_pagelist) (struct nfs_pgio_data *nfs_data);
+        enum pnfs_try_status (*read_pagelist)(struct nfs_pgio_header *);
-        enum pnfs_try_status (*write_pagelist) (struct nfs_pgio_data *nfs_data, int how);
+        enum pnfs_try_status (*write_pagelist)(struct nfs_pgio_header *, int);
        void (*free_deviceid_node) (struct nfs4_deviceid_node *);
@@ -179,6 +183,7 @@ extern int nfs4_proc_layoutreturn(struct nfs4_layoutreturn *lrp);
 /* pnfs.c */
 void pnfs_get_layout_hdr(struct pnfs_layout_hdr *lo);
 void pnfs_put_lseg(struct pnfs_layout_segment *lseg);
+void pnfs_put_lseg_async(struct pnfs_layout_segment *lseg);
 void set_pnfs_layoutdriver(struct nfs_server *, const struct nfs_fh *, u32);
 void unset_pnfs_layoutdriver(struct nfs_server *);
@@ -213,13 +218,13 @@ bool pnfs_roc(struct inode *ino);
 void pnfs_roc_release(struct inode *ino);
 void pnfs_roc_set_barrier(struct inode *ino, u32 barrier);
 bool pnfs_roc_drain(struct inode *ino, u32 *barrier, struct rpc_task *task);
-void pnfs_set_layoutcommit(struct nfs_pgio_data *wdata);
+void pnfs_set_layoutcommit(struct nfs_pgio_header *);
 void pnfs_cleanup_layoutcommit(struct nfs4_layoutcommit_data *data);
 int pnfs_layoutcommit_inode(struct inode *inode, bool sync);
 int _pnfs_return_layout(struct inode *);
 int pnfs_commit_and_return_layout(struct inode *);
-void pnfs_ld_write_done(struct nfs_pgio_data *);
+void pnfs_ld_write_done(struct nfs_pgio_header *);
-void pnfs_ld_read_done(struct nfs_pgio_data *);
+void pnfs_ld_read_done(struct nfs_pgio_header *);
 struct pnfs_layout_segment *pnfs_update_layout(struct inode *ino,
                                               struct nfs_open_context *ctx,
                                               loff_t pos,
@@ -228,12 +233,8 @@ struct pnfs_layout_segment *pnfs_update_layout(struct inode *ino,
                                               gfp_t gfp_flags);
 void nfs4_deviceid_mark_client_invalid(struct nfs_client *clp);
-int pnfs_read_done_resend_to_mds(struct inode *inode, struct list_head *head,
+int pnfs_read_done_resend_to_mds(struct nfs_pgio_header *);
-                        const struct nfs_pgio_completion_ops *compl_ops,
+int pnfs_write_done_resend_to_mds(struct nfs_pgio_header *);
-                        struct nfs_direct_req *dreq);
-int pnfs_write_done_resend_to_mds(struct inode *inode, struct list_head *head,
-                        const struct nfs_pgio_completion_ops *compl_ops,
-                        struct nfs_direct_req *dreq);
 struct nfs4_threshold *pnfs_mdsthreshold_alloc(void);
 /* nfs4_deviceid_flags */
@@ -345,6 +346,17 @@ pnfs_recover_commit_reqs(struct inode *inode, struct list_head *list,
        NFS_SERVER(inode)->pnfs_curr_ld->recover_commit_reqs(list, cinfo);
 }
+static inline struct nfs_page *
+pnfs_search_commit_reqs(struct inode *inode, struct nfs_commit_info *cinfo,
+                        struct page *page)
+{
+        struct pnfs_layoutdriver_type *ld = NFS_SERVER(inode)->pnfs_curr_ld;
+        if (ld == NULL || ld->search_commit_reqs == NULL)
+                return NULL;
+        return ld->search_commit_reqs(cinfo, page);
+}
 /* Should the pNFS client commit and return the layout upon a setattr */
 static inline bool
 pnfs_ld_layoutret_on_setattr(struct inode *inode)
@@ -410,6 +422,10 @@ static inline void pnfs_put_lseg(struct pnfs_layout_segment *lseg)
 {
 }
+static inline void pnfs_put_lseg_async(struct pnfs_layout_segment *lseg)
+{
+}
 static inline int pnfs_return_layout(struct inode *ino)
 {
        return 0;
@@ -496,6 +512,13 @@ pnfs_recover_commit_reqs(struct inode *inode, struct list_head *list,
 {
 }
+static inline struct nfs_page *
+pnfs_search_commit_reqs(struct inode *inode, struct nfs_commit_info *cinfo,
+                        struct page *page)
+{
+        return NULL;
+}
 static inline int pnfs_layoutcommit_inode(struct inode *inode, bool sync)
 {
        return 0;
diff --git a/fs/nfs/proc.c b/fs/nfs/proc.c
index c171ce1a8a30..b09cc23d6f43 100644
--- a/fs/nfs/proc.c
+++ b/fs/nfs/proc.c
@@ -578,46 +578,49 @@ nfs_proc_pathconf(struct nfs_server *server, struct nfs_fh *fhandle,
        return 0;
 }
-static int nfs_read_done(struct rpc_task *task, struct nfs_pgio_data *data)
+static int nfs_read_done(struct rpc_task *task, struct nfs_pgio_header *hdr)
 {
-        struct inode *inode = data->header->inode;
+        struct inode *inode = hdr->inode;
        nfs_invalidate_atime(inode);
        if (task->tk_status >= 0) {
-                nfs_refresh_inode(inode, data->res.fattr);
+                nfs_refresh_inode(inode, hdr->res.fattr);
                /* Emulate the eof flag, which isn't normally needed in NFSv2
                 * as it is guaranteed to always return the file attributes
                 */
-                if (data->args.offset + data->res.count >= data->res.fattr->size)
+                if (hdr->args.offset + hdr->res.count >= hdr->res.fattr->size)
-                        data->res.eof = 1;
+                        hdr->res.eof = 1;
        }
        return 0;
 }
-static void nfs_proc_read_setup(struct nfs_pgio_data *data, struct rpc_message *msg)
+static void nfs_proc_read_setup(struct nfs_pgio_header *hdr,
+                                struct rpc_message *msg)
 {
        msg->rpc_proc = &nfs_procedures[NFSPROC_READ];
 }
-static int nfs_proc_pgio_rpc_prepare(struct rpc_task *task, struct nfs_pgio_data *data)
+static int nfs_proc_pgio_rpc_prepare(struct rpc_task *task,
+                                     struct nfs_pgio_header *hdr)
 {
        rpc_call_start(task);
        return 0;
 }
-static int nfs_write_done(struct rpc_task *task, struct nfs_pgio_data *data)
+static int nfs_write_done(struct rpc_task *task, struct nfs_pgio_header *hdr)
 {
-        struct inode *inode = data->header->inode;
+        struct inode *inode = hdr->inode;
        if (task->tk_status >= 0)
-                nfs_post_op_update_inode_force_wcc(inode, data->res.fattr);
+                nfs_post_op_update_inode_force_wcc(inode, hdr->res.fattr);
        return 0;
 }
-static void nfs_proc_write_setup(struct nfs_pgio_data *data, struct rpc_message *msg)
+static void nfs_proc_write_setup(struct nfs_pgio_header *hdr,
+                                 struct rpc_message *msg)
 {
        /* Note: NFSv2 ignores @stable and always uses NFS_FILE_SYNC */
-        data->args.stable = NFS_FILE_SYNC;
+        hdr->args.stable = NFS_FILE_SYNC;
        msg->rpc_proc = &nfs_procedures[NFSPROC_WRITE];
 }
diff --git a/fs/nfs/read.c b/fs/nfs/read.c
index e818a475ca64..beff2769c5c5 100644
--- a/fs/nfs/read.c
+++ b/fs/nfs/read.c
@@ -33,12 +33,12 @@ static const struct nfs_rw_ops nfs_rw_read_ops;
 static struct kmem_cache *nfs_rdata_cachep;
-static struct nfs_rw_header *nfs_readhdr_alloc(void)
+static struct nfs_pgio_header *nfs_readhdr_alloc(void)
 {
        return kmem_cache_zalloc(nfs_rdata_cachep, GFP_KERNEL);
 }
-static void nfs_readhdr_free(struct nfs_rw_header *rhdr)
+static void nfs_readhdr_free(struct nfs_pgio_header *rhdr)
 {
        kmem_cache_free(nfs_rdata_cachep, rhdr);
 }
@@ -115,12 +115,6 @@ static void nfs_readpage_release(struct nfs_page *req)
                unlock_page(req->wb_page);
        }
-        dprintk("NFS: read done (%s/%Lu %d@%Ld)\n",
-                        req->wb_context->dentry->d_inode->i_sb->s_id,
-                        (unsigned long long)NFS_FILEID(req->wb_context->dentry->d_inode),
-                        req->wb_bytes,
-                        (long long)req_offset(req));
        nfs_release_request(req);
 }
@@ -172,14 +166,15 @@ out:
        hdr->release(hdr);
 }
-static void nfs_initiate_read(struct nfs_pgio_data *data, struct rpc_message *msg,
+static void nfs_initiate_read(struct nfs_pgio_header *hdr,
+                              struct rpc_message *msg,
                              struct rpc_task_setup *task_setup_data, int how)
 {
-        struct inode *inode = data->header->inode;
+        struct inode *inode = hdr->inode;
        int swap_flags = IS_SWAPFILE(inode) ? NFS_RPC_SWAPFLAGS : 0;
        task_setup_data->flags |= swap_flags;
-        NFS_PROTO(inode)->read_setup(data, msg);
+        NFS_PROTO(inode)->read_setup(hdr, msg);
 }
 static void
@@ -203,14 +198,15 @@ static const struct nfs_pgio_completion_ops nfs_async_read_completion_ops = {
 * This is the callback from RPC telling us whether a reply was
 * received or some error occurred (timeout or socket shutdown).
 */
-static int nfs_readpage_done(struct rpc_task *task, struct nfs_pgio_data *data,
+static int nfs_readpage_done(struct rpc_task *task,
+                             struct nfs_pgio_header *hdr,
                             struct inode *inode)
 {
-        int status = NFS_PROTO(inode)->read_done(task, data);
+        int status = NFS_PROTO(inode)->read_done(task, hdr);
        if (status != 0)
                return status;
-        nfs_add_stats(inode, NFSIOS_SERVERREADBYTES, data->res.count);
+        nfs_add_stats(inode, NFSIOS_SERVERREADBYTES, hdr->res.count);
        if (task->tk_status == -ESTALE) {
                set_bit(NFS_INO_STALE, &NFS_I(inode)->flags);
@@ -219,34 +215,34 @@ static int nfs_readpage_done(struct rpc_task *task, struct nfs_pgio_data *data,
        return 0;
 }
-static void nfs_readpage_retry(struct rpc_task *task, struct nfs_pgio_data *data)
+static void nfs_readpage_retry(struct rpc_task *task,
+                               struct nfs_pgio_header *hdr)
 {
-        struct nfs_pgio_args *argp = &data->args;
+        struct nfs_pgio_args *argp = &hdr->args;
-        struct nfs_pgio_res  *resp = &data->res;
+        struct nfs_pgio_res  *resp = &hdr->res;
        /* This is a short read! */
-        nfs_inc_stats(data->header->inode, NFSIOS_SHORTREAD);
+        nfs_inc_stats(hdr->inode, NFSIOS_SHORTREAD);
        /* Has the server at least made some progress? */
        if (resp->count == 0) {
-                nfs_set_pgio_error(data->header, -EIO, argp->offset);
+                nfs_set_pgio_error(hdr, -EIO, argp->offset);
                return;
        }
-        /* Yes, so retry the read at the end of the data */
+        /* Yes, so retry the read at the end of the hdr */
-        data->mds_offset += resp->count;
+        hdr->mds_offset += resp->count;
        argp->offset += resp->count;
        argp->pgbase += resp->count;
        argp->count -= resp->count;
        rpc_restart_call_prepare(task);
 }
-static void nfs_readpage_result(struct rpc_task *task, struct nfs_pgio_data *data)
+static void nfs_readpage_result(struct rpc_task *task,
+                                struct nfs_pgio_header *hdr)
 {
-        struct nfs_pgio_header *hdr = data->header;
+        if (hdr->res.eof) {
-        if (data->res.eof) {
                loff_t bound;
-                bound = data->args.offset + data->res.count;
+                bound = hdr->args.offset + hdr->res.count;
                spin_lock(&hdr->lock);
                if (bound < hdr->io_start + hdr->good_bytes) {
                        set_bit(NFS_IOHDR_EOF, &hdr->flags);
@@ -254,8 +250,8 @@ static void nfs_readpage_result(struct rpc_task *task, struct nfs_pgio_data *dat
                        hdr->good_bytes = bound - hdr->io_start;
                }
                spin_unlock(&hdr->lock);
-        } else if (data->res.count != data->args.count)
+        } else if (hdr->res.count != hdr->args.count)
-                nfs_readpage_retry(task, data);
+                nfs_readpage_retry(task, hdr);
 }
 /*
@@ -404,7 +400,7 @@ out:
 int __init nfs_init_readpagecache(void)
 {
        nfs_rdata_cachep = kmem_cache_create("nfs_read_data",
-                                             sizeof(struct nfs_rw_header),
+                                             sizeof(struct nfs_pgio_header),
                                             0, SLAB_HWCACHE_ALIGN,
                                             NULL);
        if (nfs_rdata_cachep == NULL)
diff --git a/fs/nfs/super.c b/fs/nfs/super.c
index 084af1060d79..e4499d5b51e8 100644
--- a/fs/nfs/super.c
+++ b/fs/nfs/super.c
@@ -1027,8 +1027,7 @@ static bool nfs_auth_info_add(struct nfs_auth_info *auth_info,
                              rpc_authflavor_t flavor)
 {
        unsigned int i;
-        unsigned int max_flavor_len = (sizeof(auth_info->flavors) /
+        unsigned int max_flavor_len = ARRAY_SIZE(auth_info->flavors);
-                                       sizeof(auth_info->flavors[0]));
        /* make sure this flavor isn't already in the list */
        for (i = 0; i < auth_info->flavor_len; i++) {
@@ -2180,7 +2179,7 @@ out_no_address:
        return -EINVAL;
 }
-#define NFS_MOUNT_CMP_FLAGMASK ~(NFS_MOUNT_INTR \
+#define NFS_REMOUNT_CMP_FLAGMASK ~(NFS_MOUNT_INTR \
                | NFS_MOUNT_SECURE \
                | NFS_MOUNT_TCP \
                | NFS_MOUNT_VER3 \
@@ -2188,15 +2187,16 @@ out_no_address:
                | NFS_MOUNT_NONLM \
                | NFS_MOUNT_BROKEN_SUID \
                | NFS_MOUNT_STRICTLOCK \
-                | NFS_MOUNT_UNSHARED \
-                | NFS_MOUNT_NORESVPORT \
                | NFS_MOUNT_LEGACY_INTERFACE)
+#define NFS_MOUNT_CMP_FLAGMASK (NFS_REMOUNT_CMP_FLAGMASK & \
+                ~(NFS_MOUNT_UNSHARED | NFS_MOUNT_NORESVPORT))
 static int
 nfs_compare_remount_data(struct nfs_server *nfss,
                         struct nfs_parsed_mount_data *data)
 {
-        if ((data->flags ^ nfss->flags) & NFS_MOUNT_CMP_FLAGMASK ||
+        if ((data->flags ^ nfss->flags) & NFS_REMOUNT_CMP_FLAGMASK ||
            data->rsize != nfss->rsize ||
            data->wsize != nfss->wsize ||
            data->version != nfss->nfs_client->rpc_ops->version ||
diff --git a/fs/nfs/write.c b/fs/nfs/write.c
index 962c9ee758be..e3b5cf28bdc5 100644
--- a/fs/nfs/write.c
+++ b/fs/nfs/write.c
@@ -47,6 +47,8 @@ static const struct nfs_pgio_completion_ops nfs_async_write_completion_ops;
 static const struct nfs_commit_completion_ops nfs_commit_completion_ops;
 static const struct nfs_rw_ops nfs_rw_write_ops;
 static void nfs_clear_request_commit(struct nfs_page *req);
+static void nfs_init_cinfo_from_inode(struct nfs_commit_info *cinfo,
+                                      struct inode *inode);
 static struct kmem_cache *nfs_wdata_cachep;
 static mempool_t *nfs_wdata_mempool;
@@ -71,18 +73,18 @@ void nfs_commit_free(struct nfs_commit_data *p)
 }
 EXPORT_SYMBOL_GPL(nfs_commit_free);
-static struct nfs_rw_header *nfs_writehdr_alloc(void)
+static struct nfs_pgio_header *nfs_writehdr_alloc(void)
 {
-        struct nfs_rw_header *p = mempool_alloc(nfs_wdata_mempool, GFP_NOIO);
+        struct nfs_pgio_header *p = mempool_alloc(nfs_wdata_mempool, GFP_NOIO);
        if (p)
                memset(p, 0, sizeof(*p));
        return p;
 }
-static void nfs_writehdr_free(struct nfs_rw_header *whdr)
+static void nfs_writehdr_free(struct nfs_pgio_header *hdr)
 {
-        mempool_free(whdr, nfs_wdata_mempool);
+        mempool_free(hdr, nfs_wdata_mempool);
 }
 static void nfs_context_set_write_error(struct nfs_open_context *ctx, int error)
@@ -93,6 +95,38 @@ static void nfs_context_set_write_error(struct nfs_open_context *ctx, int error)
 }
 /*
+ * nfs_page_search_commits_for_head_request_locked
+ *
+ * Search through commit lists on @inode for the head request for @page.
+ * Must be called while holding the inode (which is cinfo) lock.
+ *
+ * Returns the head request if found, or NULL if not found.
+ */
+static struct nfs_page *
+nfs_page_search_commits_for_head_request_locked(struct nfs_inode *nfsi,
+                                                struct page *page)
+{
+        struct nfs_page *freq, *t;
+        struct nfs_commit_info cinfo;
+        struct inode *inode = &nfsi->vfs_inode;
+        nfs_init_cinfo_from_inode(&cinfo, inode);
+        /* search through pnfs commit lists */
+        freq = pnfs_search_commit_reqs(inode, &cinfo, page);
+        if (freq)
+                return freq->wb_head;
+        /* Linearly search the commit list for the correct request */
+        list_for_each_entry_safe(freq, t, &cinfo.mds->list, wb_list) {
+                if (freq->wb_page == page)
+                        return freq->wb_head;
+        }
+        return NULL;
+}
+/*
 * nfs_page_find_head_request_locked - find head request associated with @page
 *
 * must be called while holding the inode lock.
@@ -106,21 +140,12 @@ nfs_page_find_head_request_locked(struct nfs_inode *nfsi, struct page *page)
        if (PagePrivate(page))
                req = (struct nfs_page *)page_private(page);
-        else if (unlikely(PageSwapCache(page))) {
+        else if (unlikely(PageSwapCache(page)))
-                struct nfs_page *freq, *t;
+                req = nfs_page_search_commits_for_head_request_locked(nfsi,
+                        page);
-                /* Linearly search the commit list for the correct req */
-                list_for_each_entry_safe(freq, t, &nfsi->commit_info.list, wb_list) {
-                        if (freq->wb_page == page) {
-                                req = freq->wb_head;
-                                break;
-                        }
-                }
-        }
        if (req) {
                WARN_ON_ONCE(req->wb_head != req);
                kref_get(&req->wb_kref);
        }
@@ -216,7 +241,7 @@ static bool nfs_page_group_covers_page(struct nfs_page *req)
        unsigned int pos = 0;
        unsigned int len = nfs_page_length(req->wb_page);
-        nfs_page_group_lock(req);
+        nfs_page_group_lock(req, true);
        do {
                tmp = nfs_page_group_search_locked(req->wb_head, pos);
@@ -379,8 +404,6 @@ nfs_destroy_unlinked_subrequests(struct nfs_page *destroy_list,
                subreq->wb_head = subreq;
                subreq->wb_this_page = subreq;
-                nfs_clear_request_commit(subreq);
                /* subreq is now totally disconnected from page group or any
                 * write / commit lists. last chance to wake any waiters */
                nfs_unlock_request(subreq);
@@ -456,7 +479,9 @@ try_again:
        }
        /* lock each request in the page group */
-        nfs_page_group_lock(head);
+        ret = nfs_page_group_lock(head, false);
+        if (ret < 0)
+                return ERR_PTR(ret);
        subreq = head;
        do {
                /*
@@ -488,7 +513,7 @@ try_again:
         * Commit list removal accounting is done after locks are dropped */
        subreq = head;
        do {
-                nfs_list_remove_request(subreq);
+                nfs_clear_request_commit(subreq);
                subreq = subreq->wb_this_page;
        } while (subreq != head);
@@ -518,15 +543,11 @@ try_again:
        nfs_page_group_unlock(head);
-        /* drop lock to clear_request_commit the head req and clean up
+        /* drop lock to clean uprequests on destroy list */
-         * requests on destroy list */
        spin_unlock(&inode->i_lock);
        nfs_destroy_unlinked_subrequests(destroy_list, head);
-        /* clean up commit list state */
-        nfs_clear_request_commit(head);
        /* still holds ref on head from nfs_page_find_head_request_locked
         * and still has lock on head from lock loop */
        return head;
@@ -705,6 +726,8 @@ static void nfs_inode_remove_request(struct nfs_page *req)
        if (test_and_clear_bit(PG_INODE_REF, &req->wb_flags))
                nfs_release_request(req);
+        else
+                WARN_ON_ONCE(1);
 }
 static void
@@ -808,6 +831,7 @@ nfs_clear_page_commit(struct page *page)
        dec_bdi_stat(page_file_mapping(page)->backing_dev_info, BDI_RECLAIMABLE);
 }
+/* Called holding inode (/cinfo) lock */
 static void
 nfs_clear_request_commit(struct nfs_page *req)
 {
@@ -817,20 +841,17 @@ nfs_clear_request_commit(struct nfs_page *req)
                nfs_init_cinfo_from_inode(&cinfo, inode);
                if (!pnfs_clear_request_commit(req, &cinfo)) {
-                        spin_lock(cinfo.lock);
                        nfs_request_remove_commit_list(req, &cinfo);
-                        spin_unlock(cinfo.lock);
                }
                nfs_clear_page_commit(req->wb_page);
        }
 }
-static inline
+int nfs_write_need_commit(struct nfs_pgio_header *hdr)
-int nfs_write_need_commit(struct nfs_pgio_data *data)
 {
-        if (data->verf.committed == NFS_DATA_SYNC)
+        if (hdr->verf.committed == NFS_DATA_SYNC)
-                return data->header->lseg == NULL;
+                return hdr->lseg == NULL;
-        return data->verf.committed != NFS_FILE_SYNC;
+        return hdr->verf.committed != NFS_FILE_SYNC;
 }
 #else
@@ -856,8 +877,7 @@ nfs_clear_request_commit(struct nfs_page *req)
 {
 }
-static inline
+int nfs_write_need_commit(struct nfs_pgio_header *hdr)
-int nfs_write_need_commit(struct nfs_pgio_data *data)
 {
        return 0;
 }
@@ -883,11 +903,7 @@ static void nfs_write_completion(struct nfs_pgio_header *hdr)
                        nfs_context_set_write_error(req->wb_context, hdr->error);
                        goto remove_req;
                }
-                if (test_bit(NFS_IOHDR_NEED_RESCHED, &hdr->flags)) {
+                if (nfs_write_need_commit(hdr)) {
-                        nfs_mark_request_dirty(req);
-                        goto next;
-                }
-                if (test_bit(NFS_IOHDR_NEED_COMMIT, &hdr->flags)) {
                        memcpy(&req->wb_verf, &hdr->verf.verifier, sizeof(req->wb_verf));
                        nfs_mark_request_commit(req, hdr->lseg, &cinfo);
                        goto next;
@@ -1038,9 +1054,9 @@ static struct nfs_page *nfs_try_to_update_request(struct inode *inode,
        else
                req->wb_bytes = rqend - req->wb_offset;
 out_unlock:
-        spin_unlock(&inode->i_lock);
        if (req)
                nfs_clear_request_commit(req);
+        spin_unlock(&inode->i_lock);
        return req;
 out_flushme:
        spin_unlock(&inode->i_lock);
@@ -1241,17 +1257,18 @@ static int flush_task_priority(int how)
        return RPC_PRIORITY_NORMAL;
 }
-static void nfs_initiate_write(struct nfs_pgio_data *data, struct rpc_message *msg,
+static void nfs_initiate_write(struct nfs_pgio_header *hdr,
+                               struct rpc_message *msg,
                               struct rpc_task_setup *task_setup_data, int how)
 {
-        struct inode *inode = data->header->inode;
+        struct inode *inode = hdr->inode;
        int priority = flush_task_priority(how);
        task_setup_data->priority = priority;
-        NFS_PROTO(inode)->write_setup(data, msg);
+        NFS_PROTO(inode)->write_setup(hdr, msg);
        nfs4_state_protect_write(NFS_SERVER(inode)->nfs_client,
-                                 &task_setup_data->rpc_client, msg, data);
+                                 &task_setup_data->rpc_client, msg, hdr);
 }
 /* If a nfs_flush_* function fails, it should remove reqs from @head and
@@ -1313,21 +1330,9 @@ void nfs_commit_prepare(struct rpc_task *task, void *calldata)
        NFS_PROTO(data->inode)->commit_rpc_prepare(task, data);
 }
-static void nfs_writeback_release_common(struct nfs_pgio_data *data)
+static void nfs_writeback_release_common(struct nfs_pgio_header *hdr)
 {
-        struct nfs_pgio_header *hdr = data->header;
+        /* do nothing! */
-        int status = data->task.tk_status;
-        if ((status >= 0) && nfs_write_need_commit(data)) {
-                spin_lock(&hdr->lock);
-                if (test_bit(NFS_IOHDR_NEED_RESCHED, &hdr->flags))
-                        ; /* Do nothing */
-                else if (!test_and_set_bit(NFS_IOHDR_NEED_COMMIT, &hdr->flags))
-                        memcpy(&hdr->verf, &data->verf, sizeof(hdr->verf));
-                else if (memcmp(&hdr->verf, &data->verf, sizeof(hdr->verf)))
-                        set_bit(NFS_IOHDR_NEED_RESCHED, &hdr->flags);
-                spin_unlock(&hdr->lock);
-        }
 }
 /*
@@ -1358,7 +1363,8 @@ static int nfs_should_remove_suid(const struct inode *inode)
 /*
 * This function is called when the WRITE call is complete.
 */
-static int nfs_writeback_done(struct rpc_task *task, struct nfs_pgio_data *data,
+static int nfs_writeback_done(struct rpc_task *task,
+                              struct nfs_pgio_header *hdr,
                              struct inode *inode)
 {
        int status;
@@ -1370,13 +1376,14 @@ static int nfs_writeback_done(struct rpc_task *task, struct nfs_pgio_data *data,
         * another writer had changed the file, but some applications
         * depend on tighter cache coherency when writing.
         */
-        status = NFS_PROTO(inode)->write_done(task, data);
+        status = NFS_PROTO(inode)->write_done(task, hdr);
        if (status != 0)
                return status;
-        nfs_add_stats(inode, NFSIOS_SERVERWRITTENBYTES, data->res.count);
+        nfs_add_stats(inode, NFSIOS_SERVERWRITTENBYTES, hdr->res.count);
 #if IS_ENABLED(CONFIG_NFS_V3) || IS_ENABLED(CONFIG_NFS_V4)
-        if (data->res.verf->committed < data->args.stable && task->tk_status >= 0) {
+        if (hdr->res.verf->committed < hdr->args.stable &&
+            task->tk_status >= 0) {
                /* We tried a write call, but the server did not
                 * commit data to stable storage even though we
                 * requested it.
@@ -1392,7 +1399,7 @@ static int nfs_writeback_done(struct rpc_task *task, struct nfs_pgio_data *data,
                        dprintk("NFS:       faulty NFS server %s:"
                                " (committed = %d) != (stable = %d)\n",
                                NFS_SERVER(inode)->nfs_client->cl_hostname,
-                                data->res.verf->committed, data->args.stable);
+                                hdr->res.verf->committed, hdr->args.stable);
                        complain = jiffies + 300 * HZ;
                }
        }
@@ -1407,16 +1414,17 @@ static int nfs_writeback_done(struct rpc_task *task, struct nfs_pgio_data *data,
 /*
 * This function is called when the WRITE call is complete.
 */
-static void nfs_writeback_result(struct rpc_task *task, struct nfs_pgio_data *data)
+static void nfs_writeback_result(struct rpc_task *task,
+                                 struct nfs_pgio_header *hdr)
 {
-        struct nfs_pgio_args    *argp = &data->args;
+        struct nfs_pgio_args    *argp = &hdr->args;
-        struct nfs_pgio_res     *resp = &data->res;
+        struct nfs_pgio_res     *resp = &hdr->res;
        if (resp->count < argp->count) {
                static unsigned long    complain;
                /* This a short write! */
-                nfs_inc_stats(data->header->inode, NFSIOS_SHORTWRITE);
+                nfs_inc_stats(hdr->inode, NFSIOS_SHORTWRITE);
                /* Has the server at least made some progress? */
                if (resp->count == 0) {
@@ -1426,14 +1434,14 @@ static void nfs_writeback_result(struct rpc_task *task, struct nfs_pgio_data *da
                                       argp->count);
                                complain = jiffies + 300 * HZ;
                        }
-                        nfs_set_pgio_error(data->header, -EIO, argp->offset);
+                        nfs_set_pgio_error(hdr, -EIO, argp->offset);
                        task->tk_status = -EIO;
                        return;
                }
                /* Was this an NFSv2 write or an NFSv3 stable write? */
                if (resp->verf->committed != NFS_UNSTABLE) {
                        /* Resend from where the server left off */
-                        data->mds_offset += resp->count;
+                        hdr->mds_offset += resp->count;
                        argp->offset += resp->count;
                        argp->pgbase += resp->count;
                        argp->count -= resp->count;
@@ -1884,7 +1892,7 @@ int nfs_migrate_page(struct address_space *mapping, struct page *newpage,
 int __init nfs_init_writepagecache(void)
 {
        nfs_wdata_cachep = kmem_cache_create("nfs_write_data",
-                                             sizeof(struct nfs_rw_header),
+                                             sizeof(struct nfs_pgio_header),
                                             0, SLAB_HWCACHE_ALIGN,
                                             NULL);
        if (nfs_wdata_cachep == NULL)
diff --git a/fs/nfs_common/nfsacl.c b/fs/nfs_common/nfsacl.c
index ed628f71274c..538f142935ea 100644
--- a/fs/nfs_common/nfsacl.c
+++ b/fs/nfs_common/nfsacl.c
@@ -30,9 +30,6 @@
 MODULE_LICENSE("GPL");
-EXPORT_SYMBOL_GPL(nfsacl_encode);
-EXPORT_SYMBOL_GPL(nfsacl_decode);
 struct nfsacl_encode_desc {
        struct xdr_array2_desc desc;
        unsigned int count;
@@ -136,6 +133,7 @@ int nfsacl_encode(struct xdr_buf *buf, unsigned int base, struct inode *inode,
                          nfsacl_desc.desc.array_len;
        return err;
 }
+EXPORT_SYMBOL_GPL(nfsacl_encode);
 struct nfsacl_decode_desc {
        struct xdr_array2_desc desc;
@@ -295,3 +293,4 @@ int nfsacl_decode(struct xdr_buf *buf, unsigned int base, unsigned int *aclcnt,
        return 8 + nfsacl_desc.desc.elem_size *
                   nfsacl_desc.desc.array_len;
 }
+EXPORT_SYMBOL_GPL(nfsacl_decode);
diff --git a/fs/nilfs2/super.c b/fs/nilfs2/super.c
index c519927b7b5e..228f5bdf0772 100644
--- a/fs/nilfs2/super.c
+++ b/fs/nilfs2/super.c
@@ -942,7 +942,7 @@ static int nilfs_get_root_dentry(struct super_block *sb,
                        iput(inode);
                }
        } else {
-                dentry = d_obtain_alias(inode);
+                dentry = d_obtain_root(inode);
                if (IS_ERR(dentry)) {
                        ret = PTR_ERR(dentry);
                        goto failed_dentry;
diff --git a/fs/quota/dquot.c b/fs/quota/dquot.c
index 7f30bdc57d13..f2d0eee9d1f1 100644
--- a/fs/quota/dquot.c
+++ b/fs/quota/dquot.c
@@ -96,13 +96,16 @@
 * Note that some things (eg. sb pointer, type, id) doesn't change during
 * the life of the dquot structure and so needn't to be protected by a lock
 *
- * Any operation working on dquots via inode pointers must hold dqptr_sem.  If
+ * Operation accessing dquots via inode pointers are protected by dquot_srcu.
- * operation is just reading pointers from inode (or not using them at all) the
+ * Operation of reading pointer needs srcu_read_lock(&dquot_srcu), and
- * read lock is enough. If pointers are altered function must hold write lock.
+ * synchronize_srcu(&dquot_srcu) is called after clearing pointers from
+ * inode and before dropping dquot references to avoid use of dquots after
+ * they are freed. dq_data_lock is used to serialize the pointer setting and
+ * clearing operations.
 * Special care needs to be taken about S_NOQUOTA inode flag (marking that
 * inode is a quota file). Functions adding pointers from inode to dquots have
- * to check this flag under dqptr_sem and then (if S_NOQUOTA is not set) they
+ * to check this flag under dq_data_lock and then (if S_NOQUOTA is not set) they
- * have to do all pointer modifications before dropping dqptr_sem. This makes
+ * have to do all pointer modifications before dropping dq_data_lock. This makes
 * sure they cannot race with quotaon which first sets S_NOQUOTA flag and
 * then drops all pointers to dquots from an inode.
 *
@@ -116,21 +119,15 @@
 * spinlock to internal buffers before writing.
 *
 * Lock ordering (including related VFS locks) is the following:
- *   dqonoff_mutex > i_mutex > journal_lock > dqptr_sem > dquot->dq_lock >
+ *   dqonoff_mutex > i_mutex > journal_lock > dquot->dq_lock > dqio_mutex
- *   dqio_mutex
 * dqonoff_mutex > i_mutex comes from dquot_quota_sync, dquot_enable, etc.
- * The lock ordering of dqptr_sem imposed by quota code is only dqonoff_sem >
- * dqptr_sem. But filesystem has to count with the fact that functions such as
- * dquot_alloc_space() acquire dqptr_sem and they usually have to be called
- * from inside a transaction to keep filesystem consistency after a crash. Also
- * filesystems usually want to do some IO on dquot from ->mark_dirty which is
- * called with dqptr_sem held.
 */
 static __cacheline_aligned_in_smp DEFINE_SPINLOCK(dq_list_lock);
 static __cacheline_aligned_in_smp DEFINE_SPINLOCK(dq_state_lock);
 __cacheline_aligned_in_smp DEFINE_SPINLOCK(dq_data_lock);
 EXPORT_SYMBOL(dq_data_lock);
+DEFINE_STATIC_SRCU(dquot_srcu);
 void __quota_error(struct super_block *sb, const char *func,
                   const char *fmt, ...)
@@ -733,7 +730,6 @@ static struct shrinker dqcache_shrinker = {
 /*
 * Put reference to dquot
- * NOTE: If you change this function please check whether dqput_blocks() works right...
 */
 void dqput(struct dquot *dquot)
 {
@@ -963,46 +959,33 @@ static void add_dquot_ref(struct super_block *sb, int type)
 }
 /*
- * Return 0 if dqput() won't block.
- * (note that 1 doesn't necessarily mean blocking)
- */
-static inline int dqput_blocks(struct dquot *dquot)
-{
-        if (atomic_read(&dquot->dq_count) <= 1)
-                return 1;
-        return 0;
-}
-/*
 * Remove references to dquots from inode and add dquot to list for freeing
 * if we have the last reference to dquot
- * We can't race with anybody because we hold dqptr_sem for writing...
 */
-static int remove_inode_dquot_ref(struct inode *inode, int type,
+static void remove_inode_dquot_ref(struct inode *inode, int type,
-                                  struct list_head *tofree_head)
+                                   struct list_head *tofree_head)
 {
        struct dquot *dquot = inode->i_dquot[type];
        inode->i_dquot[type] = NULL;
-        if (dquot) {
+        if (!dquot)
-                if (dqput_blocks(dquot)) {
+                return;
-#ifdef CONFIG_QUOTA_DEBUG
-                        if (atomic_read(&dquot->dq_count) != 1)
+        if (list_empty(&dquot->dq_free)) {
-                                quota_error(inode->i_sb, "Adding dquot with "
+                /*
-                                            "dq_count %d to dispose list",
+                 * The inode still has reference to dquot so it can't be in the
-                                            atomic_read(&dquot->dq_count));
+                 * free list
-#endif
+                 */
-                        spin_lock(&dq_list_lock);
+                spin_lock(&dq_list_lock);
-                        /* As dquot must have currently users it can't be on
+                list_add(&dquot->dq_free, tofree_head);
-                         * the free list... */
+                spin_unlock(&dq_list_lock);
-                        list_add(&dquot->dq_free, tofree_head);
+        } else {
-                        spin_unlock(&dq_list_lock);
+                /*
-                        return 1;
+                 * Dquot is already in a list to put so we won't drop the last
-                }
+                 * reference here.
-                else
+                 */
-                        dqput(dquot);   /* We have guaranteed we won't block */
+                dqput(dquot);
        }
-        return 0;
 }
 /*
@@ -1037,13 +1020,15 @@ static void remove_dquot_ref(struct super_block *sb, int type,
                 *  We have to scan also I_NEW inodes because they can already
                 *  have quota pointer initialized. Luckily, we need to touch
                 *  only quota pointers and these have separate locking
-                 *  (dqptr_sem).
+                 *  (dq_data_lock).
                 */
+                spin_lock(&dq_data_lock);
                if (!IS_NOQUOTA(inode)) {
                        if (unlikely(inode_get_rsv_space(inode) > 0))
                                reserved = 1;
                        remove_inode_dquot_ref(inode, type, tofree_head);
                }
+                spin_unlock(&dq_data_lock);
        }
        spin_unlock(&inode_sb_list_lock);
 #ifdef CONFIG_QUOTA_DEBUG
@@ -1061,9 +1046,8 @@ static void drop_dquot_ref(struct super_block *sb, int type)
        LIST_HEAD(tofree_head);
        if (sb->dq_op) {
-                down_write(&sb_dqopt(sb)->dqptr_sem);
                remove_dquot_ref(sb, type, &tofree_head);
-                up_write(&sb_dqopt(sb)->dqptr_sem);
+                synchronize_srcu(&dquot_srcu);
                put_dquot_list(&tofree_head);
        }
 }
@@ -1394,21 +1378,16 @@ static int dquot_active(const struct inode *inode)
 /*
 * Initialize quota pointers in inode
 *
- * We do things in a bit complicated way but by that we avoid calling
- * dqget() and thus filesystem callbacks under dqptr_sem.
- *
 * It is better to call this function outside of any transaction as it
 * might need a lot of space in journal for dquot structure allocation.
 */
 static void __dquot_initialize(struct inode *inode, int type)
 {
-        int cnt;
+        int cnt, init_needed = 0;
        struct dquot *got[MAXQUOTAS];
        struct super_block *sb = inode->i_sb;
        qsize_t rsv;
-        /* First test before acquiring mutex - solves deadlocks when we
-         * re-enter the quota code and are already holding the mutex */
        if (!dquot_active(inode))
                return;
@@ -1418,6 +1397,15 @@ static void __dquot_initialize(struct inode *inode, int type)
                got[cnt] = NULL;
                if (type != -1 && cnt != type)
                        continue;
+                /*
+                 * The i_dquot should have been initialized in most cases,
+                 * we check it without locking here to avoid unnecessary
+                 * dqget()/dqput() calls.
+                 */
+                if (inode->i_dquot[cnt])
+                        continue;
+                init_needed = 1;
                switch (cnt) {
                case USRQUOTA:
                        qid = make_kqid_uid(inode->i_uid);
@@ -1429,7 +1417,11 @@ static void __dquot_initialize(struct inode *inode, int type)
                got[cnt] = dqget(sb, qid);
        }
-        down_write(&sb_dqopt(sb)->dqptr_sem);
+        /* All required i_dquot has been initialized */
+        if (!init_needed)
+                return;
+        spin_lock(&dq_data_lock);
        if (IS_NOQUOTA(inode))
                goto out_err;
        for (cnt = 0; cnt < MAXQUOTAS; cnt++) {
@@ -1449,15 +1441,12 @@ static void __dquot_initialize(struct inode *inode, int type)
                         * did a write before quota was turned on
                         */
                        rsv = inode_get_rsv_space(inode);
-                        if (unlikely(rsv)) {
+                        if (unlikely(rsv))
-                                spin_lock(&dq_data_lock);
                                dquot_resv_space(inode->i_dquot[cnt], rsv);
-                                spin_unlock(&dq_data_lock);
-                        }
                }
        }
 out_err:
-        up_write(&sb_dqopt(sb)->dqptr_sem);
+        spin_unlock(&dq_data_lock);
        /* Drop unused references */
        dqput_all(got);
 }
@@ -1469,19 +1458,24 @@ void dquot_initialize(struct inode *inode)
 EXPORT_SYMBOL(dquot_initialize);
 /*
- *      Release all quotas referenced by inode
+ * Release all quotas referenced by inode.
+ *
+ * This function only be called on inode free or converting
+ * a file to quota file, no other users for the i_dquot in
+ * both cases, so we needn't call synchronize_srcu() after
+ * clearing i_dquot.
 */
 static void __dquot_drop(struct inode *inode)
 {
        int cnt;
        struct dquot *put[MAXQUOTAS];
-        down_write(&sb_dqopt(inode->i_sb)->dqptr_sem);
+        spin_lock(&dq_data_lock);
        for (cnt = 0; cnt < MAXQUOTAS; cnt++) {
                put[cnt] = inode->i_dquot[cnt];
                inode->i_dquot[cnt] = NULL;
        }
-        up_write(&sb_dqopt(inode->i_sb)->dqptr_sem);
+        spin_unlock(&dq_data_lock);
        dqput_all(put);
 }
@@ -1599,15 +1593,11 @@ static void inode_decr_space(struct inode *inode, qsize_t number, int reserve)
 */
 int __dquot_alloc_space(struct inode *inode, qsize_t number, int flags)
 {
-        int cnt, ret = 0;
+        int cnt, ret = 0, index;
        struct dquot_warn warn[MAXQUOTAS];
        struct dquot **dquots = inode->i_dquot;
        int reserve = flags & DQUOT_SPACE_RESERVE;
-        /*
-         * First test before acquiring mutex - solves deadlocks when we
-         * re-enter the quota code and are already holding the mutex
-         */
        if (!dquot_active(inode)) {
                inode_incr_space(inode, number, reserve);
                goto out;
@@ -1616,7 +1606,7 @@ int __dquot_alloc_space(struct inode *inode, qsize_t number, int flags)
        for (cnt = 0; cnt < MAXQUOTAS; cnt++)
                warn[cnt].w_type = QUOTA_NL_NOWARN;
-        down_read(&sb_dqopt(inode->i_sb)->dqptr_sem);
+        index = srcu_read_lock(&dquot_srcu);
        spin_lock(&dq_data_lock);
        for (cnt = 0; cnt < MAXQUOTAS; cnt++) {
                if (!dquots[cnt])
@@ -1643,7 +1633,7 @@ int __dquot_alloc_space(struct inode *inode, qsize_t number, int flags)
                goto out_flush_warn;
        mark_all_dquot_dirty(dquots);
 out_flush_warn:
-        up_read(&sb_dqopt(inode->i_sb)->dqptr_sem);
+        srcu_read_unlock(&dquot_srcu, index);
        flush_warnings(warn);
 out:
        return ret;
@@ -1655,17 +1645,16 @@ EXPORT_SYMBOL(__dquot_alloc_space);
 */
 int dquot_alloc_inode(const struct inode *inode)
 {
-        int cnt, ret = 0;
+        int cnt, ret = 0, index;
        struct dquot_warn warn[MAXQUOTAS];
        struct dquot * const *dquots = inode->i_dquot;
-        /* First test before acquiring mutex - solves deadlocks when we
-         * re-enter the quota code and are already holding the mutex */
        if (!dquot_active(inode))
                return 0;
        for (cnt = 0; cnt < MAXQUOTAS; cnt++)
                warn[cnt].w_type = QUOTA_NL_NOWARN;
-        down_read(&sb_dqopt(inode->i_sb)->dqptr_sem);
+        index = srcu_read_lock(&dquot_srcu);
        spin_lock(&dq_data_lock);
        for (cnt = 0; cnt < MAXQUOTAS; cnt++) {
                if (!dquots[cnt])
@@ -1685,7 +1674,7 @@ warn_put_all:
        spin_unlock(&dq_data_lock);
        if (ret == 0)
                mark_all_dquot_dirty(dquots);
-        up_read(&sb_dqopt(inode->i_sb)->dqptr_sem);
+        srcu_read_unlock(&dquot_srcu, index);
        flush_warnings(warn);
        return ret;
 }
@@ -1696,14 +1685,14 @@ EXPORT_SYMBOL(dquot_alloc_inode);
 */
 int dquot_claim_space_nodirty(struct inode *inode, qsize_t number)
 {
-        int cnt;
+        int cnt, index;
        if (!dquot_active(inode)) {
                inode_claim_rsv_space(inode, number);
                return 0;
        }
-        down_read(&sb_dqopt(inode->i_sb)->dqptr_sem);
+        index = srcu_read_lock(&dquot_srcu);
        spin_lock(&dq_data_lock);
        /* Claim reserved quotas to allocated quotas */
        for (cnt = 0; cnt < MAXQUOTAS; cnt++) {
@@ -1715,7 +1704,7 @@ int dquot_claim_space_nodirty(struct inode *inode, qsize_t number)
        inode_claim_rsv_space(inode, number);
        spin_unlock(&dq_data_lock);
        mark_all_dquot_dirty(inode->i_dquot);
-        up_read(&sb_dqopt(inode->i_sb)->dqptr_sem);
+        srcu_read_unlock(&dquot_srcu, index);
        return 0;
 }
 EXPORT_SYMBOL(dquot_claim_space_nodirty);
@@ -1725,14 +1714,14 @@ EXPORT_SYMBOL(dquot_claim_space_nodirty);
 */
 void dquot_reclaim_space_nodirty(struct inode *inode, qsize_t number)
 {
-        int cnt;
+        int cnt, index;
        if (!dquot_active(inode)) {
                inode_reclaim_rsv_space(inode, number);
                return;
        }
-        down_read(&sb_dqopt(inode->i_sb)->dqptr_sem);
+        index = srcu_read_lock(&dquot_srcu);
        spin_lock(&dq_data_lock);
        /* Claim reserved quotas to allocated quotas */
        for (cnt = 0; cnt < MAXQUOTAS; cnt++) {
@@ -1744,7 +1733,7 @@ void dquot_reclaim_space_nodirty(struct inode *inode, qsize_t number)
        inode_reclaim_rsv_space(inode, number);
        spin_unlock(&dq_data_lock);
        mark_all_dquot_dirty(inode->i_dquot);
-        up_read(&sb_dqopt(inode->i_sb)->dqptr_sem);
+        srcu_read_unlock(&dquot_srcu, index);
        return;
 }
 EXPORT_SYMBOL(dquot_reclaim_space_nodirty);
@@ -1757,16 +1746,14 @@ void __dquot_free_space(struct inode *inode, qsize_t number, int flags)
        unsigned int cnt;
        struct dquot_warn warn[MAXQUOTAS];
        struct dquot **dquots = inode->i_dquot;
-        int reserve = flags & DQUOT_SPACE_RESERVE;
+        int reserve = flags & DQUOT_SPACE_RESERVE, index;
-        /* First test before acquiring mutex - solves deadlocks when we
-         * re-enter the quota code and are already holding the mutex */
        if (!dquot_active(inode)) {
                inode_decr_space(inode, number, reserve);
                return;
        }
-        down_read(&sb_dqopt(inode->i_sb)->dqptr_sem);
+        index = srcu_read_lock(&dquot_srcu);
        spin_lock(&dq_data_lock);
        for (cnt = 0; cnt < MAXQUOTAS; cnt++) {
                int wtype;
@@ -1789,7 +1776,7 @@ void __dquot_free_space(struct inode *inode, qsize_t number, int flags)
                goto out_unlock;
        mark_all_dquot_dirty(dquots);
 out_unlock:
-        up_read(&sb_dqopt(inode->i_sb)->dqptr_sem);
+        srcu_read_unlock(&dquot_srcu, index);
        flush_warnings(warn);
 }
 EXPORT_SYMBOL(__dquot_free_space);
@@ -1802,13 +1789,12 @@ void dquot_free_inode(const struct inode *inode)
        unsigned int cnt;
        struct dquot_warn warn[MAXQUOTAS];
        struct dquot * const *dquots = inode->i_dquot;
+        int index;
-        /* First test before acquiring mutex - solves deadlocks when we
-         * re-enter the quota code and are already holding the mutex */
        if (!dquot_active(inode))
                return;
-        down_read(&sb_dqopt(inode->i_sb)->dqptr_sem);
+        index = srcu_read_lock(&dquot_srcu);
        spin_lock(&dq_data_lock);
        for (cnt = 0; cnt < MAXQUOTAS; cnt++) {
                int wtype;
@@ -1823,7 +1809,7 @@ void dquot_free_inode(const struct inode *inode)
        }
        spin_unlock(&dq_data_lock);
        mark_all_dquot_dirty(dquots);
-        up_read(&sb_dqopt(inode->i_sb)->dqptr_sem);
+        srcu_read_unlock(&dquot_srcu, index);
        flush_warnings(warn);
 }
 EXPORT_SYMBOL(dquot_free_inode);
@@ -1837,6 +1823,8 @@ EXPORT_SYMBOL(dquot_free_inode);
 * This operation can block, but only after everything is updated
 * A transaction must be started when entering this function.
 *
+ * We are holding reference on transfer_from & transfer_to, no need to
+ * protect them by srcu_read_lock().
 */
 int __dquot_transfer(struct inode *inode, struct dquot **transfer_to)
 {
@@ -1849,8 +1837,6 @@ int __dquot_transfer(struct inode *inode, struct dquot **transfer_to)
        struct dquot_warn warn_from_inodes[MAXQUOTAS];
        struct dquot_warn warn_from_space[MAXQUOTAS];
-        /* First test before acquiring mutex - solves deadlocks when we
-         * re-enter the quota code and are already holding the mutex */
        if (IS_NOQUOTA(inode))
                return 0;
        /* Initialize the arrays */
@@ -1859,12 +1845,12 @@ int __dquot_transfer(struct inode *inode, struct dquot **transfer_to)
                warn_from_inodes[cnt].w_type = QUOTA_NL_NOWARN;
                warn_from_space[cnt].w_type = QUOTA_NL_NOWARN;
        }
-        down_write(&sb_dqopt(inode->i_sb)->dqptr_sem);
+        spin_lock(&dq_data_lock);
        if (IS_NOQUOTA(inode)) {        /* File without quota accounting? */
-                up_write(&sb_dqopt(inode->i_sb)->dqptr_sem);
+                spin_unlock(&dq_data_lock);
                return 0;
        }
-        spin_lock(&dq_data_lock);
        cur_space = inode_get_bytes(inode);
        rsv_space = inode_get_rsv_space(inode);
        space = cur_space + rsv_space;
@@ -1918,7 +1904,6 @@ int __dquot_transfer(struct inode *inode, struct dquot **transfer_to)
                inode->i_dquot[cnt] = transfer_to[cnt];
        }
        spin_unlock(&dq_data_lock);
-        up_write(&sb_dqopt(inode->i_sb)->dqptr_sem);
        mark_all_dquot_dirty(transfer_from);
        mark_all_dquot_dirty(transfer_to);
@@ -1932,7 +1917,6 @@ int __dquot_transfer(struct inode *inode, struct dquot **transfer_to)
        return 0;
 over_quota:
        spin_unlock(&dq_data_lock);
-        up_write(&sb_dqopt(inode->i_sb)->dqptr_sem);
        flush_warnings(warn_to);
        return ret;
 }
diff --git a/fs/quota/kqid.c b/fs/quota/kqid.c
index 2f97b0e2c501..ebc5e6285800 100644
--- a/fs/quota/kqid.c
+++ b/fs/quota/kqid.c
@@ -55,7 +55,7 @@ EXPORT_SYMBOL(qid_lt);
 /**
 *      from_kqid - Create a qid from a kqid user-namespace pair.
 *      @targ: The user namespace we want a qid in.
- *      @kuid: The kernel internal quota identifier to start with.
+ *      @kqid: The kernel internal quota identifier to start with.
 *
 *      Map @kqid into the user-namespace specified by @targ and
 *      return the resulting qid.
diff --git a/fs/quota/netlink.c b/fs/quota/netlink.c
index 72d29177998e..bb2869f5dfd8 100644
--- a/fs/quota/netlink.c
+++ b/fs/quota/netlink.c
@@ -32,8 +32,7 @@ static struct genl_family quota_genl_family = {
 /**
 * quota_send_warning - Send warning to userspace about exceeded quota
- * @type: The quota type: USRQQUOTA, GRPQUOTA,...
+ * @qid: The kernel internal quota identifier.
- * @id: The user or group id of the quota that was exceeded
 * @dev: The device on which the fs is mounted (sb->s_dev)
 * @warntype: The type of the warning: QUOTA_NL_...
 *
diff --git a/fs/quota/quota.c b/fs/quota/quota.c
index ff3f0b3cfdb3..75621649dbd7 100644
--- a/fs/quota/quota.c
+++ b/fs/quota/quota.c
@@ -79,13 +79,13 @@ static int quota_getfmt(struct super_block *sb, int type, void __user *addr)
 {
        __u32 fmt;
-        down_read(&sb_dqopt(sb)->dqptr_sem);
+        mutex_lock(&sb_dqopt(sb)->dqonoff_mutex);
        if (!sb_has_quota_active(sb, type)) {
-                up_read(&sb_dqopt(sb)->dqptr_sem);
+                mutex_unlock(&sb_dqopt(sb)->dqonoff_mutex);
                return -ESRCH;
        }
        fmt = sb_dqopt(sb)->info[type].dqi_format->qf_fmt_id;
-        up_read(&sb_dqopt(sb)->dqptr_sem);
+        mutex_unlock(&sb_dqopt(sb)->dqonoff_mutex);
        if (copy_to_user(addr, &fmt, sizeof(fmt)))
                return -EFAULT;
        return 0;
diff --git a/fs/reiserfs/do_balan.c b/fs/reiserfs/do_balan.c
index 5739cb99de7b..9c02d96d3a42 100644
--- a/fs/reiserfs/do_balan.c
+++ b/fs/reiserfs/do_balan.c
@@ -286,12 +286,14 @@ static int balance_leaf_when_delete(struct tree_balance *tb, int flag)
        return 0;
 }
-static void balance_leaf_insert_left(struct tree_balance *tb,
+static unsigned int balance_leaf_insert_left(struct tree_balance *tb,
-                                     struct item_head *ih, const char *body)
+                                             struct item_head *const ih,
+                                             const char * const body)
 {
        int ret;
        struct buffer_info bi;
        int n = B_NR_ITEMS(tb->L[0]);
+        unsigned body_shift_bytes = 0;
        if (tb->item_pos == tb->lnum[0] - 1 && tb->lbytes != -1) {
                /* part of new item falls into L[0] */
@@ -329,7 +331,7 @@ static void balance_leaf_insert_left(struct tree_balance *tb,
                put_ih_item_len(ih, new_item_len);
                if (tb->lbytes > tb->zeroes_num) {
-                        body += (tb->lbytes - tb->zeroes_num);
+                        body_shift_bytes = tb->lbytes - tb->zeroes_num;
                        tb->zeroes_num = 0;
                } else
                        tb->zeroes_num -= tb->lbytes;
@@ -349,11 +351,12 @@ static void balance_leaf_insert_left(struct tree_balance *tb,
                tb->insert_size[0] = 0;
                tb->zeroes_num = 0;
        }
+        return body_shift_bytes;
 }
 static void balance_leaf_paste_left_shift_dirent(struct tree_balance *tb,
-                                                 struct item_head *ih,
+                                                 struct item_head * const ih,
-                                                 const char *body)
+                                                 const char * const body)
 {
        int n = B_NR_ITEMS(tb->L[0]);
        struct buffer_info bi;
@@ -413,17 +416,18 @@ static void balance_leaf_paste_left_shift_dirent(struct tree_balance *tb,
        tb->pos_in_item -= tb->lbytes;
 }
-static void balance_leaf_paste_left_shift(struct tree_balance *tb,
+static unsigned int balance_leaf_paste_left_shift(struct tree_balance *tb,
-                                          struct item_head *ih,
+                                                  struct item_head * const ih,
-                                          const char *body)
+                                                  const char * const body)
 {
        struct buffer_head *tbS0 = PATH_PLAST_BUFFER(tb->tb_path);
        int n = B_NR_ITEMS(tb->L[0]);
        struct buffer_info bi;
+        int body_shift_bytes = 0;
        if (is_direntry_le_ih(item_head(tbS0, tb->item_pos))) {
                balance_leaf_paste_left_shift_dirent(tb, ih, body);
-                return;
+                return 0;
        }
        RFALSE(tb->lbytes <= 0,
@@ -497,7 +501,7 @@ static void balance_leaf_paste_left_shift(struct tree_balance *tb,
                 * insert_size[0]
                 */
                if (l_n > tb->zeroes_num) {
-                        body += (l_n - tb->zeroes_num);
+                        body_shift_bytes = l_n - tb->zeroes_num;
                        tb->zeroes_num = 0;
                } else
                        tb->zeroes_num -= l_n;
@@ -526,13 +530,14 @@ static void balance_leaf_paste_left_shift(struct tree_balance *tb,
                 */
                leaf_shift_left(tb, tb->lnum[0], tb->lbytes);
        }
+        return body_shift_bytes;
 }
 /* appended item will be in L[0] in whole */
 static void balance_leaf_paste_left_whole(struct tree_balance *tb,
-                                          struct item_head *ih,
+                                          struct item_head * const ih,
-                                          const char *body)
+                                          const char * const body)
 {
        struct buffer_head *tbS0 = PATH_PLAST_BUFFER(tb->tb_path);
        int n = B_NR_ITEMS(tb->L[0]);
@@ -584,39 +589,44 @@ static void balance_leaf_paste_left_whole(struct tree_balance *tb,
        tb->zeroes_num = 0;
 }
-static void balance_leaf_paste_left(struct tree_balance *tb,
+static unsigned int balance_leaf_paste_left(struct tree_balance *tb,
-                                    struct item_head *ih, const char *body)
+                                            struct item_head * const ih,
+                                            const char * const body)
 {
        /* we must shift the part of the appended item */
        if (tb->item_pos == tb->lnum[0] - 1 && tb->lbytes != -1)
-                balance_leaf_paste_left_shift(tb, ih, body);
+                return balance_leaf_paste_left_shift(tb, ih, body);
        else
                balance_leaf_paste_left_whole(tb, ih, body);
+        return 0;
 }
 /* Shift lnum[0] items from S[0] to the left neighbor L[0] */
-static void balance_leaf_left(struct tree_balance *tb, struct item_head *ih,
+static unsigned int balance_leaf_left(struct tree_balance *tb,
-                              const char *body, int flag)
+                                      struct item_head * const ih,
+                                      const char * const body, int flag)
 {
        if (tb->lnum[0] <= 0)
-                return;
+                return 0;
        /* new item or it part falls to L[0], shift it too */
        if (tb->item_pos < tb->lnum[0]) {
                BUG_ON(flag != M_INSERT && flag != M_PASTE);
                if (flag == M_INSERT)
-                        balance_leaf_insert_left(tb, ih, body);
+                        return balance_leaf_insert_left(tb, ih, body);
                else /* M_PASTE */
-                        balance_leaf_paste_left(tb, ih, body);
+                        return balance_leaf_paste_left(tb, ih, body);
        } else
                /* new item doesn't fall into L[0] */
                leaf_shift_left(tb, tb->lnum[0], tb->lbytes);
+        return 0;
 }
 static void balance_leaf_insert_right(struct tree_balance *tb,
-                                      struct item_head *ih, const char *body)
+                                      struct item_head * const ih,
+                                      const char * const body)
 {
        struct buffer_head *tbS0 = PATH_PLAST_BUFFER(tb->tb_path);
@@ -704,7 +714,8 @@ static void balance_leaf_insert_right(struct tree_balance *tb,
 static void balance_leaf_paste_right_shift_dirent(struct tree_balance *tb,
-                                     struct item_head *ih, const char *body)
+                                     struct item_head * const ih,
+                                     const char * const body)
 {
        struct buffer_head *tbS0 = PATH_PLAST_BUFFER(tb->tb_path);
        struct buffer_info bi;
@@ -754,7 +765,8 @@ static void balance_leaf_paste_right_shift_dirent(struct tree_balance *tb,
 }
 static void balance_leaf_paste_right_shift(struct tree_balance *tb,
-                                     struct item_head *ih, const char *body)
+                                     struct item_head * const ih,
+                                     const char * const body)
 {
        struct buffer_head *tbS0 = PATH_PLAST_BUFFER(tb->tb_path);
        int n_shift, n_rem, r_zeroes_number, version;
@@ -831,7 +843,8 @@ static void balance_leaf_paste_right_shift(struct tree_balance *tb,
 }
 static void balance_leaf_paste_right_whole(struct tree_balance *tb,
-                                     struct item_head *ih, const char *body)
+                                     struct item_head * const ih,
+                                     const char * const body)
 {
        struct buffer_head *tbS0 = PATH_PLAST_BUFFER(tb->tb_path);
        int n = B_NR_ITEMS(tbS0);
@@ -874,7 +887,8 @@ static void balance_leaf_paste_right_whole(struct tree_balance *tb,
 }
 static void balance_leaf_paste_right(struct tree_balance *tb,
-                                     struct item_head *ih, const char *body)
+                                     struct item_head * const ih,
+                                     const char * const body)
 {
        struct buffer_head *tbS0 = PATH_PLAST_BUFFER(tb->tb_path);
        int n = B_NR_ITEMS(tbS0);
@@ -896,8 +910,9 @@ static void balance_leaf_paste_right(struct tree_balance *tb,
 }
 /* shift rnum[0] items from S[0] to the right neighbor R[0] */
-static void balance_leaf_right(struct tree_balance *tb, struct item_head *ih,
+static void balance_leaf_right(struct tree_balance *tb,
-                               const char *body, int flag)
+                               struct item_head * const ih,
+                               const char * const body, int flag)
 {
        if (tb->rnum[0] <= 0)
                return;
@@ -911,8 +926,8 @@ static void balance_leaf_right(struct tree_balance *tb, struct item_head *ih,
 }
 static void balance_leaf_new_nodes_insert(struct tree_balance *tb,
-                                          struct item_head *ih,
+                                          struct item_head * const ih,
-                                          const char *body,
+                                          const char * const body,
                                          struct item_head *insert_key,
                                          struct buffer_head **insert_ptr,
                                          int i)
@@ -1003,8 +1018,8 @@ static void balance_leaf_new_nodes_insert(struct tree_balance *tb,
 /* we append to directory item */
 static void balance_leaf_new_nodes_paste_dirent(struct tree_balance *tb,
-                                         struct item_head *ih,
+                                         struct item_head * const ih,
-                                         const char *body,
+                                         const char * const body,
                                         struct item_head *insert_key,
                                         struct buffer_head **insert_ptr,
                                         int i)
@@ -1058,8 +1073,8 @@ static void balance_leaf_new_nodes_paste_dirent(struct tree_balance *tb,
 }
 static void balance_leaf_new_nodes_paste_shift(struct tree_balance *tb,
-                                         struct item_head *ih,
+                                         struct item_head * const ih,
-                                         const char *body,
+                                         const char * const body,
                                         struct item_head *insert_key,
                                         struct buffer_head **insert_ptr,
                                         int i)
@@ -1131,8 +1146,8 @@ static void balance_leaf_new_nodes_paste_shift(struct tree_balance *tb,
 }
 static void balance_leaf_new_nodes_paste_whole(struct tree_balance *tb,
-                                               struct item_head *ih,
+                                               struct item_head * const ih,
-                                               const char *body,
+                                               const char * const body,
                                               struct item_head *insert_key,
                                               struct buffer_head **insert_ptr,
                                               int i)
@@ -1184,8 +1199,8 @@ static void balance_leaf_new_nodes_paste_whole(struct tree_balance *tb,
 }
 static void balance_leaf_new_nodes_paste(struct tree_balance *tb,
-                                         struct item_head *ih,
+                                         struct item_head * const ih,
-                                         const char *body,
+                                         const char * const body,
                                         struct item_head *insert_key,
                                         struct buffer_head **insert_ptr,
                                         int i)
@@ -1214,8 +1229,8 @@ static void balance_leaf_new_nodes_paste(struct tree_balance *tb,
 /* Fill new nodes that appear in place of S[0] */
 static void balance_leaf_new_nodes(struct tree_balance *tb,
-                                   struct item_head *ih,
+                                   struct item_head * const ih,
-                                   const char *body,
+                                   const char * const body,
                                   struct item_head *insert_key,
                                   struct buffer_head **insert_ptr,
                                   int flag)
@@ -1254,8 +1269,8 @@ static void balance_leaf_new_nodes(struct tree_balance *tb,
 }
 static void balance_leaf_finish_node_insert(struct tree_balance *tb,
-                                            struct item_head *ih,
+                                            struct item_head * const ih,
-                                            const char *body)
+                                            const char * const body)
 {
        struct buffer_head *tbS0 = PATH_PLAST_BUFFER(tb->tb_path);
        struct buffer_info bi;
@@ -1271,8 +1286,8 @@ static void balance_leaf_finish_node_insert(struct tree_balance *tb,
 }
 static void balance_leaf_finish_node_paste_dirent(struct tree_balance *tb,
-                                                  struct item_head *ih,
+                                                  struct item_head * const ih,
-                                                  const char *body)
+                                                  const char * const body)
 {
        struct buffer_head *tbS0 = PATH_PLAST_BUFFER(tb->tb_path);
        struct item_head *pasted = item_head(tbS0, tb->item_pos);
@@ -1305,8 +1320,8 @@ static void balance_leaf_finish_node_paste_dirent(struct tree_balance *tb,
 }
 static void balance_leaf_finish_node_paste(struct tree_balance *tb,
-                                           struct item_head *ih,
+                                           struct item_head * const ih,
-                                           const char *body)
+                                           const char * const body)
 {
        struct buffer_head *tbS0 = PATH_PLAST_BUFFER(tb->tb_path);
        struct buffer_info bi;
@@ -1349,8 +1364,8 @@ static void balance_leaf_finish_node_paste(struct tree_balance *tb,
 * of the affected item which remains in S
 */
 static void balance_leaf_finish_node(struct tree_balance *tb,
-                                      struct item_head *ih,
+                                      struct item_head * const ih,
-                                      const char *body, int flag)
+                                      const char * const body, int flag)
 {
        /* if we must insert or append into buffer S[0] */
        if (0 <= tb->item_pos && tb->item_pos < tb->s0num) {
@@ -1402,7 +1417,7 @@ static int balance_leaf(struct tree_balance *tb, struct item_head *ih,
            && is_indirect_le_ih(item_head(tbS0, tb->item_pos)))
                tb->pos_in_item *= UNFM_P_SIZE;
-        balance_leaf_left(tb, ih, body, flag);
+        body += balance_leaf_left(tb, ih, body, flag);
        /* tb->lnum[0] > 0 */
        /* Calculate new item position */
diff --git a/fs/reiserfs/journal.c b/fs/reiserfs/journal.c
index e8870de4627e..a88b1b3e7db3 100644
--- a/fs/reiserfs/journal.c
+++ b/fs/reiserfs/journal.c
@@ -1947,8 +1947,6 @@ static int do_journal_release(struct reiserfs_transaction_handle *th,
                }
        }
-        /* wait for all commits to finish */
-        cancel_delayed_work(&SB_JOURNAL(sb)->j_work);
        /*
         * We must release the write lock here because
@@ -1956,8 +1954,14 @@ static int do_journal_release(struct reiserfs_transaction_handle *th,
         */
        reiserfs_write_unlock(sb);
+        /*
+         * Cancel flushing of old commits. Note that neither of these works
+         * will be requeued because superblock is being shutdown and doesn't
+         * have MS_ACTIVE set.
+         */
        cancel_delayed_work_sync(&REISERFS_SB(sb)->old_work);
-        flush_workqueue(REISERFS_SB(sb)->commit_wq);
+        /* wait for all commits to finish */
+        cancel_delayed_work_sync(&SB_JOURNAL(sb)->j_work);
        free_journal_ram(sb);
@@ -4292,9 +4296,15 @@ static int do_journal_end(struct reiserfs_transaction_handle *th, int flags)
        if (flush) {
                flush_commit_list(sb, jl, 1);
                flush_journal_list(sb, jl, 1);
-        } else if (!(jl->j_state & LIST_COMMIT_PENDING))
+        } else if (!(jl->j_state & LIST_COMMIT_PENDING)) {
-                queue_delayed_work(REISERFS_SB(sb)->commit_wq,
+                /*
-                                   &journal->j_work, HZ / 10);
+                 * Avoid queueing work when sb is being shut down. Transaction
+                 * will be flushed on journal shutdown.
+                 */
+                if (sb->s_flags & MS_ACTIVE)
+                        queue_delayed_work(REISERFS_SB(sb)->commit_wq,
+                                           &journal->j_work, HZ / 10);
+        }
        /*
         * if the next transaction has any chance of wrapping, flush
diff --git a/fs/reiserfs/lbalance.c b/fs/reiserfs/lbalance.c
index 814dda3ec998..249594a821e0 100644
--- a/fs/reiserfs/lbalance.c
+++ b/fs/reiserfs/lbalance.c
@@ -899,8 +899,9 @@ void leaf_delete_items(struct buffer_info *cur_bi, int last_first,
 /* insert item into the leaf node in position before */
 void leaf_insert_into_buf(struct buffer_info *bi, int before,
-                          struct item_head *inserted_item_ih,
+                          struct item_head * const inserted_item_ih,
-                          const char *inserted_item_body, int zeros_number)
+                          const char * const inserted_item_body,
+                          int zeros_number)
 {
        struct buffer_head *bh = bi->bi_bh;
        int nr, free_space;
diff --git a/fs/reiserfs/reiserfs.h b/fs/reiserfs/reiserfs.h
index bf53888c7f59..735c2c2b4536 100644
--- a/fs/reiserfs/reiserfs.h
+++ b/fs/reiserfs/reiserfs.h
@@ -3216,11 +3216,12 @@ int leaf_shift_right(struct tree_balance *tb, int shift_num, int shift_bytes);
 void leaf_delete_items(struct buffer_info *cur_bi, int last_first, int first,
                       int del_num, int del_bytes);
 void leaf_insert_into_buf(struct buffer_info *bi, int before,
-                          struct item_head *inserted_item_ih,
+                          struct item_head * const inserted_item_ih,
-                          const char *inserted_item_body, int zeros_number);
+                          const char * const inserted_item_body,
-void leaf_paste_in_buffer(struct buffer_info *bi, int pasted_item_num,
-                          int pos_in_item, int paste_size, const char *body,
                          int zeros_number);
+void leaf_paste_in_buffer(struct buffer_info *bi, int pasted_item_num,
+                          int pos_in_item, int paste_size,
+                          const char * const body, int zeros_number);
 void leaf_cut_from_buffer(struct buffer_info *bi, int cut_item_num,
                          int pos_in_item, int cut_size);
 void leaf_paste_entries(struct buffer_info *bi, int item_num, int before,
diff --git a/fs/reiserfs/super.c b/fs/reiserfs/super.c
index 709ea92d716f..d46e88a33b02 100644
--- a/fs/reiserfs/super.c
+++ b/fs/reiserfs/super.c
@@ -100,7 +100,11 @@ void reiserfs_schedule_old_flush(struct super_block *s)
        struct reiserfs_sb_info *sbi = REISERFS_SB(s);
        unsigned long delay;
-        if (s->s_flags & MS_RDONLY)
+        /*
+         * Avoid scheduling flush when sb is being shut down. It can race
+         * with journal shutdown and free still queued delayed work.
+         */
+        if (s->s_flags & MS_RDONLY || !(s->s_flags & MS_ACTIVE))
                return;
        spin_lock(&sbi->old_work_lock);
diff --git a/fs/super.c b/fs/super.c
index d20d5b11dedf..b9a214d2fe98 100644
--- a/fs/super.c
+++ b/fs/super.c
@@ -22,7 +22,6 @@
 #include <linux/export.h>
 #include <linux/slab.h>
-#include <linux/acct.h>
 #include <linux/blkdev.h>
 #include <linux/mount.h>
 #include <linux/security.h>
@@ -218,7 +217,6 @@ static struct super_block *alloc_super(struct file_system_type *type, int flags)
        lockdep_set_class(&s->s_vfs_rename_mutex, &type->s_vfs_rename_key);
        mutex_init(&s->s_dquot.dqio_mutex);
        mutex_init(&s->s_dquot.dqonoff_mutex);
-        init_rwsem(&s->s_dquot.dqptr_sem);
        s->s_maxbytes = MAX_NON_LFS;
        s->s_op = &default_op;
        s->s_time_gran = 1000000000;
@@ -702,12 +700,22 @@ int do_remount_sb(struct super_block *sb, int flags, void *data, int force)
                return -EACCES;
 #endif
-        if (flags & MS_RDONLY)
-                acct_auto_close(sb);
-        shrink_dcache_sb(sb);
        remount_ro = (flags & MS_RDONLY) && !(sb->s_flags & MS_RDONLY);
+        if (remount_ro) {
+                if (sb->s_pins.first) {
+                        up_write(&sb->s_umount);
+                        sb_pin_kill(sb);
+                        down_write(&sb->s_umount);
+                        if (!sb->s_root)
+                                return 0;
+                        if (sb->s_writers.frozen != SB_UNFROZEN)
+                                return -EBUSY;
+                        remount_ro = (flags & MS_RDONLY) && !(sb->s_flags & MS_RDONLY);
+                }
+        }
+        shrink_dcache_sb(sb);
        /* If we are remounting RDONLY and current sb is read/write,
           make sure there are no rw files opened */
        if (remount_ro) {
diff --git a/fs/ubifs/commit.c b/fs/ubifs/commit.c
index ff8229340cd5..aa13ad053b14 100644
--- a/fs/ubifs/commit.c
+++ b/fs/ubifs/commit.c
@@ -174,7 +174,6 @@ static int do_commit(struct ubifs_info *c)
        if (err)
                goto out;
-        mutex_lock(&c->mst_mutex);
        c->mst_node->cmt_no      = cpu_to_le64(c->cmt_no);
        c->mst_node->log_lnum    = cpu_to_le32(new_ltail_lnum);
        c->mst_node->root_lnum   = cpu_to_le32(zroot.lnum);
@@ -204,7 +203,6 @@ static int do_commit(struct ubifs_info *c)
        else
                c->mst_node->flags &= ~cpu_to_le32(UBIFS_MST_NO_ORPHS);
        err = ubifs_write_master(c);
-        mutex_unlock(&c->mst_mutex);
        if (err)
                goto out;
diff --git a/fs/ubifs/io.c b/fs/ubifs/io.c
index 2290d5866725..fb08b0c514b6 100644
--- a/fs/ubifs/io.c
+++ b/fs/ubifs/io.c
@@ -431,7 +431,7 @@ void ubifs_prep_grp_node(struct ubifs_info *c, void *node, int len, int last)
 /**
 * wbuf_timer_callback - write-buffer timer callback function.
- * @data: timer data (write-buffer descriptor)
+ * @timer: timer data (write-buffer descriptor)
 *
 * This function is called when the write-buffer timer expires.
 */
diff --git a/fs/ubifs/log.c b/fs/ubifs/log.c
index a902c5919e42..a47ddfc9be6b 100644
--- a/fs/ubifs/log.c
+++ b/fs/ubifs/log.c
@@ -240,6 +240,7 @@ int ubifs_add_bud_to_log(struct ubifs_info *c, int jhead, int lnum, int offs)
        if (c->lhead_offs > c->leb_size - c->ref_node_alsz) {
                c->lhead_lnum = ubifs_next_log_lnum(c, c->lhead_lnum);
+                ubifs_assert(c->lhead_lnum != c->ltail_lnum);
                c->lhead_offs = 0;
        }
@@ -404,15 +405,14 @@ int ubifs_log_start_commit(struct ubifs_info *c, int *ltail_lnum)
        /* Switch to the next log LEB */
        if (c->lhead_offs) {
                c->lhead_lnum = ubifs_next_log_lnum(c, c->lhead_lnum);
+                ubifs_assert(c->lhead_lnum != c->ltail_lnum);
                c->lhead_offs = 0;
        }
-        if (c->lhead_offs == 0) {
+        /* Must ensure next LEB has been unmapped */
-                /* Must ensure next LEB has been unmapped */
+        err = ubifs_leb_unmap(c, c->lhead_lnum);
-                err = ubifs_leb_unmap(c, c->lhead_lnum);
+        if (err)
-                if (err)
+                goto out;
-                        goto out;
-        }
        len = ALIGN(len, c->min_io_size);
        dbg_log("writing commit start at LEB %d:0, len %d", c->lhead_lnum, len);
diff --git a/fs/ubifs/lpt.c b/fs/ubifs/lpt.c
index d46b19ec1815..421bd0a80424 100644
--- a/fs/ubifs/lpt.c
+++ b/fs/ubifs/lpt.c
@@ -1464,7 +1464,6 @@ struct ubifs_lprops *ubifs_lpt_lookup(struct ubifs_info *c, int lnum)
                        return ERR_CAST(nnode);
        }
        iip = ((i >> shft) & (UBIFS_LPT_FANOUT - 1));
-        shft -= UBIFS_LPT_FANOUT_SHIFT;
        pnode = ubifs_get_pnode(c, nnode, iip);
        if (IS_ERR(pnode))
                return ERR_CAST(pnode);
@@ -1604,7 +1603,6 @@ struct ubifs_lprops *ubifs_lpt_lookup_dirty(struct ubifs_info *c, int lnum)
                        return ERR_CAST(nnode);
        }
        iip = ((i >> shft) & (UBIFS_LPT_FANOUT - 1));
-        shft -= UBIFS_LPT_FANOUT_SHIFT;
        pnode = ubifs_get_pnode(c, nnode, iip);
        if (IS_ERR(pnode))
                return ERR_CAST(pnode);
@@ -1964,7 +1962,6 @@ again:
                }
        }
        iip = ((i >> shft) & (UBIFS_LPT_FANOUT - 1));
-        shft -= UBIFS_LPT_FANOUT_SHIFT;
        pnode = scan_get_pnode(c, path + h, nnode, iip);
        if (IS_ERR(pnode)) {
                err = PTR_ERR(pnode);
@@ -2198,6 +2195,7 @@ static int dbg_chk_pnode(struct ubifs_info *c, struct ubifs_pnode *pnode,
                                          lprops->dirty);
                                return -EINVAL;
                        }
+                        break;
                case LPROPS_FREEABLE:
                case LPROPS_FRDI_IDX:
                        if (lprops->free + lprops->dirty != c->leb_size) {
@@ -2206,6 +2204,7 @@ static int dbg_chk_pnode(struct ubifs_info *c, struct ubifs_pnode *pnode,
                                          lprops->dirty);
                                return -EINVAL;
                        }
+                        break;
                }
        }
        return 0;
diff --git a/fs/ubifs/lpt_commit.c b/fs/ubifs/lpt_commit.c
index 45d4e96a6bac..d9c02928e992 100644
--- a/fs/ubifs/lpt_commit.c
+++ b/fs/ubifs/lpt_commit.c
@@ -304,7 +304,6 @@ static int layout_cnodes(struct ubifs_info *c)
                        ubifs_assert(lnum >= c->lpt_first &&
                                     lnum <= c->lpt_last);
                }
-                done_ltab = 1;
                c->ltab_lnum = lnum;
                c->ltab_offs = offs;
                offs += c->ltab_sz;
@@ -514,7 +513,6 @@ static int write_cnodes(struct ubifs_info *c)
                        if (err)
                                return err;
                }
-                done_ltab = 1;
                ubifs_pack_ltab(c, buf + offs, c->ltab_cmt);
                offs += c->ltab_sz;
                dbg_chk_lpt_sz(c, 1, c->ltab_sz);
@@ -1941,6 +1939,11 @@ static void dump_lpt_leb(const struct ubifs_info *c, int lnum)
                                pr_err("LEB %d:%d, nnode, ",
                                       lnum, offs);
                        err = ubifs_unpack_nnode(c, p, &nnode);
+                        if (err) {
+                                pr_err("failed to unpack_node, error %d\n",
+                                       err);
+                                break;
+                        }
                        for (i = 0; i < UBIFS_LPT_FANOUT; i++) {
                                pr_cont("%d:%d", nnode.nbranch[i].lnum,
                                       nnode.nbranch[i].offs);
diff --git a/fs/ubifs/master.c b/fs/ubifs/master.c
index ab83ace9910a..1a4bb9e8b3b8 100644
--- a/fs/ubifs/master.c
+++ b/fs/ubifs/master.c
@@ -352,10 +352,9 @@ int ubifs_read_master(struct ubifs_info *c)
 * ubifs_write_master - write master node.
 * @c: UBIFS file-system description object
 *
- * This function writes the master node. The caller has to take the
+ * This function writes the master node. Returns zero in case of success and a
- * @c->mst_mutex lock before calling this function. Returns zero in case of
+ * negative error code in case of failure. The master node is written twice to
- * success and a negative error code in case of failure. The master node is
+ * enable recovery.
- * written twice to enable recovery.
 */
 int ubifs_write_master(struct ubifs_info *c)
 {
diff --git a/fs/ubifs/orphan.c b/fs/ubifs/orphan.c
index f1c3e5a1b315..4409f486ecef 100644
--- a/fs/ubifs/orphan.c
+++ b/fs/ubifs/orphan.c
@@ -346,7 +346,6 @@ static int write_orph_nodes(struct ubifs_info *c, int atomic)
                int lnum;
                /* Unmap any unused LEBs after consolidation */
-                lnum = c->ohead_lnum + 1;
                for (lnum = c->ohead_lnum + 1; lnum <= c->orph_last; lnum++) {
                        err = ubifs_leb_unmap(c, lnum);
                        if (err)
diff --git a/fs/ubifs/recovery.c b/fs/ubifs/recovery.c
index c14adb2f420c..c640938f62f0 100644
--- a/fs/ubifs/recovery.c
+++ b/fs/ubifs/recovery.c
@@ -596,7 +596,6 @@ static void drop_last_group(struct ubifs_scan_leb *sleb, int *offs)
 * drop_last_node - drop the last node.
 * @sleb: scanned LEB information
 * @offs: offset of dropped nodes is returned here
- * @grouped: non-zero if whole group of nodes have to be dropped
 *
 * This is a helper function for 'ubifs_recover_leb()' which drops the last
 * node of the scanned LEB.
@@ -629,8 +628,8 @@ static void drop_last_node(struct ubifs_scan_leb *sleb, int *offs)
 *
 * This function does a scan of a LEB, but caters for errors that might have
 * been caused by the unclean unmount from which we are attempting to recover.
- * Returns %0 in case of success, %-EUCLEAN if an unrecoverable corruption is
+ * Returns the scanned information on success and a negative error code on
- * found, and a negative error code in case of failure.
+ * failure.
 */
 struct ubifs_scan_leb *ubifs_recover_leb(struct ubifs_info *c, int lnum,
                                         int offs, void *sbuf, int jhead)
diff --git a/fs/ubifs/sb.c b/fs/ubifs/sb.c
index 4c37607a958e..79c6dbbc0e04 100644
--- a/fs/ubifs/sb.c
+++ b/fs/ubifs/sb.c
@@ -332,6 +332,8 @@ static int create_default_filesystem(struct ubifs_info *c)
        cs->ch.node_type = UBIFS_CS_NODE;
        err = ubifs_write_node(c, cs, UBIFS_CS_NODE_SZ, UBIFS_LOG_LNUM, 0);
        kfree(cs);
+        if (err)
+                return err;
        ubifs_msg("default file-system created");
        return 0;
@@ -447,7 +449,7 @@ static int validate_sb(struct ubifs_info *c, struct ubifs_sb_node *sup)
                goto failed;
        }
-        if (c->default_compr < 0 || c->default_compr >= UBIFS_COMPR_TYPES_CNT) {
+        if (c->default_compr >= UBIFS_COMPR_TYPES_CNT) {
                err = 13;
                goto failed;
        }
diff --git a/fs/ubifs/scan.c b/fs/ubifs/scan.c
index 58aa05df2bb6..89adbc4d08ac 100644
--- a/fs/ubifs/scan.c
+++ b/fs/ubifs/scan.c
@@ -131,7 +131,8 @@ int ubifs_scan_a_node(const struct ubifs_info *c, void *buf, int len, int lnum,
 * @offs: offset to start at (usually zero)
 * @sbuf: scan buffer (must be c->leb_size)
 *
- * This function returns %0 on success and a negative error code on failure.
+ * This function returns the scanned information on success and a negative error
+ * code on failure.
 */
 struct ubifs_scan_leb *ubifs_start_scan(const struct ubifs_info *c, int lnum,
                                        int offs, void *sbuf)
@@ -157,9 +158,10 @@ struct ubifs_scan_leb *ubifs_start_scan(const struct ubifs_info *c, int lnum,
                return ERR_PTR(err);
        }
-        if (err == -EBADMSG)
+        /*
-                sleb->ecc = 1;
+         * Note, we ignore integrity errors (EBASMSG) because all the nodes are
+         * protected by CRC checksums.
+         */
        return sleb;
 }
@@ -169,8 +171,6 @@ struct ubifs_scan_leb *ubifs_start_scan(const struct ubifs_info *c, int lnum,
 * @sleb: scanning information
 * @lnum: logical eraseblock number
 * @offs: offset to start at (usually zero)
- *
- * This function returns %0 on success and a negative error code on failure.
 */
 void ubifs_end_scan(const struct ubifs_info *c, struct ubifs_scan_leb *sleb,
                    int lnum, int offs)
@@ -257,7 +257,7 @@ void ubifs_scanned_corruption(const struct ubifs_info *c, int lnum, int offs,
 * @quiet: print no messages
 *
 * This function scans LEB number @lnum and returns complete information about
- * its contents. Returns the scaned information in case of success and,
+ * its contents. Returns the scanned information in case of success and,
 * %-EUCLEAN if the LEB neads recovery, and other negative error codes in case
 * of failure.
 *
diff --git a/fs/ubifs/super.c b/fs/ubifs/super.c
index 3904c8574ef9..106bf20629ce 100644
--- a/fs/ubifs/super.c
+++ b/fs/ubifs/super.c
@@ -75,7 +75,7 @@ static int validate_inode(struct ubifs_info *c, const struct inode *inode)
                return 1;
        }
-        if (ui->compr_type < 0 || ui->compr_type >= UBIFS_COMPR_TYPES_CNT) {
+        if (ui->compr_type >= UBIFS_COMPR_TYPES_CNT) {
                ubifs_err("unknown compression type %d", ui->compr_type);
                return 2;
        }
@@ -424,19 +424,19 @@ static int ubifs_show_options(struct seq_file *s, struct dentry *root)
        struct ubifs_info *c = root->d_sb->s_fs_info;
        if (c->mount_opts.unmount_mode == 2)
-                seq_printf(s, ",fast_unmount");
+                seq_puts(s, ",fast_unmount");
        else if (c->mount_opts.unmount_mode == 1)
-                seq_printf(s, ",norm_unmount");
+                seq_puts(s, ",norm_unmount");
        if (c->mount_opts.bulk_read == 2)
-                seq_printf(s, ",bulk_read");
+                seq_puts(s, ",bulk_read");
        else if (c->mount_opts.bulk_read == 1)
-                seq_printf(s, ",no_bulk_read");
+                seq_puts(s, ",no_bulk_read");
        if (c->mount_opts.chk_data_crc == 2)
-                seq_printf(s, ",chk_data_crc");
+                seq_puts(s, ",chk_data_crc");
        else if (c->mount_opts.chk_data_crc == 1)
-                seq_printf(s, ",no_chk_data_crc");
+                seq_puts(s, ",no_chk_data_crc");
        if (c->mount_opts.override_compr) {
                seq_printf(s, ",compr=%s",
@@ -796,8 +796,8 @@ static int alloc_wbufs(struct ubifs_info *c)
 {
        int i, err;
-        c->jheads = kzalloc(c->jhead_cnt * sizeof(struct ubifs_jhead),
+        c->jheads = kcalloc(c->jhead_cnt, sizeof(struct ubifs_jhead),
-                           GFP_KERNEL);
+                            GFP_KERNEL);
        if (!c->jheads)
                return -ENOMEM;
@@ -1963,7 +1963,6 @@ static struct ubifs_info *alloc_ubifs_info(struct ubi_volume_desc *ubi)
                mutex_init(&c->lp_mutex);
                mutex_init(&c->tnc_mutex);
                mutex_init(&c->log_mutex);
-                mutex_init(&c->mst_mutex);
                mutex_init(&c->umount_mutex);
                mutex_init(&c->bu_mutex);
                mutex_init(&c->write_reserve_mutex);
diff --git a/fs/ubifs/tnc.c b/fs/ubifs/tnc.c
index 8a40cf9c02d7..6793db0754f6 100644
--- a/fs/ubifs/tnc.c
+++ b/fs/ubifs/tnc.c
@@ -3294,7 +3294,6 @@ int dbg_check_inode_size(struct ubifs_info *c, const struct inode *inode,
                goto out_unlock;
        if (err) {
-                err = -EINVAL;
                key = &from_key;
                goto out_dump;
        }
diff --git a/fs/ubifs/tnc_commit.c b/fs/ubifs/tnc_commit.c
index 3600994f8411..7a205e046776 100644
--- a/fs/ubifs/tnc_commit.c
+++ b/fs/ubifs/tnc_commit.c
@@ -389,7 +389,6 @@ static int layout_in_gaps(struct ubifs_info *c, int cnt)
                                ubifs_dump_lprops(c);
                        }
                        /* Try to commit anyway */
-                        err = 0;
                        break;
                }
                p++;
diff --git a/fs/ubifs/ubifs.h b/fs/ubifs/ubifs.h
index c1f71fe17cc0..c4fe900c67ab 100644
--- a/fs/ubifs/ubifs.h
+++ b/fs/ubifs/ubifs.h
@@ -314,7 +314,6 @@ struct ubifs_scan_node {
 * @nodes_cnt: number of nodes scanned
 * @nodes: list of struct ubifs_scan_node
 * @endpt: end point (and therefore the start of empty space)
- * @ecc: read returned -EBADMSG
 * @buf: buffer containing entire LEB scanned
 */
 struct ubifs_scan_leb {
@@ -322,7 +321,6 @@ struct ubifs_scan_leb {
        int nodes_cnt;
        struct list_head nodes;
        int endpt;
-        int ecc;
        void *buf;
 };
@@ -1051,7 +1049,6 @@ struct ubifs_debug_info;
 *
 * @mst_node: master node
 * @mst_offs: offset of valid master node
- * @mst_mutex: protects the master node area, @mst_node, and @mst_offs
 *
 * @max_bu_buf_len: maximum bulk-read buffer length
 * @bu_mutex: protects the pre-allocated bulk-read buffer and @c->bu
@@ -1292,7 +1289,6 @@ struct ubifs_info {
        struct ubifs_mst_node *mst_node;
        int mst_offs;
-        struct mutex mst_mutex;
        int max_bu_buf_len;
        struct mutex bu_mutex;
diff --git a/fs/udf/file.c b/fs/udf/file.c
index d80738fdf424..86c6743ec1fe 100644
--- a/fs/udf/file.c
+++ b/fs/udf/file.c
@@ -27,7 +27,7 @@
 #include "udfdecl.h"
 #include <linux/fs.h>
-#include <asm/uaccess.h>
+#include <linux/uaccess.h>
 #include <linux/kernel.h>
 #include <linux/string.h> /* memset */
 #include <linux/capability.h>
@@ -100,24 +100,6 @@ static int udf_adinicb_write_begin(struct file *file,
        return 0;
 }
-static int udf_adinicb_write_end(struct file *file,
-                        struct address_space *mapping,
-                        loff_t pos, unsigned len, unsigned copied,
-                        struct page *page, void *fsdata)
-{
-        struct inode *inode = mapping->host;
-        unsigned offset = pos & (PAGE_CACHE_SIZE - 1);
-        char *kaddr;
-        struct udf_inode_info *iinfo = UDF_I(inode);
-        kaddr = kmap_atomic(page);
-        memcpy(iinfo->i_ext.i_data + iinfo->i_lenEAttr + offset,
-                kaddr + offset, copied);
-        kunmap_atomic(kaddr);
-        return simple_write_end(file, mapping, pos, len, copied, page, fsdata);
-}
 static ssize_t udf_adinicb_direct_IO(int rw, struct kiocb *iocb,
                                     struct iov_iter *iter,
                                     loff_t offset)
@@ -130,7 +112,7 @@ const struct address_space_operations udf_adinicb_aops = {
        .readpage       = udf_adinicb_readpage,
        .writepage      = udf_adinicb_writepage,
        .write_begin    = udf_adinicb_write_begin,
-        .write_end      = udf_adinicb_write_end,
+        .write_end      = simple_write_end,
        .direct_IO      = udf_adinicb_direct_IO,
 };
diff --git a/fs/udf/lowlevel.c b/fs/udf/lowlevel.c
index 6583fe9b0645..6ad5a453af97 100644
--- a/fs/udf/lowlevel.c
+++ b/fs/udf/lowlevel.c
@@ -21,7 +21,7 @@
 #include <linux/blkdev.h>
 #include <linux/cdrom.h>
-#include <asm/uaccess.h>
+#include <linux/uaccess.h>
 #include "udf_sb.h"
diff --git a/fs/udf/super.c b/fs/udf/super.c
index 3286db047a40..813da94d447b 100644
--- a/fs/udf/super.c
+++ b/fs/udf/super.c
@@ -63,7 +63,7 @@
 #include "udf_i.h"
 #include <linux/init.h>
-#include <asm/uaccess.h>
+#include <linux/uaccess.h>
 #define VDS_POS_PRIMARY_VOL_DESC        0
 #define VDS_POS_UNALLOC_SPACE_DESC      1
diff --git a/fs/udf/symlink.c b/fs/udf/symlink.c
index d7c6dbe4194b..6fb7945c1e6e 100644
--- a/fs/udf/symlink.c
+++ b/fs/udf/symlink.c
@@ -20,7 +20,7 @@
 */
 #include "udfdecl.h"
-#include <asm/uaccess.h>
+#include <linux/uaccess.h>
 #include <linux/errno.h>
 #include <linux/fs.h>
 #include <linux/time.h>
diff --git a/fs/udf/unicode.c b/fs/udf/unicode.c
index 44b815e57f94..afd470e588ff 100644
--- a/fs/udf/unicode.c
+++ b/fs/udf/unicode.c
@@ -412,7 +412,6 @@ static int udf_translate_to_linux(uint8_t *newName, uint8_t *udfName,
        int extIndex = 0, newExtIndex = 0, hasExt = 0;
        unsigned short valueCRC;
        uint8_t curr;
-        const uint8_t hexChar[] = "0123456789ABCDEF";
        if (udfName[0] == '.' &&
            (udfLen == 1 || (udfLen == 2 && udfName[1] == '.'))) {
@@ -477,10 +476,10 @@ static int udf_translate_to_linux(uint8_t *newName, uint8_t *udfName,
                        newIndex = 250;
                newName[newIndex++] = CRC_MARK;
                valueCRC = crc_itu_t(0, fidName, fidNameLen);
-                newName[newIndex++] = hexChar[(valueCRC & 0xf000) >> 12];
+                newName[newIndex++] = hex_asc_upper_hi(valueCRC >> 8);
-                newName[newIndex++] = hexChar[(valueCRC & 0x0f00) >> 8];
+                newName[newIndex++] = hex_asc_upper_lo(valueCRC >> 8);
-                newName[newIndex++] = hexChar[(valueCRC & 0x00f0) >> 4];
+                newName[newIndex++] = hex_asc_upper_hi(valueCRC);
-                newName[newIndex++] = hexChar[(valueCRC & 0x000f)];
+                newName[newIndex++] = hex_asc_upper_lo(valueCRC);
                if (hasExt) {
                        newName[newIndex++] = EXT_MARK;
diff --git a/fs/xfs/Kconfig b/fs/xfs/Kconfig
index 399e8cec6e60..5d47b4df61ea 100644
--- a/fs/xfs/Kconfig
+++ b/fs/xfs/Kconfig
@@ -1,6 +1,7 @@
 config XFS_FS
        tristate "XFS filesystem support"
        depends on BLOCK
+        depends on (64BIT || LBDAF)
        select EXPORTFS
        select LIBCRC32C
        help
diff --git a/fs/xfs/Makefile b/fs/xfs/Makefile
index c21f43506661..d61799949580 100644
--- a/fs/xfs/Makefile
+++ b/fs/xfs/Makefile
@@ -17,6 +17,7 @@
 #
 ccflags-y += -I$(src)                   # needed for trace events
+ccflags-y += -I$(src)/libxfs
 ccflags-$(CONFIG_XFS_DEBUG) += -g
@@ -25,6 +26,39 @@ obj-$(CONFIG_XFS_FS)		+= xfs.o
 # this one should be compiled first, as the tracing macros can easily blow up
 xfs-y                           += xfs_trace.o
+# build the libxfs code first
+xfs-y                           += $(addprefix libxfs/, \
+                                   xfs_alloc.o \
+                                   xfs_alloc_btree.o \
+                                   xfs_attr.o \
+                                   xfs_attr_leaf.o \
+                                   xfs_attr_remote.o \
+                                   xfs_bmap.o \
+                                   xfs_bmap_btree.o \
+                                   xfs_btree.o \
+                                   xfs_da_btree.o \
+                                   xfs_da_format.o \
+                                   xfs_dir2.o \
+                                   xfs_dir2_block.o \
+                                   xfs_dir2_data.o \
+                                   xfs_dir2_leaf.o \
+                                   xfs_dir2_node.o \
+                                   xfs_dir2_sf.o \
+                                   xfs_dquot_buf.o \
+                                   xfs_ialloc.o \
+                                   xfs_ialloc_btree.o \
+                                   xfs_inode_fork.o \
+                                   xfs_inode_buf.o \
+                                   xfs_log_rlimit.o \
+                                   xfs_sb.o \
+                                   xfs_symlink_remote.o \
+                                   xfs_trans_resv.o \
+                                   )
+# xfs_rtbitmap is shared with libxfs
+xfs-$(CONFIG_XFS_RT)            += $(addprefix libxfs/, \
+                                   xfs_rtbitmap.o \
+                                   )
 # highlevel code
 xfs-y                           += xfs_aops.o \
                                   xfs_attr_inactive.o \
@@ -45,53 +79,27 @@ xfs-y				+= xfs_aops.o \
                                   xfs_ioctl.o \
                                   xfs_iomap.o \
                                   xfs_iops.o \
+                                   xfs_inode.o \
                                   xfs_itable.o \
                                   xfs_message.o \
                                   xfs_mount.o \
                                   xfs_mru_cache.o \
                                   xfs_super.o \
                                   xfs_symlink.o \
+                                   xfs_sysfs.o \
                                   xfs_trans.o \
                                   xfs_xattr.o \
                                   kmem.o \
                                   uuid.o
-# code shared with libxfs
-xfs-y                           += xfs_alloc.o \
-                                   xfs_alloc_btree.o \
-                                   xfs_attr.o \
-                                   xfs_attr_leaf.o \
-                                   xfs_attr_remote.o \
-                                   xfs_bmap.o \
-                                   xfs_bmap_btree.o \
-                                   xfs_btree.o \
-                                   xfs_da_btree.o \
-                                   xfs_da_format.o \
-                                   xfs_dir2.o \
-                                   xfs_dir2_block.o \
-                                   xfs_dir2_data.o \
-                                   xfs_dir2_leaf.o \
-                                   xfs_dir2_node.o \
-                                   xfs_dir2_sf.o \
-                                   xfs_dquot_buf.o \
-                                   xfs_ialloc.o \
-                                   xfs_ialloc_btree.o \
-                                   xfs_icreate_item.o \
-                                   xfs_inode.o \
-                                   xfs_inode_fork.o \
-                                   xfs_inode_buf.o \
-                                   xfs_log_recover.o \
-                                   xfs_log_rlimit.o \
-                                   xfs_sb.o \
-                                   xfs_symlink_remote.o \
-                                   xfs_trans_resv.o
 # low-level transaction/log code
 xfs-y                           += xfs_log.o \
                                   xfs_log_cil.o \
                                   xfs_buf_item.o \
                                   xfs_extfree_item.o \
+                                   xfs_icreate_item.o \
                                   xfs_inode_item.o \
+                                   xfs_log_recover.o \
                                   xfs_trans_ail.o \
                                   xfs_trans_buf.o \
                                   xfs_trans_extfree.o \
@@ -107,8 +115,7 @@ xfs-$(CONFIG_XFS_QUOTA)		+= xfs_dquot.o \
                                   xfs_quotaops.o
 # xfs_rtbitmap is shared with libxfs
-xfs-$(CONFIG_XFS_RT)            += xfs_rtalloc.o \
+xfs-$(CONFIG_XFS_RT)            += xfs_rtalloc.o
-                                   xfs_rtbitmap.o
 xfs-$(CONFIG_XFS_POSIX_ACL)     += xfs_acl.o
 xfs-$(CONFIG_PROC_FS)           += xfs_stats.o
diff --git a/fs/xfs/xfs_ag.h b/fs/xfs/libxfs/xfs_ag.h
index 6e247a99f5db..6e247a99f5db 100644
--- a/fs/xfs/xfs_ag.h
+++ b/fs/xfs/libxfs/xfs_ag.h
diff --git a/fs/xfs/xfs_alloc.c b/fs/xfs/libxfs/xfs_alloc.c
index d43813267a80..4bffffe038a1 100644
--- a/fs/xfs/xfs_alloc.c
+++ b/fs/xfs/libxfs/xfs_alloc.c
@@ -483,9 +483,9 @@ xfs_agfl_read_verify(
                return;
        if (!xfs_buf_verify_cksum(bp, XFS_AGFL_CRC_OFF))
-                xfs_buf_ioerror(bp, EFSBADCRC);
+                xfs_buf_ioerror(bp, -EFSBADCRC);
        else if (!xfs_agfl_verify(bp))
-                xfs_buf_ioerror(bp, EFSCORRUPTED);
+                xfs_buf_ioerror(bp, -EFSCORRUPTED);
        if (bp->b_error)
                xfs_verifier_error(bp);
@@ -503,7 +503,7 @@ xfs_agfl_write_verify(
                return;
        if (!xfs_agfl_verify(bp)) {
-                xfs_buf_ioerror(bp, EFSCORRUPTED);
+                xfs_buf_ioerror(bp, -EFSCORRUPTED);
                xfs_verifier_error(bp);
                return;
        }
@@ -559,7 +559,7 @@ xfs_alloc_update_counters(
        xfs_trans_agblocks_delta(tp, len);
        if (unlikely(be32_to_cpu(agf->agf_freeblks) >
                     be32_to_cpu(agf->agf_length)))
-                return EFSCORRUPTED;
+                return -EFSCORRUPTED;
        xfs_alloc_log_agf(tp, agbp, XFS_AGF_FREEBLKS);
        return 0;
@@ -2234,11 +2234,11 @@ xfs_agf_read_verify(
        if (xfs_sb_version_hascrc(&mp->m_sb) &&
            !xfs_buf_verify_cksum(bp, XFS_AGF_CRC_OFF))
-                xfs_buf_ioerror(bp, EFSBADCRC);
+                xfs_buf_ioerror(bp, -EFSBADCRC);
        else if (XFS_TEST_ERROR(!xfs_agf_verify(mp, bp), mp,
                                XFS_ERRTAG_ALLOC_READ_AGF,
                                XFS_RANDOM_ALLOC_READ_AGF))
-                xfs_buf_ioerror(bp, EFSCORRUPTED);
+                xfs_buf_ioerror(bp, -EFSCORRUPTED);
        if (bp->b_error)
                xfs_verifier_error(bp);
@@ -2252,7 +2252,7 @@ xfs_agf_write_verify(
        struct xfs_buf_log_item *bip = bp->b_fspriv;
        if (!xfs_agf_verify(mp, bp)) {
-                xfs_buf_ioerror(bp, EFSCORRUPTED);
+                xfs_buf_ioerror(bp, -EFSCORRUPTED);
                xfs_verifier_error(bp);
                return;
        }
@@ -2601,11 +2601,11 @@ xfs_free_extent(
         */
        args.agno = XFS_FSB_TO_AGNO(args.mp, bno);
        if (args.agno >= args.mp->m_sb.sb_agcount)
-                return EFSCORRUPTED;
+                return -EFSCORRUPTED;
        args.agbno = XFS_FSB_TO_AGBNO(args.mp, bno);
        if (args.agbno >= args.mp->m_sb.sb_agblocks)
-                return EFSCORRUPTED;
+                return -EFSCORRUPTED;
        args.pag = xfs_perag_get(args.mp, args.agno);
        ASSERT(args.pag);
@@ -2617,7 +2617,7 @@ xfs_free_extent(
        /* validate the extent size is legal now we have the agf locked */
        if (args.agbno + len >
                        be32_to_cpu(XFS_BUF_TO_AGF(args.agbp)->agf_length)) {
-                error = EFSCORRUPTED;
+                error = -EFSCORRUPTED;
                goto error0;
        }
diff --git a/fs/xfs/xfs_alloc.h b/fs/xfs/libxfs/xfs_alloc.h
index feacb061bab7..feacb061bab7 100644
--- a/fs/xfs/xfs_alloc.h
+++ b/fs/xfs/libxfs/xfs_alloc.h
diff --git a/fs/xfs/xfs_alloc_btree.c b/fs/xfs/libxfs/xfs_alloc_btree.c
index 8358f1ded94d..e0e83e24d3ef 100644
--- a/fs/xfs/xfs_alloc_btree.c
+++ b/fs/xfs/libxfs/xfs_alloc_btree.c
@@ -355,9 +355,9 @@ xfs_allocbt_read_verify(
        struct xfs_buf  *bp)
 {
        if (!xfs_btree_sblock_verify_crc(bp))
-                xfs_buf_ioerror(bp, EFSBADCRC);
+                xfs_buf_ioerror(bp, -EFSBADCRC);
        else if (!xfs_allocbt_verify(bp))
-                xfs_buf_ioerror(bp, EFSCORRUPTED);
+                xfs_buf_ioerror(bp, -EFSCORRUPTED);
        if (bp->b_error) {
                trace_xfs_btree_corrupt(bp, _RET_IP_);
@@ -371,7 +371,7 @@ xfs_allocbt_write_verify(
 {
        if (!xfs_allocbt_verify(bp)) {
                trace_xfs_btree_corrupt(bp, _RET_IP_);
-                xfs_buf_ioerror(bp, EFSCORRUPTED);
+                xfs_buf_ioerror(bp, -EFSCORRUPTED);
                xfs_verifier_error(bp);
                return;
        }
diff --git a/fs/xfs/xfs_alloc_btree.h b/fs/xfs/libxfs/xfs_alloc_btree.h
index 45e189e7e81c..45e189e7e81c 100644
--- a/fs/xfs/xfs_alloc_btree.h
+++ b/fs/xfs/libxfs/xfs_alloc_btree.h
diff --git a/fs/xfs/xfs_attr.c b/fs/xfs/libxfs/xfs_attr.c
index bfe36fc2cdc2..353fb425faef 100644
--- a/fs/xfs/xfs_attr.c
+++ b/fs/xfs/libxfs/xfs_attr.c
@@ -85,7 +85,7 @@ xfs_attr_args_init(
 {
        if (!name)
-                return EINVAL;
+                return -EINVAL;
        memset(args, 0, sizeof(*args));
        args->geo = dp->i_mount->m_attr_geo;
@@ -95,7 +95,7 @@ xfs_attr_args_init(
        args->name = name;
        args->namelen = strlen((const char *)name);
        if (args->namelen >= MAXNAMELEN)
-                return EFAULT;          /* match IRIX behaviour */
+                return -EFAULT;         /* match IRIX behaviour */
        args->hashval = xfs_da_hashname(args->name, args->namelen);
        return 0;
@@ -131,10 +131,10 @@ xfs_attr_get(
        XFS_STATS_INC(xs_attr_get);
        if (XFS_FORCED_SHUTDOWN(ip->i_mount))
-                return EIO;
+                return -EIO;
        if (!xfs_inode_hasattr(ip))
-                return ENOATTR;
+                return -ENOATTR;
        error = xfs_attr_args_init(&args, ip, name, flags);
        if (error)
@@ -145,7 +145,7 @@ xfs_attr_get(
        lock_mode = xfs_ilock_attr_map_shared(ip);
        if (!xfs_inode_hasattr(ip))
-                error = ENOATTR;
+                error = -ENOATTR;
        else if (ip->i_d.di_aformat == XFS_DINODE_FMT_LOCAL)
                error = xfs_attr_shortform_getvalue(&args);
        else if (xfs_bmap_one_block(ip, XFS_ATTR_FORK))
@@ -155,7 +155,7 @@ xfs_attr_get(
        xfs_iunlock(ip, lock_mode);
        *valuelenp = args.valuelen;
-        return error == EEXIST ? 0 : error;
+        return error == -EEXIST ? 0 : error;
 }
 /*
@@ -213,7 +213,7 @@ xfs_attr_set(
        XFS_STATS_INC(xs_attr_set);
        if (XFS_FORCED_SHUTDOWN(dp->i_mount))
-                return EIO;
+                return -EIO;
        error = xfs_attr_args_init(&args, dp, name, flags);
        if (error)
@@ -304,7 +304,7 @@ xfs_attr_set(
                 * the inode.
                 */
                error = xfs_attr_shortform_addname(&args);
-                if (error != ENOSPC) {
+                if (error != -ENOSPC) {
                        /*
                         * Commit the shortform mods, and we're done.
                         * NOTE: this is also the error path (EEXIST, etc).
@@ -419,10 +419,10 @@ xfs_attr_remove(
        XFS_STATS_INC(xs_attr_remove);
        if (XFS_FORCED_SHUTDOWN(dp->i_mount))
-                return EIO;
+                return -EIO;
        if (!xfs_inode_hasattr(dp))
-                return ENOATTR;
+                return -ENOATTR;
        error = xfs_attr_args_init(&args, dp, name, flags);
        if (error)
@@ -477,7 +477,7 @@ xfs_attr_remove(
        xfs_trans_ijoin(args.trans, dp, 0);
        if (!xfs_inode_hasattr(dp)) {
-                error = XFS_ERROR(ENOATTR);
+                error = -ENOATTR;
        } else if (dp->i_d.di_aformat == XFS_DINODE_FMT_LOCAL) {
                ASSERT(dp->i_afp->if_flags & XFS_IFINLINE);
                error = xfs_attr_shortform_remove(&args);
@@ -534,28 +534,28 @@ xfs_attr_shortform_addname(xfs_da_args_t *args)
        trace_xfs_attr_sf_addname(args);
        retval = xfs_attr_shortform_lookup(args);
-        if ((args->flags & ATTR_REPLACE) && (retval == ENOATTR)) {
+        if ((args->flags & ATTR_REPLACE) && (retval == -ENOATTR)) {
-                return(retval);
+                return retval;
-        } else if (retval == EEXIST) {
+        } else if (retval == -EEXIST) {
                if (args->flags & ATTR_CREATE)
-                        return(retval);
+                        return retval;
                retval = xfs_attr_shortform_remove(args);
                ASSERT(retval == 0);
        }
        if (args->namelen >= XFS_ATTR_SF_ENTSIZE_MAX ||
            args->valuelen >= XFS_ATTR_SF_ENTSIZE_MAX)
-                return(XFS_ERROR(ENOSPC));
+                return -ENOSPC;
        newsize = XFS_ATTR_SF_TOTSIZE(args->dp);
        newsize += XFS_ATTR_SF_ENTSIZE_BYNAME(args->namelen, args->valuelen);
        forkoff = xfs_attr_shortform_bytesfit(args->dp, newsize);
        if (!forkoff)
-                return(XFS_ERROR(ENOSPC));
+                return -ENOSPC;
        xfs_attr_shortform_add(args, forkoff);
-        return(0);
+        return 0;
 }
@@ -592,10 +592,10 @@ xfs_attr_leaf_addname(xfs_da_args_t *args)
         * the given flags produce an error or call for an atomic rename.
         */
        retval = xfs_attr3_leaf_lookup_int(bp, args);
-        if ((args->flags & ATTR_REPLACE) && (retval == ENOATTR)) {
+        if ((args->flags & ATTR_REPLACE) && (retval == -ENOATTR)) {
                xfs_trans_brelse(args->trans, bp);
                return retval;
-        } else if (retval == EEXIST) {
+        } else if (retval == -EEXIST) {
                if (args->flags & ATTR_CREATE) {        /* pure create op */
                        xfs_trans_brelse(args->trans, bp);
                        return retval;
@@ -626,7 +626,7 @@ xfs_attr_leaf_addname(xfs_da_args_t *args)
         * if required.
         */
        retval = xfs_attr3_leaf_add(bp, args);
-        if (retval == ENOSPC) {
+        if (retval == -ENOSPC) {
                /*
                 * Promote the attribute list to the Btree format, then
                 * Commit that transaction so that the node_addname() call
@@ -642,7 +642,7 @@ xfs_attr_leaf_addname(xfs_da_args_t *args)
                        ASSERT(committed);
                        args->trans = NULL;
                        xfs_bmap_cancel(args->flist);
-                        return(error);
+                        return error;
                }
                /*
@@ -658,13 +658,13 @@ xfs_attr_leaf_addname(xfs_da_args_t *args)
                 */
                error = xfs_trans_roll(&args->trans, dp);
                if (error)
-                        return (error);
+                        return error;
                /*
                 * Fob the whole rest of the problem off on the Btree code.
                 */
                error = xfs_attr_node_addname(args);
-                return(error);
+                return error;
        }
        /*
@@ -673,7 +673,7 @@ xfs_attr_leaf_addname(xfs_da_args_t *args)
         */
        error = xfs_trans_roll(&args->trans, dp);
        if (error)
-                return (error);
+                return error;
        /*
         * If there was an out-of-line value, allocate the blocks we
@@ -684,7 +684,7 @@ xfs_attr_leaf_addname(xfs_da_args_t *args)
        if (args->rmtblkno > 0) {
                error = xfs_attr_rmtval_set(args);
                if (error)
-                        return(error);
+                        return error;
        }
        /*
@@ -700,7 +700,7 @@ xfs_attr_leaf_addname(xfs_da_args_t *args)
                 */
                error = xfs_attr3_leaf_flipflags(args);
                if (error)
-                        return(error);
+                        return error;
                /*
                 * Dismantle the "old" attribute/value pair by removing
@@ -714,7 +714,7 @@ xfs_attr_leaf_addname(xfs_da_args_t *args)
                if (args->rmtblkno) {
                        error = xfs_attr_rmtval_remove(args);
                        if (error)
-                                return(error);
+                                return error;
                }
                /*
@@ -744,7 +744,7 @@ xfs_attr_leaf_addname(xfs_da_args_t *args)
                                ASSERT(committed);
                                args->trans = NULL;
                                xfs_bmap_cancel(args->flist);
-                                return(error);
+                                return error;
                        }
                        /*
@@ -795,7 +795,7 @@ xfs_attr_leaf_removename(xfs_da_args_t *args)
                return error;
        error = xfs_attr3_leaf_lookup_int(bp, args);
-        if (error == ENOATTR) {
+        if (error == -ENOATTR) {
                xfs_trans_brelse(args->trans, bp);
                return error;
        }
@@ -850,7 +850,7 @@ xfs_attr_leaf_get(xfs_da_args_t *args)
                return error;
        error = xfs_attr3_leaf_lookup_int(bp, args);
-        if (error != EEXIST)  {
+        if (error != -EEXIST)  {
                xfs_trans_brelse(args->trans, bp);
                return error;
        }
@@ -906,9 +906,9 @@ restart:
                goto out;
        blk = &state->path.blk[ state->path.active-1 ];
        ASSERT(blk->magic == XFS_ATTR_LEAF_MAGIC);
-        if ((args->flags & ATTR_REPLACE) && (retval == ENOATTR)) {
+        if ((args->flags & ATTR_REPLACE) && (retval == -ENOATTR)) {
                goto out;
-        } else if (retval == EEXIST) {
+        } else if (retval == -EEXIST) {
                if (args->flags & ATTR_CREATE)
                        goto out;
@@ -933,7 +933,7 @@ restart:
        }
        retval = xfs_attr3_leaf_add(blk->bp, state->args);
-        if (retval == ENOSPC) {
+        if (retval == -ENOSPC) {
                if (state->path.active == 1) {
                        /*
                         * Its really a single leaf node, but it had
@@ -1031,7 +1031,7 @@ restart:
        if (args->rmtblkno > 0) {
                error = xfs_attr_rmtval_set(args);
                if (error)
-                        return(error);
+                        return error;
        }
        /*
@@ -1061,7 +1061,7 @@ restart:
                if (args->rmtblkno) {
                        error = xfs_attr_rmtval_remove(args);
                        if (error)
-                                return(error);
+                                return error;
                }
                /*
@@ -1134,8 +1134,8 @@ out:
        if (state)
                xfs_da_state_free(state);
        if (error)
-                return(error);
+                return error;
-        return(retval);
+        return retval;
 }
 /*
@@ -1168,7 +1168,7 @@ xfs_attr_node_removename(xfs_da_args_t *args)
         * Search to see if name exists, and get back a pointer to it.
         */
        error = xfs_da3_node_lookup_int(state, &retval);
-        if (error || (retval != EEXIST)) {
+        if (error || (retval != -EEXIST)) {
                if (error == 0)
                        error = retval;
                goto out;
@@ -1297,7 +1297,7 @@ xfs_attr_node_removename(xfs_da_args_t *args)
 out:
        xfs_da_state_free(state);
-        return(error);
+        return error;
 }
 /*
@@ -1345,7 +1345,7 @@ xfs_attr_fillstate(xfs_da_state_t *state)
                }
        }
-        return(0);
+        return 0;
 }
 /*
@@ -1376,7 +1376,7 @@ xfs_attr_refillstate(xfs_da_state_t *state)
                                                blk->blkno, blk->disk_blkno,
                                                &blk->bp, XFS_ATTR_FORK);
                        if (error)
-                                return(error);
+                                return error;
                } else {
                        blk->bp = NULL;
                }
@@ -1395,13 +1395,13 @@ xfs_attr_refillstate(xfs_da_state_t *state)
                                                blk->blkno, blk->disk_blkno,
                                                &blk->bp, XFS_ATTR_FORK);
                        if (error)
-                                return(error);
+                                return error;
                } else {
                        blk->bp = NULL;
                }
        }
-        return(0);
+        return 0;
 }
 /*
@@ -1431,7 +1431,7 @@ xfs_attr_node_get(xfs_da_args_t *args)
        error = xfs_da3_node_lookup_int(state, &retval);
        if (error) {
                retval = error;
-        } else if (retval == EEXIST) {
+        } else if (retval == -EEXIST) {
                blk = &state->path.blk[ state->path.active-1 ];
                ASSERT(blk->bp != NULL);
                ASSERT(blk->magic == XFS_ATTR_LEAF_MAGIC);
@@ -1455,5 +1455,5 @@ xfs_attr_node_get(xfs_da_args_t *args)
        }
        xfs_da_state_free(state);
-        return(retval);
+        return retval;
 }
diff --git a/fs/xfs/xfs_attr_leaf.c b/fs/xfs/libxfs/xfs_attr_leaf.c
index 28712d29e43c..b1f73dbbf3d8 100644
--- a/fs/xfs/xfs_attr_leaf.c
+++ b/fs/xfs/libxfs/xfs_attr_leaf.c
@@ -214,7 +214,7 @@ xfs_attr3_leaf_write_verify(
        struct xfs_attr3_leaf_hdr *hdr3 = bp->b_addr;
        if (!xfs_attr3_leaf_verify(bp)) {
-                xfs_buf_ioerror(bp, EFSCORRUPTED);
+                xfs_buf_ioerror(bp, -EFSCORRUPTED);
                xfs_verifier_error(bp);
                return;
        }
@@ -242,9 +242,9 @@ xfs_attr3_leaf_read_verify(
        if (xfs_sb_version_hascrc(&mp->m_sb) &&
             !xfs_buf_verify_cksum(bp, XFS_ATTR3_LEAF_CRC_OFF))
-                xfs_buf_ioerror(bp, EFSBADCRC);
+                xfs_buf_ioerror(bp, -EFSBADCRC);
        else if (!xfs_attr3_leaf_verify(bp))
-                xfs_buf_ioerror(bp, EFSCORRUPTED);
+                xfs_buf_ioerror(bp, -EFSCORRUPTED);
        if (bp->b_error)
                xfs_verifier_error(bp);
@@ -547,7 +547,7 @@ xfs_attr_shortform_remove(xfs_da_args_t *args)
                break;
        }
        if (i == end)
-                return(XFS_ERROR(ENOATTR));
+                return -ENOATTR;
        /*
         * Fix up the attribute fork data, covering the hole
@@ -582,7 +582,7 @@ xfs_attr_shortform_remove(xfs_da_args_t *args)
        xfs_sbversion_add_attr2(mp, args->trans);
-        return(0);
+        return 0;
 }
 /*
@@ -611,9 +611,9 @@ xfs_attr_shortform_lookup(xfs_da_args_t *args)
                        continue;
                if (!xfs_attr_namesp_match(args->flags, sfe->flags))
                        continue;
-                return(XFS_ERROR(EEXIST));
+                return -EEXIST;
        }
-        return(XFS_ERROR(ENOATTR));
+        return -ENOATTR;
 }
 /*
@@ -640,18 +640,18 @@ xfs_attr_shortform_getvalue(xfs_da_args_t *args)
                        continue;
                if (args->flags & ATTR_KERNOVAL) {
                        args->valuelen = sfe->valuelen;
-                        return(XFS_ERROR(EEXIST));
+                        return -EEXIST;
                }
                if (args->valuelen < sfe->valuelen) {
                        args->valuelen = sfe->valuelen;
-                        return(XFS_ERROR(ERANGE));
+                        return -ERANGE;
                }
                args->valuelen = sfe->valuelen;
                memcpy(args->value, &sfe->nameval[args->namelen],
                                                    args->valuelen);
-                return(XFS_ERROR(EEXIST));
+                return -EEXIST;
        }
-        return(XFS_ERROR(ENOATTR));
+        return -ENOATTR;
 }
 /*
@@ -691,7 +691,7 @@ xfs_attr_shortform_to_leaf(xfs_da_args_t *args)
                 * If we hit an IO error middle of the transaction inside
                 * grow_inode(), we may have inconsistent data. Bail out.
                 */
-                if (error == EIO)
+                if (error == -EIO)
                        goto out;
                xfs_idata_realloc(dp, size, XFS_ATTR_FORK);     /* try to put */
                memcpy(ifp->if_u1.if_data, tmpbuffer, size);    /* it back */
@@ -730,9 +730,9 @@ xfs_attr_shortform_to_leaf(xfs_da_args_t *args)
                                                sfe->namelen);
                nargs.flags = XFS_ATTR_NSP_ONDISK_TO_ARGS(sfe->flags);
                error = xfs_attr3_leaf_lookup_int(bp, &nargs); /* set a->index */
-                ASSERT(error == ENOATTR);
+                ASSERT(error == -ENOATTR);
                error = xfs_attr3_leaf_add(bp, &nargs);
-                ASSERT(error != ENOSPC);
+                ASSERT(error != -ENOSPC);
                if (error)
                        goto out;
                sfe = XFS_ATTR_SF_NEXTENTRY(sfe);
@@ -741,7 +741,7 @@ xfs_attr_shortform_to_leaf(xfs_da_args_t *args)
 out:
        kmem_free(tmpbuffer);
-        return(error);
+        return error;
 }
 /*
@@ -769,12 +769,12 @@ xfs_attr_shortform_allfit(
                if (entry->flags & XFS_ATTR_INCOMPLETE)
                        continue;               /* don't copy partial entries */
                if (!(entry->flags & XFS_ATTR_LOCAL))
-                        return(0);
+                        return 0;
                name_loc = xfs_attr3_leaf_name_local(leaf, i);
                if (name_loc->namelen >= XFS_ATTR_SF_ENTSIZE_MAX)
-                        return(0);
+                        return 0;
                if (be16_to_cpu(name_loc->valuelen) >= XFS_ATTR_SF_ENTSIZE_MAX)
-                        return(0);
+                        return 0;
                bytes += sizeof(struct xfs_attr_sf_entry) - 1
                                + name_loc->namelen
                                + be16_to_cpu(name_loc->valuelen);
@@ -809,7 +809,7 @@ xfs_attr3_leaf_to_shortform(
        tmpbuffer = kmem_alloc(args->geo->blksize, KM_SLEEP);
        if (!tmpbuffer)
-                return ENOMEM;
+                return -ENOMEM;
        memcpy(tmpbuffer, bp->b_addr, args->geo->blksize);
@@ -1017,10 +1017,10 @@ xfs_attr3_leaf_split(
        ASSERT(oldblk->magic == XFS_ATTR_LEAF_MAGIC);
        error = xfs_da_grow_inode(state->args, &blkno);
        if (error)
-                return(error);
+                return error;
        error = xfs_attr3_leaf_create(state->args, blkno, &newblk->bp);
        if (error)
-                return(error);
+                return error;
        newblk->blkno = blkno;
        newblk->magic = XFS_ATTR_LEAF_MAGIC;
@@ -1031,7 +1031,7 @@ xfs_attr3_leaf_split(
        xfs_attr3_leaf_rebalance(state, oldblk, newblk);
        error = xfs_da3_blk_link(state, oldblk, newblk);
        if (error)
-                return(error);
+                return error;
        /*
         * Save info on "old" attribute for "atomic rename" ops, leaf_add()
@@ -1053,7 +1053,7 @@ xfs_attr3_leaf_split(
         */
        oldblk->hashval = xfs_attr_leaf_lasthash(oldblk->bp, NULL);
        newblk->hashval = xfs_attr_leaf_lasthash(newblk->bp, NULL);
-        return(error);
+        return error;
 }
 /*
@@ -1108,7 +1108,7 @@ xfs_attr3_leaf_add(
         * no good and we should just give up.
         */
        if (!ichdr.holes && sum < entsize)
-                return XFS_ERROR(ENOSPC);
+                return -ENOSPC;
        /*
         * Compact the entries to coalesce free space.
@@ -1121,7 +1121,7 @@ xfs_attr3_leaf_add(
         * free region, in freemap[0].  If it is not big enough, give up.
         */
        if (ichdr.freemap[0].size < (entsize + sizeof(xfs_attr_leaf_entry_t))) {
-                tmp = ENOSPC;
+                tmp = -ENOSPC;
                goto out_log_hdr;
        }
@@ -1692,7 +1692,7 @@ xfs_attr3_leaf_toosmall(
                ichdr.usedbytes;
        if (bytes > (state->args->geo->blksize >> 1)) {
                *action = 0;    /* blk over 50%, don't try to join */
-                return(0);
+                return 0;
        }
        /*
@@ -1711,7 +1711,7 @@ xfs_attr3_leaf_toosmall(
                error = xfs_da3_path_shift(state, &state->altpath, forward,
                                                 0, &retval);
                if (error)
-                        return(error);
+                        return error;
                if (retval) {
                        *action = 0;
                } else {
@@ -1740,7 +1740,7 @@ xfs_attr3_leaf_toosmall(
                error = xfs_attr3_leaf_read(state->args->trans, state->args->dp,
                                        blkno, -1, &bp);
                if (error)
-                        return(error);
+                        return error;
                xfs_attr3_leaf_hdr_from_disk(&ichdr2, bp->b_addr);
@@ -1757,7 +1757,7 @@ xfs_attr3_leaf_toosmall(
        }
        if (i >= 2) {
                *action = 0;
-                return(0);
+                return 0;
        }
        /*
@@ -1773,13 +1773,13 @@ xfs_attr3_leaf_toosmall(
                                                 0, &retval);
        }
        if (error)
-                return(error);
+                return error;
        if (retval) {
                *action = 0;
        } else {
                *action = 1;
        }
-        return(0);
+        return 0;
 }
 /*
@@ -2123,7 +2123,7 @@ xfs_attr3_leaf_lookup_int(
        }
        if (probe == ichdr.count || be32_to_cpu(entry->hashval) != hashval) {
                args->index = probe;
-                return XFS_ERROR(ENOATTR);
+                return -ENOATTR;
        }
        /*
@@ -2152,7 +2152,7 @@ xfs_attr3_leaf_lookup_int(
                        if (!xfs_attr_namesp_match(args->flags, entry->flags))
                                continue;
                        args->index = probe;
-                        return XFS_ERROR(EEXIST);
+                        return -EEXIST;
                } else {
                        name_rmt = xfs_attr3_leaf_name_remote(leaf, probe);
                        if (name_rmt->namelen != args->namelen)
@@ -2168,11 +2168,11 @@ xfs_attr3_leaf_lookup_int(
                        args->rmtblkcnt = xfs_attr3_rmt_blocks(
                                                        args->dp->i_mount,
                                                        args->rmtvaluelen);
-                        return XFS_ERROR(EEXIST);
+                        return -EEXIST;
                }
        }
        args->index = probe;
-        return XFS_ERROR(ENOATTR);
+        return -ENOATTR;
 }
 /*
@@ -2208,7 +2208,7 @@ xfs_attr3_leaf_getvalue(
                }
                if (args->valuelen < valuelen) {
                        args->valuelen = valuelen;
-                        return XFS_ERROR(ERANGE);
+                        return -ERANGE;
                }
                args->valuelen = valuelen;
                memcpy(args->value, &name_loc->nameval[args->namelen], valuelen);
@@ -2226,7 +2226,7 @@ xfs_attr3_leaf_getvalue(
                }
                if (args->valuelen < args->rmtvaluelen) {
                        args->valuelen = args->rmtvaluelen;
-                        return XFS_ERROR(ERANGE);
+                        return -ERANGE;
                }
                args->valuelen = args->rmtvaluelen;
        }
@@ -2481,7 +2481,7 @@ xfs_attr3_leaf_clearflag(
         */
        error = xfs_attr3_leaf_read(args->trans, args->dp, args->blkno, -1, &bp);
        if (error)
-                return(error);
+                return error;
        leaf = bp->b_addr;
        entry = &xfs_attr3_leaf_entryp(leaf)[args->index];
@@ -2548,7 +2548,7 @@ xfs_attr3_leaf_setflag(
         */
        error = xfs_attr3_leaf_read(args->trans, args->dp, args->blkno, -1, &bp);
        if (error)
-                return(error);
+                return error;
        leaf = bp->b_addr;
 #ifdef DEBUG
diff --git a/fs/xfs/xfs_attr_leaf.h b/fs/xfs/libxfs/xfs_attr_leaf.h
index e2929da7c3ba..e2929da7c3ba 100644
--- a/fs/xfs/xfs_attr_leaf.h
+++ b/fs/xfs/libxfs/xfs_attr_leaf.h
diff --git a/fs/xfs/xfs_attr_remote.c b/fs/xfs/libxfs/xfs_attr_remote.c
index b5adfecbb8ee..7510ab8058a4 100644
--- a/fs/xfs/xfs_attr_remote.c
+++ b/fs/xfs/libxfs/xfs_attr_remote.c
@@ -138,11 +138,11 @@ xfs_attr3_rmt_read_verify(
        while (len > 0) {
                if (!xfs_verify_cksum(ptr, blksize, XFS_ATTR3_RMT_CRC_OFF)) {
-                        xfs_buf_ioerror(bp, EFSBADCRC);
+                        xfs_buf_ioerror(bp, -EFSBADCRC);
                        break;
                }
                if (!xfs_attr3_rmt_verify(mp, ptr, blksize, bno)) {
-                        xfs_buf_ioerror(bp, EFSCORRUPTED);
+                        xfs_buf_ioerror(bp, -EFSCORRUPTED);
                        break;
                }
                len -= blksize;
@@ -178,7 +178,7 @@ xfs_attr3_rmt_write_verify(
        while (len > 0) {
                if (!xfs_attr3_rmt_verify(mp, ptr, blksize, bno)) {
-                        xfs_buf_ioerror(bp, EFSCORRUPTED);
+                        xfs_buf_ioerror(bp, -EFSCORRUPTED);
                        xfs_verifier_error(bp);
                        return;
                }
@@ -257,7 +257,7 @@ xfs_attr_rmtval_copyout(
                                xfs_alert(mp,
 "remote attribute header mismatch bno/off/len/owner (0x%llx/0x%x/Ox%x/0x%llx)",
                                        bno, *offset, byte_cnt, ino);
-                                return EFSCORRUPTED;
+                                return -EFSCORRUPTED;
                        }
                        hdr_size = sizeof(struct xfs_attr3_rmt_hdr);
                }
@@ -452,7 +452,7 @@ xfs_attr_rmtval_set(
                        ASSERT(committed);
                        args->trans = NULL;
                        xfs_bmap_cancel(args->flist);
-                        return(error);
+                        return error;
                }
                /*
@@ -473,7 +473,7 @@ xfs_attr_rmtval_set(
                 */
                error = xfs_trans_roll(&args->trans, dp);
                if (error)
-                        return (error);
+                        return error;
        }
        /*
@@ -498,7 +498,7 @@ xfs_attr_rmtval_set(
                                       blkcnt, &map, &nmap,
                                       XFS_BMAPI_ATTRFORK);
                if (error)
-                        return(error);
+                        return error;
                ASSERT(nmap == 1);
                ASSERT((map.br_startblock != DELAYSTARTBLOCK) &&
                       (map.br_startblock != HOLESTARTBLOCK));
@@ -508,7 +508,7 @@ xfs_attr_rmtval_set(
                bp = xfs_buf_get(mp->m_ddev_targp, dblkno, dblkcnt, 0);
                if (!bp)
-                        return ENOMEM;
+                        return -ENOMEM;
                bp->b_ops = &xfs_attr3_rmt_buf_ops;
                xfs_attr_rmtval_copyin(mp, bp, args->dp->i_ino, &offset,
@@ -563,7 +563,7 @@ xfs_attr_rmtval_remove(
                error = xfs_bmapi_read(args->dp, (xfs_fileoff_t)lblkno,
                                       blkcnt, &map, &nmap, XFS_BMAPI_ATTRFORK);
                if (error)
-                        return(error);
+                        return error;
                ASSERT(nmap == 1);
                ASSERT((map.br_startblock != DELAYSTARTBLOCK) &&
                       (map.br_startblock != HOLESTARTBLOCK));
@@ -622,7 +622,7 @@ xfs_attr_rmtval_remove(
                 */
                error = xfs_trans_roll(&args->trans, args->dp);
                if (error)
-                        return (error);
+                        return error;
        }
-        return(0);
+        return 0;
 }
diff --git a/fs/xfs/xfs_attr_remote.h b/fs/xfs/libxfs/xfs_attr_remote.h
index 5a9acfa156d7..5a9acfa156d7 100644
--- a/fs/xfs/xfs_attr_remote.h
+++ b/fs/xfs/libxfs/xfs_attr_remote.h
diff --git a/fs/xfs/xfs_attr_sf.h b/fs/xfs/libxfs/xfs_attr_sf.h
index 919756e3ba53..919756e3ba53 100644
--- a/fs/xfs/xfs_attr_sf.h
+++ b/fs/xfs/libxfs/xfs_attr_sf.h
diff --git a/fs/xfs/xfs_bit.h b/fs/xfs/libxfs/xfs_bit.h
index e1649c0d3e02..e1649c0d3e02 100644
--- a/fs/xfs/xfs_bit.h
+++ b/fs/xfs/libxfs/xfs_bit.h
diff --git a/fs/xfs/xfs_bmap.c b/fs/xfs/libxfs/xfs_bmap.c
index 75c3fe5f3d9d..de2d26d32844 100644
--- a/fs/xfs/xfs_bmap.c
+++ b/fs/xfs/libxfs/xfs_bmap.c
@@ -392,7 +392,7 @@ xfs_bmap_check_leaf_extents(
        pp = XFS_BMAP_BROOT_PTR_ADDR(mp, block, 1, ifp->if_broot_bytes);
        bno = be64_to_cpu(*pp);
-        ASSERT(bno != NULLDFSBNO);
+        ASSERT(bno != NULLFSBLOCK);
        ASSERT(XFS_FSB_TO_AGNO(mp, bno) < mp->m_sb.sb_agcount);
        ASSERT(XFS_FSB_TO_AGBNO(mp, bno) < mp->m_sb.sb_agblocks);
@@ -1033,7 +1033,7 @@ xfs_bmap_add_attrfork_btree(
                        goto error0;
                if (stat == 0) {
                        xfs_btree_del_cursor(cur, XFS_BTREE_NOERROR);
-                        return XFS_ERROR(ENOSPC);
+                        return -ENOSPC;
                }
                *firstblock = cur->bc_private.b.firstblock;
                cur->bc_private.b.allocated = 0;
@@ -1115,7 +1115,7 @@ xfs_bmap_add_attrfork_local(
        /* should only be called for types that support local format data */
        ASSERT(0);
-        return EFSCORRUPTED;
+        return -EFSCORRUPTED;
 }
 /*
@@ -1192,7 +1192,7 @@ xfs_bmap_add_attrfork(
                break;
        default:
                ASSERT(0);
-                error = XFS_ERROR(EINVAL);
+                error = -EINVAL;
                goto trans_cancel;
        }
@@ -1299,7 +1299,7 @@ xfs_bmap_read_extents(
        ASSERT(level > 0);
        pp = XFS_BMAP_BROOT_PTR_ADDR(mp, block, 1, ifp->if_broot_bytes);
        bno = be64_to_cpu(*pp);
-        ASSERT(bno != NULLDFSBNO);
+        ASSERT(bno != NULLFSBLOCK);
        ASSERT(XFS_FSB_TO_AGNO(mp, bno) < mp->m_sb.sb_agcount);
        ASSERT(XFS_FSB_TO_AGBNO(mp, bno) < mp->m_sb.sb_agblocks);
        /*
@@ -1399,7 +1399,7 @@ xfs_bmap_read_extents(
        return 0;
 error0:
        xfs_trans_brelse(tp, bp);
-        return XFS_ERROR(EFSCORRUPTED);
+        return -EFSCORRUPTED;
 }
@@ -1429,11 +1429,7 @@ xfs_bmap_search_multi_extents(
        gotp->br_startoff = 0xffa5a5a5a5a5a5a5LL;
        gotp->br_blockcount = 0xa55a5a5a5a5a5a5aLL;
        gotp->br_state = XFS_EXT_INVALID;
-#if XFS_BIG_BLKNOS
        gotp->br_startblock = 0xffffa5a5a5a5a5a5LL;
-#else
-        gotp->br_startblock = 0xffffa5a5;
-#endif
        prevp->br_startoff = NULLFILEOFF;
        ep = xfs_iext_bno_to_ext(ifp, bno, &lastx);
@@ -1576,7 +1572,7 @@ xfs_bmap_last_before(
        if (XFS_IFORK_FORMAT(ip, whichfork) != XFS_DINODE_FMT_BTREE &&
            XFS_IFORK_FORMAT(ip, whichfork) != XFS_DINODE_FMT_EXTENTS &&
            XFS_IFORK_FORMAT(ip, whichfork) != XFS_DINODE_FMT_LOCAL)
-               return XFS_ERROR(EIO);
+               return -EIO;
        if (XFS_IFORK_FORMAT(ip, whichfork) == XFS_DINODE_FMT_LOCAL) {
                *last_block = 0;
                return 0;
@@ -1690,7 +1686,7 @@ xfs_bmap_last_offset(
        if (XFS_IFORK_FORMAT(ip, whichfork) != XFS_DINODE_FMT_BTREE &&
            XFS_IFORK_FORMAT(ip, whichfork) != XFS_DINODE_FMT_EXTENTS)
-               return XFS_ERROR(EIO);
+               return -EIO;
        error = xfs_bmap_last_extent(NULL, ip, whichfork, &rec, &is_empty);
        if (error || is_empty)
@@ -3323,7 +3319,7 @@ xfs_bmap_extsize_align(
                if (orig_off < align_off ||
                    orig_end > align_off + align_alen ||
                    align_alen - temp < orig_alen)
-                        return XFS_ERROR(EINVAL);
+                        return -EINVAL;
                /*
                 * Try to fix it by moving the start up.
                 */
@@ -3348,7 +3344,7 @@ xfs_bmap_extsize_align(
                 * Result doesn't cover the request, fail it.
                 */
                if (orig_off < align_off || orig_end > align_off + align_alen)
-                        return XFS_ERROR(EINVAL);
+                        return -EINVAL;
        } else {
                ASSERT(orig_off >= align_off);
                ASSERT(orig_end <= align_off + align_alen);
@@ -4051,11 +4047,11 @@ xfs_bmapi_read(
             XFS_IFORK_FORMAT(ip, whichfork) != XFS_DINODE_FMT_BTREE),
             mp, XFS_ERRTAG_BMAPIFORMAT, XFS_RANDOM_BMAPIFORMAT))) {
                XFS_ERROR_REPORT("xfs_bmapi_read", XFS_ERRLEVEL_LOW, mp);
-                return XFS_ERROR(EFSCORRUPTED);
+                return -EFSCORRUPTED;
        }
        if (XFS_FORCED_SHUTDOWN(mp))
-                return XFS_ERROR(EIO);
+                return -EIO;
        XFS_STATS_INC(xs_blk_mapr);
@@ -4246,11 +4242,11 @@ xfs_bmapi_delay(
             XFS_IFORK_FORMAT(ip, XFS_DATA_FORK) != XFS_DINODE_FMT_BTREE),
             mp, XFS_ERRTAG_BMAPIFORMAT, XFS_RANDOM_BMAPIFORMAT))) {
                XFS_ERROR_REPORT("xfs_bmapi_delay", XFS_ERRLEVEL_LOW, mp);
-                return XFS_ERROR(EFSCORRUPTED);
+                return -EFSCORRUPTED;
        }
        if (XFS_FORCED_SHUTDOWN(mp))
-                return XFS_ERROR(EIO);
+                return -EIO;
        XFS_STATS_INC(xs_blk_mapw);
@@ -4469,7 +4465,7 @@ xfs_bmapi_convert_unwritten(
         * so generate another request.
         */
        if (mval->br_blockcount < len)
-                return EAGAIN;
+                return -EAGAIN;
        return 0;
 }
@@ -4540,11 +4536,11 @@ xfs_bmapi_write(
             XFS_IFORK_FORMAT(ip, whichfork) != XFS_DINODE_FMT_BTREE),
             mp, XFS_ERRTAG_BMAPIFORMAT, XFS_RANDOM_BMAPIFORMAT))) {
                XFS_ERROR_REPORT("xfs_bmapi_write", XFS_ERRLEVEL_LOW, mp);
-                return XFS_ERROR(EFSCORRUPTED);
+                return -EFSCORRUPTED;
        }
        if (XFS_FORCED_SHUTDOWN(mp))
-                return XFS_ERROR(EIO);
+                return -EIO;
        ifp = XFS_IFORK_PTR(ip, whichfork);
@@ -4620,7 +4616,7 @@ xfs_bmapi_write(
                /* Execute unwritten extent conversion if necessary */
                error = xfs_bmapi_convert_unwritten(&bma, mval, len, flags);
-                if (error == EAGAIN)
+                if (error == -EAGAIN)
                        continue;
                if (error)
                        goto error0;
@@ -4922,7 +4918,7 @@ xfs_bmap_del_extent(
                                        goto done;
                                cur->bc_rec.b = new;
                                error = xfs_btree_insert(cur, &i);
-                                if (error && error != ENOSPC)
+                                if (error && error != -ENOSPC)
                                        goto done;
                                /*
                                 * If get no-space back from btree insert,
@@ -4930,7 +4926,7 @@ xfs_bmap_del_extent(
                                 * block reservation.
                                 * Fix up our state and return the error.
                                 */
-                                if (error == ENOSPC) {
+                                if (error == -ENOSPC) {
                                        /*
                                         * Reset the cursor, don't trust
                                         * it after any insert operation.
@@ -4958,7 +4954,7 @@ xfs_bmap_del_extent(
                                        xfs_bmbt_set_blockcount(ep,
                                                got.br_blockcount);
                                        flags = 0;
-                                        error = XFS_ERROR(ENOSPC);
+                                        error = -ENOSPC;
                                        goto done;
                                }
                                XFS_WANT_CORRUPTED_GOTO(i == 1, done);
@@ -5076,11 +5072,11 @@ xfs_bunmapi(
            XFS_IFORK_FORMAT(ip, whichfork) != XFS_DINODE_FMT_BTREE)) {
                XFS_ERROR_REPORT("xfs_bunmapi", XFS_ERRLEVEL_LOW,
                                 ip->i_mount);
-                return XFS_ERROR(EFSCORRUPTED);
+                return -EFSCORRUPTED;
        }
        mp = ip->i_mount;
        if (XFS_FORCED_SHUTDOWN(mp))
-                return XFS_ERROR(EIO);
+                return -EIO;
        ASSERT(xfs_isilocked(ip, XFS_ILOCK_EXCL));
        ASSERT(len > 0);
@@ -5325,7 +5321,7 @@ xfs_bunmapi(
                    del.br_startoff > got.br_startoff &&
                    del.br_startoff + del.br_blockcount <
                    got.br_startoff + got.br_blockcount) {
-                        error = XFS_ERROR(ENOSPC);
+                        error = -ENOSPC;
                        goto error0;
                }
                error = xfs_bmap_del_extent(ip, tp, &lastx, flist, cur, &del,
@@ -5449,11 +5445,11 @@ xfs_bmap_shift_extents(
             mp, XFS_ERRTAG_BMAPIFORMAT, XFS_RANDOM_BMAPIFORMAT))) {
                XFS_ERROR_REPORT("xfs_bmap_shift_extents",
                                 XFS_ERRLEVEL_LOW, mp);
-                return XFS_ERROR(EFSCORRUPTED);
+                return -EFSCORRUPTED;
        }
        if (XFS_FORCED_SHUTDOWN(mp))
-                return XFS_ERROR(EIO);
+                return -EIO;
        ASSERT(current_ext != NULL);
@@ -5516,14 +5512,14 @@ xfs_bmap_shift_extents(
                                                *current_ext - 1), &left);
                        if (startoff < left.br_startoff + left.br_blockcount)
-                                error = XFS_ERROR(EINVAL);
+                                error = -EINVAL;
                } else if (offset_shift_fsb > got.br_startoff) {
                        /*
                         * When first extent is shifted, offset_shift_fsb
                         * should be less than the stating offset of
                         * the first extent.
                         */
-                        error = XFS_ERROR(EINVAL);
+                        error = -EINVAL;
                }
                if (error)
diff --git a/fs/xfs/xfs_bmap.h b/fs/xfs/libxfs/xfs_bmap.h
index b879ca56a64c..b879ca56a64c 100644
--- a/fs/xfs/xfs_bmap.h
+++ b/fs/xfs/libxfs/xfs_bmap.h
diff --git a/fs/xfs/xfs_bmap_btree.c b/fs/xfs/libxfs/xfs_bmap_btree.c
index 948836c4fd90..fba753308f31 100644
--- a/fs/xfs/xfs_bmap_btree.c
+++ b/fs/xfs/libxfs/xfs_bmap_btree.c
@@ -111,23 +111,8 @@ __xfs_bmbt_get_all(
        ext_flag = (int)(l0 >> (64 - BMBT_EXNTFLAG_BITLEN));
        s->br_startoff = ((xfs_fileoff_t)l0 &
                           xfs_mask64lo(64 - BMBT_EXNTFLAG_BITLEN)) >> 9;
-#if XFS_BIG_BLKNOS
        s->br_startblock = (((xfs_fsblock_t)l0 & xfs_mask64lo(9)) << 43) |
                           (((xfs_fsblock_t)l1) >> 21);
-#else
-#ifdef DEBUG
-        {
-                xfs_dfsbno_t    b;
-                b = (((xfs_dfsbno_t)l0 & xfs_mask64lo(9)) << 43) |
-                    (((xfs_dfsbno_t)l1) >> 21);
-                ASSERT((b >> 32) == 0 || isnulldstartblock(b));
-                s->br_startblock = (xfs_fsblock_t)b;
-        }
-#else   /* !DEBUG */
-        s->br_startblock = (xfs_fsblock_t)(((xfs_dfsbno_t)l1) >> 21);
-#endif  /* DEBUG */
-#endif  /* XFS_BIG_BLKNOS */
        s->br_blockcount = (xfs_filblks_t)(l1 & xfs_mask64lo(21));
        /* This is xfs_extent_state() in-line */
        if (ext_flag) {
@@ -163,21 +148,8 @@ xfs_fsblock_t
 xfs_bmbt_get_startblock(
        xfs_bmbt_rec_host_t     *r)
 {
-#if XFS_BIG_BLKNOS
        return (((xfs_fsblock_t)r->l0 & xfs_mask64lo(9)) << 43) |
               (((xfs_fsblock_t)r->l1) >> 21);
-#else
-#ifdef DEBUG
-        xfs_dfsbno_t    b;
-        b = (((xfs_dfsbno_t)r->l0 & xfs_mask64lo(9)) << 43) |
-            (((xfs_dfsbno_t)r->l1) >> 21);
-        ASSERT((b >> 32) == 0 || isnulldstartblock(b));
-        return (xfs_fsblock_t)b;
-#else   /* !DEBUG */
-        return (xfs_fsblock_t)(((xfs_dfsbno_t)r->l1) >> 21);
-#endif  /* DEBUG */
-#endif  /* XFS_BIG_BLKNOS */
 }
 /*
@@ -241,7 +213,6 @@ xfs_bmbt_set_allf(
        ASSERT((startoff & xfs_mask64hi(64-BMBT_STARTOFF_BITLEN)) == 0);
        ASSERT((blockcount & xfs_mask64hi(64-BMBT_BLOCKCOUNT_BITLEN)) == 0);
-#if XFS_BIG_BLKNOS
        ASSERT((startblock & xfs_mask64hi(64-BMBT_STARTBLOCK_BITLEN)) == 0);
        r->l0 = ((xfs_bmbt_rec_base_t)extent_flag << 63) |
@@ -250,23 +221,6 @@ xfs_bmbt_set_allf(
        r->l1 = ((xfs_bmbt_rec_base_t)startblock << 21) |
                ((xfs_bmbt_rec_base_t)blockcount &
                (xfs_bmbt_rec_base_t)xfs_mask64lo(21));
-#else   /* !XFS_BIG_BLKNOS */
-        if (isnullstartblock(startblock)) {
-                r->l0 = ((xfs_bmbt_rec_base_t)extent_flag << 63) |
-                        ((xfs_bmbt_rec_base_t)startoff << 9) |
-                         (xfs_bmbt_rec_base_t)xfs_mask64lo(9);
-                r->l1 = xfs_mask64hi(11) |
-                          ((xfs_bmbt_rec_base_t)startblock << 21) |
-                          ((xfs_bmbt_rec_base_t)blockcount &
-                           (xfs_bmbt_rec_base_t)xfs_mask64lo(21));
-        } else {
-                r->l0 = ((xfs_bmbt_rec_base_t)extent_flag << 63) |
-                        ((xfs_bmbt_rec_base_t)startoff << 9);
-                r->l1 = ((xfs_bmbt_rec_base_t)startblock << 21) |
-                         ((xfs_bmbt_rec_base_t)blockcount &
-                         (xfs_bmbt_rec_base_t)xfs_mask64lo(21));
-        }
-#endif  /* XFS_BIG_BLKNOS */
 }
 /*
@@ -298,8 +252,6 @@ xfs_bmbt_disk_set_allf(
        ASSERT(state == XFS_EXT_NORM || state == XFS_EXT_UNWRITTEN);
        ASSERT((startoff & xfs_mask64hi(64-BMBT_STARTOFF_BITLEN)) == 0);
        ASSERT((blockcount & xfs_mask64hi(64-BMBT_BLOCKCOUNT_BITLEN)) == 0);
-#if XFS_BIG_BLKNOS
        ASSERT((startblock & xfs_mask64hi(64-BMBT_STARTBLOCK_BITLEN)) == 0);
        r->l0 = cpu_to_be64(
@@ -310,26 +262,6 @@ xfs_bmbt_disk_set_allf(
                ((xfs_bmbt_rec_base_t)startblock << 21) |
                 ((xfs_bmbt_rec_base_t)blockcount &
                  (xfs_bmbt_rec_base_t)xfs_mask64lo(21)));
-#else   /* !XFS_BIG_BLKNOS */
-        if (isnullstartblock(startblock)) {
-                r->l0 = cpu_to_be64(
-                        ((xfs_bmbt_rec_base_t)extent_flag << 63) |
-                         ((xfs_bmbt_rec_base_t)startoff << 9) |
-                          (xfs_bmbt_rec_base_t)xfs_mask64lo(9));
-                r->l1 = cpu_to_be64(xfs_mask64hi(11) |
-                          ((xfs_bmbt_rec_base_t)startblock << 21) |
-                          ((xfs_bmbt_rec_base_t)blockcount &
-                           (xfs_bmbt_rec_base_t)xfs_mask64lo(21)));
-        } else {
-                r->l0 = cpu_to_be64(
-                        ((xfs_bmbt_rec_base_t)extent_flag << 63) |
-                         ((xfs_bmbt_rec_base_t)startoff << 9));
-                r->l1 = cpu_to_be64(
-                        ((xfs_bmbt_rec_base_t)startblock << 21) |
-                         ((xfs_bmbt_rec_base_t)blockcount &
-                          (xfs_bmbt_rec_base_t)xfs_mask64lo(21)));
-        }
-#endif  /* XFS_BIG_BLKNOS */
 }
 /*
@@ -365,24 +297,11 @@ xfs_bmbt_set_startblock(
        xfs_bmbt_rec_host_t *r,
        xfs_fsblock_t   v)
 {
-#if XFS_BIG_BLKNOS
        ASSERT((v & xfs_mask64hi(12)) == 0);
        r->l0 = (r->l0 & (xfs_bmbt_rec_base_t)xfs_mask64hi(55)) |
                  (xfs_bmbt_rec_base_t)(v >> 43);
        r->l1 = (r->l1 & (xfs_bmbt_rec_base_t)xfs_mask64lo(21)) |
                  (xfs_bmbt_rec_base_t)(v << 21);
-#else   /* !XFS_BIG_BLKNOS */
-        if (isnullstartblock(v)) {
-                r->l0 |= (xfs_bmbt_rec_base_t)xfs_mask64lo(9);
-                r->l1 = (xfs_bmbt_rec_base_t)xfs_mask64hi(11) |
-                          ((xfs_bmbt_rec_base_t)v << 21) |
-                          (r->l1 & (xfs_bmbt_rec_base_t)xfs_mask64lo(21));
-        } else {
-                r->l0 &= ~(xfs_bmbt_rec_base_t)xfs_mask64lo(9);
-                r->l1 = ((xfs_bmbt_rec_base_t)v << 21) |
-                          (r->l1 & (xfs_bmbt_rec_base_t)xfs_mask64lo(21));
-        }
-#endif  /* XFS_BIG_BLKNOS */
 }
 /*
@@ -438,8 +357,8 @@ xfs_bmbt_to_bmdr(
                       cpu_to_be64(XFS_BUF_DADDR_NULL));
        } else
                ASSERT(rblock->bb_magic == cpu_to_be32(XFS_BMAP_MAGIC));
-        ASSERT(rblock->bb_u.l.bb_leftsib == cpu_to_be64(NULLDFSBNO));
+        ASSERT(rblock->bb_u.l.bb_leftsib == cpu_to_be64(NULLFSBLOCK));
-        ASSERT(rblock->bb_u.l.bb_rightsib == cpu_to_be64(NULLDFSBNO));
+        ASSERT(rblock->bb_u.l.bb_rightsib == cpu_to_be64(NULLFSBLOCK));
        ASSERT(rblock->bb_level != 0);
        dblock->bb_level = rblock->bb_level;
        dblock->bb_numrecs = rblock->bb_numrecs;
@@ -554,7 +473,7 @@ xfs_bmbt_alloc_block(
        args.minlen = args.maxlen = args.prod = 1;
        args.wasdel = cur->bc_private.b.flags & XFS_BTCUR_BPRV_WASDEL;
        if (!args.wasdel && xfs_trans_get_block_res(args.tp) == 0) {
-                error = XFS_ERROR(ENOSPC);
+                error = -ENOSPC;
                goto error0;
        }
        error = xfs_alloc_vextent(&args);
@@ -763,11 +682,11 @@ xfs_bmbt_verify(
        /* sibling pointer verification */
        if (!block->bb_u.l.bb_leftsib ||
-            (block->bb_u.l.bb_leftsib != cpu_to_be64(NULLDFSBNO) &&
+            (block->bb_u.l.bb_leftsib != cpu_to_be64(NULLFSBLOCK) &&
             !XFS_FSB_SANITY_CHECK(mp, be64_to_cpu(block->bb_u.l.bb_leftsib))))
                return false;
        if (!block->bb_u.l.bb_rightsib ||
-            (block->bb_u.l.bb_rightsib != cpu_to_be64(NULLDFSBNO) &&
+            (block->bb_u.l.bb_rightsib != cpu_to_be64(NULLFSBLOCK) &&
             !XFS_FSB_SANITY_CHECK(mp, be64_to_cpu(block->bb_u.l.bb_rightsib))))
                return false;
@@ -779,9 +698,9 @@ xfs_bmbt_read_verify(
        struct xfs_buf  *bp)
 {
        if (!xfs_btree_lblock_verify_crc(bp))
-                xfs_buf_ioerror(bp, EFSBADCRC);
+                xfs_buf_ioerror(bp, -EFSBADCRC);
        else if (!xfs_bmbt_verify(bp))
-                xfs_buf_ioerror(bp, EFSCORRUPTED);
+                xfs_buf_ioerror(bp, -EFSCORRUPTED);
        if (bp->b_error) {
                trace_xfs_btree_corrupt(bp, _RET_IP_);
@@ -795,7 +714,7 @@ xfs_bmbt_write_verify(
 {
        if (!xfs_bmbt_verify(bp)) {
                trace_xfs_btree_corrupt(bp, _RET_IP_);
-                xfs_buf_ioerror(bp, EFSCORRUPTED);
+                xfs_buf_ioerror(bp, -EFSCORRUPTED);
                xfs_verifier_error(bp);
                return;
        }
@@ -959,7 +878,7 @@ xfs_bmbt_change_owner(
        cur = xfs_bmbt_init_cursor(ip->i_mount, tp, ip, whichfork);
        if (!cur)
-                return ENOMEM;
+                return -ENOMEM;
        error = xfs_btree_change_owner(cur, new_owner, buffer_list);
        xfs_btree_del_cursor(cur, error ? XFS_BTREE_ERROR : XFS_BTREE_NOERROR);
diff --git a/fs/xfs/xfs_bmap_btree.h b/fs/xfs/libxfs/xfs_bmap_btree.h
index 819a8a4dee95..819a8a4dee95 100644
--- a/fs/xfs/xfs_bmap_btree.h
+++ b/fs/xfs/libxfs/xfs_bmap_btree.h
diff --git a/fs/xfs/xfs_btree.c b/fs/xfs/libxfs/xfs_btree.c
index cf893bc1e373..8fe6a93ff473 100644
--- a/fs/xfs/xfs_btree.c
+++ b/fs/xfs/libxfs/xfs_btree.c
@@ -78,11 +78,11 @@ xfs_btree_check_lblock(
                be16_to_cpu(block->bb_numrecs) <=
                        cur->bc_ops->get_maxrecs(cur, level) &&
                block->bb_u.l.bb_leftsib &&
-                (block->bb_u.l.bb_leftsib == cpu_to_be64(NULLDFSBNO) ||
+                (block->bb_u.l.bb_leftsib == cpu_to_be64(NULLFSBLOCK) ||
                 XFS_FSB_SANITY_CHECK(mp,
                        be64_to_cpu(block->bb_u.l.bb_leftsib))) &&
                block->bb_u.l.bb_rightsib &&
-                (block->bb_u.l.bb_rightsib == cpu_to_be64(NULLDFSBNO) ||
+                (block->bb_u.l.bb_rightsib == cpu_to_be64(NULLFSBLOCK) ||
                 XFS_FSB_SANITY_CHECK(mp,
                        be64_to_cpu(block->bb_u.l.bb_rightsib)));
@@ -92,7 +92,7 @@ xfs_btree_check_lblock(
                if (bp)
                        trace_xfs_btree_corrupt(bp, _RET_IP_);
                XFS_ERROR_REPORT(__func__, XFS_ERRLEVEL_LOW, mp);
-                return XFS_ERROR(EFSCORRUPTED);
+                return -EFSCORRUPTED;
        }
        return 0;
 }
@@ -140,7 +140,7 @@ xfs_btree_check_sblock(
                if (bp)
                        trace_xfs_btree_corrupt(bp, _RET_IP_);
                XFS_ERROR_REPORT(__func__, XFS_ERRLEVEL_LOW, mp);
-                return XFS_ERROR(EFSCORRUPTED);
+                return -EFSCORRUPTED;
        }
        return 0;
 }
@@ -167,12 +167,12 @@ xfs_btree_check_block(
 int                                     /* error (0 or EFSCORRUPTED) */
 xfs_btree_check_lptr(
        struct xfs_btree_cur    *cur,   /* btree cursor */
-        xfs_dfsbno_t            bno,    /* btree block disk address */
+        xfs_fsblock_t           bno,    /* btree block disk address */
        int                     level)  /* btree block level */
 {
        XFS_WANT_CORRUPTED_RETURN(
                level > 0 &&
-                bno != NULLDFSBNO &&
+                bno != NULLFSBLOCK &&
                XFS_FSB_SANITY_CHECK(cur->bc_mp, bno));
        return 0;
 }
@@ -595,7 +595,7 @@ xfs_btree_islastblock(
        block = xfs_btree_get_block(cur, level, &bp);
        xfs_btree_check_block(cur, block, level, bp);
        if (cur->bc_flags & XFS_BTREE_LONG_PTRS)
-                return block->bb_u.l.bb_rightsib == cpu_to_be64(NULLDFSBNO);
+                return block->bb_u.l.bb_rightsib == cpu_to_be64(NULLFSBLOCK);
        else
                return block->bb_u.s.bb_rightsib == cpu_to_be32(NULLAGBLOCK);
 }
@@ -771,16 +771,16 @@ xfs_btree_readahead_lblock(
        struct xfs_btree_block  *block)
 {
        int                     rval = 0;
-        xfs_dfsbno_t            left = be64_to_cpu(block->bb_u.l.bb_leftsib);
+        xfs_fsblock_t           left = be64_to_cpu(block->bb_u.l.bb_leftsib);
-        xfs_dfsbno_t            right = be64_to_cpu(block->bb_u.l.bb_rightsib);
+        xfs_fsblock_t           right = be64_to_cpu(block->bb_u.l.bb_rightsib);
-        if ((lr & XFS_BTCUR_LEFTRA) && left != NULLDFSBNO) {
+        if ((lr & XFS_BTCUR_LEFTRA) && left != NULLFSBLOCK) {
                xfs_btree_reada_bufl(cur->bc_mp, left, 1,
                                     cur->bc_ops->buf_ops);
                rval++;
        }
-        if ((lr & XFS_BTCUR_RIGHTRA) && right != NULLDFSBNO) {
+        if ((lr & XFS_BTCUR_RIGHTRA) && right != NULLFSBLOCK) {
                xfs_btree_reada_bufl(cur->bc_mp, right, 1,
                                     cur->bc_ops->buf_ops);
                rval++;
@@ -852,7 +852,7 @@ xfs_btree_ptr_to_daddr(
        union xfs_btree_ptr     *ptr)
 {
        if (cur->bc_flags & XFS_BTREE_LONG_PTRS) {
-                ASSERT(ptr->l != cpu_to_be64(NULLDFSBNO));
+                ASSERT(ptr->l != cpu_to_be64(NULLFSBLOCK));
                return XFS_FSB_TO_DADDR(cur->bc_mp, be64_to_cpu(ptr->l));
        } else {
@@ -900,9 +900,9 @@ xfs_btree_setbuf(
        b = XFS_BUF_TO_BLOCK(bp);
        if (cur->bc_flags & XFS_BTREE_LONG_PTRS) {
-                if (b->bb_u.l.bb_leftsib == cpu_to_be64(NULLDFSBNO))
+                if (b->bb_u.l.bb_leftsib == cpu_to_be64(NULLFSBLOCK))
                        cur->bc_ra[lev] |= XFS_BTCUR_LEFTRA;
-                if (b->bb_u.l.bb_rightsib == cpu_to_be64(NULLDFSBNO))
+                if (b->bb_u.l.bb_rightsib == cpu_to_be64(NULLFSBLOCK))
                        cur->bc_ra[lev] |= XFS_BTCUR_RIGHTRA;
        } else {
                if (b->bb_u.s.bb_leftsib == cpu_to_be32(NULLAGBLOCK))
@@ -918,7 +918,7 @@ xfs_btree_ptr_is_null(
        union xfs_btree_ptr     *ptr)
 {
        if (cur->bc_flags & XFS_BTREE_LONG_PTRS)
-                return ptr->l == cpu_to_be64(NULLDFSBNO);
+                return ptr->l == cpu_to_be64(NULLFSBLOCK);
        else
                return ptr->s == cpu_to_be32(NULLAGBLOCK);
 }
@@ -929,7 +929,7 @@ xfs_btree_set_ptr_null(
        union xfs_btree_ptr     *ptr)
 {
        if (cur->bc_flags & XFS_BTREE_LONG_PTRS)
-                ptr->l = cpu_to_be64(NULLDFSBNO);
+                ptr->l = cpu_to_be64(NULLFSBLOCK);
        else
                ptr->s = cpu_to_be32(NULLAGBLOCK);
 }
@@ -997,8 +997,8 @@ xfs_btree_init_block_int(
        buf->bb_numrecs = cpu_to_be16(numrecs);
        if (flags & XFS_BTREE_LONG_PTRS) {
-                buf->bb_u.l.bb_leftsib = cpu_to_be64(NULLDFSBNO);
+                buf->bb_u.l.bb_leftsib = cpu_to_be64(NULLFSBLOCK);
-                buf->bb_u.l.bb_rightsib = cpu_to_be64(NULLDFSBNO);
+                buf->bb_u.l.bb_rightsib = cpu_to_be64(NULLFSBLOCK);
                if (flags & XFS_BTREE_CRC_BLOCKS) {
                        buf->bb_u.l.bb_blkno = cpu_to_be64(blkno);
                        buf->bb_u.l.bb_owner = cpu_to_be64(owner);
@@ -1140,7 +1140,7 @@ xfs_btree_get_buf_block(
                                 mp->m_bsize, flags);
        if (!*bpp)
-                return ENOMEM;
+                return -ENOMEM;
        (*bpp)->b_ops = cur->bc_ops->buf_ops;
        *block = XFS_BUF_TO_BLOCK(*bpp);
@@ -1498,7 +1498,7 @@ xfs_btree_increment(
                if (cur->bc_flags & XFS_BTREE_ROOT_IN_INODE)
                        goto out0;
                ASSERT(0);
-                error = EFSCORRUPTED;
+                error = -EFSCORRUPTED;
                goto error0;
        }
        ASSERT(lev < cur->bc_nlevels);
@@ -1597,7 +1597,7 @@ xfs_btree_decrement(
                if (cur->bc_flags & XFS_BTREE_ROOT_IN_INODE)
                        goto out0;
                ASSERT(0);
-                error = EFSCORRUPTED;
+                error = -EFSCORRUPTED;
                goto error0;
        }
        ASSERT(lev < cur->bc_nlevels);
@@ -4018,7 +4018,7 @@ xfs_btree_block_change_owner(
        /* now read rh sibling block for next iteration */
        xfs_btree_get_sibling(cur, block, &rptr, XFS_BB_RIGHTSIB);
        if (xfs_btree_ptr_is_null(cur, &rptr))
-                return ENOENT;
+                return -ENOENT;
        return xfs_btree_lookup_get_block(cur, level, &rptr, &block);
 }
@@ -4061,7 +4061,7 @@ xfs_btree_change_owner(
                                                             buffer_list);
                } while (!error);
-                if (error != ENOENT)
+                if (error != -ENOENT)
                        return error;
        }
diff --git a/fs/xfs/xfs_btree.h b/fs/xfs/libxfs/xfs_btree.h
index a04b69422f67..8f18bab73ea5 100644
--- a/fs/xfs/xfs_btree.h
+++ b/fs/xfs/libxfs/xfs_btree.h
@@ -258,7 +258,7 @@ xfs_btree_check_block(
 int                                     /* error (0 or EFSCORRUPTED) */
 xfs_btree_check_lptr(
        struct xfs_btree_cur    *cur,   /* btree cursor */
-        xfs_dfsbno_t            ptr,    /* btree block disk address */
+        xfs_fsblock_t           ptr,    /* btree block disk address */
        int                     level); /* btree block level */
 /*
diff --git a/fs/xfs/xfs_cksum.h b/fs/xfs/libxfs/xfs_cksum.h
index fad1676ad8cd..fad1676ad8cd 100644
--- a/fs/xfs/xfs_cksum.h
+++ b/fs/xfs/libxfs/xfs_cksum.h
diff --git a/fs/xfs/xfs_da_btree.c b/fs/xfs/libxfs/xfs_da_btree.c
index a514ab616650..2c42ae28d027 100644
--- a/fs/xfs/xfs_da_btree.c
+++ b/fs/xfs/libxfs/xfs_da_btree.c
@@ -185,7 +185,7 @@ xfs_da3_node_write_verify(
        struct xfs_da3_node_hdr *hdr3 = bp->b_addr;
        if (!xfs_da3_node_verify(bp)) {
-                xfs_buf_ioerror(bp, EFSCORRUPTED);
+                xfs_buf_ioerror(bp, -EFSCORRUPTED);
                xfs_verifier_error(bp);
                return;
        }
@@ -214,13 +214,13 @@ xfs_da3_node_read_verify(
        switch (be16_to_cpu(info->magic)) {
                case XFS_DA3_NODE_MAGIC:
                        if (!xfs_buf_verify_cksum(bp, XFS_DA3_NODE_CRC_OFF)) {
-                                xfs_buf_ioerror(bp, EFSBADCRC);
+                                xfs_buf_ioerror(bp, -EFSBADCRC);
                                break;
                        }
                        /* fall through */
                case XFS_DA_NODE_MAGIC:
                        if (!xfs_da3_node_verify(bp)) {
-                                xfs_buf_ioerror(bp, EFSCORRUPTED);
+                                xfs_buf_ioerror(bp, -EFSCORRUPTED);
                                break;
                        }
                        return;
@@ -315,7 +315,7 @@ xfs_da3_node_create(
        error = xfs_da_get_buf(tp, dp, blkno, -1, &bp, whichfork);
        if (error)
-                return(error);
+                return error;
        bp->b_ops = &xfs_da3_node_buf_ops;
        xfs_trans_buf_set_type(tp, bp, XFS_BLFT_DA_NODE_BUF);
        node = bp->b_addr;
@@ -337,7 +337,7 @@ xfs_da3_node_create(
                XFS_DA_LOGRANGE(node, &node->hdr, dp->d_ops->node_hdr_size));
        *bpp = bp;
-        return(0);
+        return 0;
 }
 /*
@@ -385,8 +385,8 @@ xfs_da3_split(
                switch (oldblk->magic) {
                case XFS_ATTR_LEAF_MAGIC:
                        error = xfs_attr3_leaf_split(state, oldblk, newblk);
-                        if ((error != 0) && (error != ENOSPC)) {
+                        if ((error != 0) && (error != -ENOSPC)) {
-                                return(error);  /* GROT: attr is inconsistent */
+                                return error;   /* GROT: attr is inconsistent */
                        }
                        if (!error) {
                                addblk = newblk;
@@ -408,7 +408,7 @@ xfs_da3_split(
                                                            &state->extrablk);
                        }
                        if (error)
-                                return(error);  /* GROT: attr inconsistent */
+                                return error;   /* GROT: attr inconsistent */
                        addblk = newblk;
                        break;
                case XFS_DIR2_LEAFN_MAGIC:
@@ -422,7 +422,7 @@ xfs_da3_split(
                                                         max - i, &action);
                        addblk->bp = NULL;
                        if (error)
-                                return(error);  /* GROT: dir is inconsistent */
+                                return error;   /* GROT: dir is inconsistent */
                        /*
                         * Record the newly split block for the next time thru?
                         */
@@ -439,7 +439,7 @@ xfs_da3_split(
                xfs_da3_fixhashpath(state, &state->path);
        }
        if (!addblk)
-                return(0);
+                return 0;
        /*
         * Split the root node.
@@ -449,7 +449,7 @@ xfs_da3_split(
        error = xfs_da3_root_split(state, oldblk, addblk);
        if (error) {
                addblk->bp = NULL;
-                return(error);  /* GROT: dir is inconsistent */
+                return error;   /* GROT: dir is inconsistent */
        }
        /*
@@ -492,7 +492,7 @@ xfs_da3_split(
                    sizeof(node->hdr.info)));
        }
        addblk->bp = NULL;
-        return(0);
+        return 0;
 }
 /*
@@ -670,18 +670,18 @@ xfs_da3_node_split(
                 */
                error = xfs_da_grow_inode(state->args, &blkno);
                if (error)
-                        return(error);  /* GROT: dir is inconsistent */
+                        return error;   /* GROT: dir is inconsistent */
                error = xfs_da3_node_create(state->args, blkno, treelevel,
                                           &newblk->bp, state->args->whichfork);
                if (error)
-                        return(error);  /* GROT: dir is inconsistent */
+                        return error;   /* GROT: dir is inconsistent */
                newblk->blkno = blkno;
                newblk->magic = XFS_DA_NODE_MAGIC;
                xfs_da3_node_rebalance(state, oldblk, newblk);
                error = xfs_da3_blk_link(state, oldblk, newblk);
                if (error)
-                        return(error);
+                        return error;
                *result = 1;
        } else {
                *result = 0;
@@ -721,7 +721,7 @@ xfs_da3_node_split(
                }
        }
-        return(0);
+        return 0;
 }
 /*
@@ -963,9 +963,9 @@ xfs_da3_join(
                case XFS_ATTR_LEAF_MAGIC:
                        error = xfs_attr3_leaf_toosmall(state, &action);
                        if (error)
-                                return(error);
+                                return error;
                        if (action == 0)
-                                return(0);
+                                return 0;
                        xfs_attr3_leaf_unbalance(state, drop_blk, save_blk);
                        break;
                case XFS_DIR2_LEAFN_MAGIC:
@@ -985,7 +985,7 @@ xfs_da3_join(
                        xfs_da3_fixhashpath(state, &state->path);
                        error = xfs_da3_node_toosmall(state, &action);
                        if (error)
-                                return(error);
+                                return error;
                        if (action == 0)
                                return 0;
                        xfs_da3_node_unbalance(state, drop_blk, save_blk);
@@ -995,12 +995,12 @@ xfs_da3_join(
                error = xfs_da3_blk_unlink(state, drop_blk, save_blk);
                xfs_da_state_kill_altpath(state);
                if (error)
-                        return(error);
+                        return error;
                error = xfs_da_shrink_inode(state->args, drop_blk->blkno,
                                                         drop_blk->bp);
                drop_blk->bp = NULL;
                if (error)
-                        return(error);
+                        return error;
        }
        /*
         * We joined all the way to the top.  If it turns out that
@@ -1010,7 +1010,7 @@ xfs_da3_join(
        xfs_da3_node_remove(state, drop_blk);
        xfs_da3_fixhashpath(state, &state->path);
        error = xfs_da3_root_join(state, &state->path.blk[0]);
-        return(error);
+        return error;
 }
 #ifdef  DEBUG
@@ -1099,7 +1099,7 @@ xfs_da3_root_join(
        xfs_trans_log_buf(args->trans, root_blk->bp, 0,
                          args->geo->blksize - 1);
        error = xfs_da_shrink_inode(args, child, bp);
-        return(error);
+        return error;
 }
 /*
@@ -1142,7 +1142,7 @@ xfs_da3_node_toosmall(
        dp->d_ops->node_hdr_from_disk(&nodehdr, node);
        if (nodehdr.count > (state->args->geo->node_ents >> 1)) {
                *action = 0;    /* blk over 50%, don't try to join */
-                return(0);      /* blk over 50%, don't try to join */
+                return 0;       /* blk over 50%, don't try to join */
        }
        /*
@@ -1161,13 +1161,13 @@ xfs_da3_node_toosmall(
                error = xfs_da3_path_shift(state, &state->altpath, forward,
                                                 0, &retval);
                if (error)
-                        return(error);
+                        return error;
                if (retval) {
                        *action = 0;
                } else {
                        *action = 2;
                }
-                return(0);
+                return 0;
        }
        /*
@@ -1194,7 +1194,7 @@ xfs_da3_node_toosmall(
                error = xfs_da3_node_read(state->args->trans, dp,
                                        blkno, -1, &bp, state->args->whichfork);
                if (error)
-                        return(error);
+                        return error;
                node = bp->b_addr;
                dp->d_ops->node_hdr_from_disk(&thdr, node);
@@ -1486,7 +1486,7 @@ xfs_da3_node_lookup_int(
                if (error) {
                        blk->blkno = 0;
                        state->path.active--;
-                        return(error);
+                        return error;
                }
                curr = blk->bp->b_addr;
                blk->magic = be16_to_cpu(curr->magic);
@@ -1579,25 +1579,25 @@ xfs_da3_node_lookup_int(
                        args->blkno = blk->blkno;
                } else {
                        ASSERT(0);
-                        return XFS_ERROR(EFSCORRUPTED);
+                        return -EFSCORRUPTED;
                }
-                if (((retval == ENOENT) || (retval == ENOATTR)) &&
+                if (((retval == -ENOENT) || (retval == -ENOATTR)) &&
                    (blk->hashval == args->hashval)) {
                        error = xfs_da3_path_shift(state, &state->path, 1, 1,
                                                         &retval);
                        if (error)
-                                return(error);
+                                return error;
                        if (retval == 0) {
                                continue;
                        } else if (blk->magic == XFS_ATTR_LEAF_MAGIC) {
                                /* path_shift() gives ENOENT */
-                                retval = XFS_ERROR(ENOATTR);
+                                retval = -ENOATTR;
                        }
                }
                break;
        }
        *result = retval;
-        return(0);
+        return 0;
 }
 /*========================================================================
@@ -1692,7 +1692,7 @@ xfs_da3_blk_link(
                                                be32_to_cpu(old_info->back),
                                                -1, &bp, args->whichfork);
                        if (error)
-                                return(error);
+                                return error;
                        ASSERT(bp != NULL);
                        tmp_info = bp->b_addr;
                        ASSERT(tmp_info->magic == old_info->magic);
@@ -1713,7 +1713,7 @@ xfs_da3_blk_link(
                                                be32_to_cpu(old_info->forw),
                                                -1, &bp, args->whichfork);
                        if (error)
-                                return(error);
+                                return error;
                        ASSERT(bp != NULL);
                        tmp_info = bp->b_addr;
                        ASSERT(tmp_info->magic == old_info->magic);
@@ -1726,7 +1726,7 @@ xfs_da3_blk_link(
        xfs_trans_log_buf(args->trans, old_blk->bp, 0, sizeof(*tmp_info) - 1);
        xfs_trans_log_buf(args->trans, new_blk->bp, 0, sizeof(*tmp_info) - 1);
-        return(0);
+        return 0;
 }
 /*
@@ -1772,7 +1772,7 @@ xfs_da3_blk_unlink(
                                                be32_to_cpu(drop_info->back),
                                                -1, &bp, args->whichfork);
                        if (error)
-                                return(error);
+                                return error;
                        ASSERT(bp != NULL);
                        tmp_info = bp->b_addr;
                        ASSERT(tmp_info->magic == save_info->magic);
@@ -1789,7 +1789,7 @@ xfs_da3_blk_unlink(
                                                be32_to_cpu(drop_info->forw),
                                                -1, &bp, args->whichfork);
                        if (error)
-                                return(error);
+                                return error;
                        ASSERT(bp != NULL);
                        tmp_info = bp->b_addr;
                        ASSERT(tmp_info->magic == save_info->magic);
@@ -1801,7 +1801,7 @@ xfs_da3_blk_unlink(
        }
        xfs_trans_log_buf(args->trans, save_blk->bp, 0, sizeof(*save_info) - 1);
-        return(0);
+        return 0;
 }
 /*
@@ -1859,9 +1859,9 @@ xfs_da3_path_shift(
                }
        }
        if (level < 0) {
-                *result = XFS_ERROR(ENOENT);    /* we're out of our tree */
+                *result = -ENOENT;      /* we're out of our tree */
                ASSERT(args->op_flags & XFS_DA_OP_OKNOENT);
-                return(0);
+                return 0;
        }
        /*
@@ -1883,7 +1883,7 @@ xfs_da3_path_shift(
                error = xfs_da3_node_read(args->trans, dp, blkno, -1,
                                        &blk->bp, args->whichfork);
                if (error)
-                        return(error);
+                        return error;
                info = blk->bp->b_addr;
                ASSERT(info->magic == cpu_to_be16(XFS_DA_NODE_MAGIC) ||
                       info->magic == cpu_to_be16(XFS_DA3_NODE_MAGIC) ||
@@ -2004,7 +2004,7 @@ xfs_da_grow_inode_int(
        struct xfs_trans        *tp = args->trans;
        struct xfs_inode        *dp = args->dp;
        int                     w = args->whichfork;
-        xfs_drfsbno_t           nblks = dp->i_d.di_nblocks;
+        xfs_rfsblock_t          nblks = dp->i_d.di_nblocks;
        struct xfs_bmbt_irec    map, *mapp;
        int                     nmap, error, got, i, mapi;
@@ -2068,7 +2068,7 @@ xfs_da_grow_inode_int(
        if (got != count || mapp[0].br_startoff != *bno ||
            mapp[mapi - 1].br_startoff + mapp[mapi - 1].br_blockcount !=
            *bno + count) {
-                error = XFS_ERROR(ENOSPC);
+                error = -ENOSPC;
                goto out_free_map;
        }
@@ -2158,7 +2158,7 @@ xfs_da3_swap_lastblock(
        if (unlikely(lastoff == 0)) {
                XFS_ERROR_REPORT("xfs_da_swap_lastblock(1)", XFS_ERRLEVEL_LOW,
                                 mp);
-                return XFS_ERROR(EFSCORRUPTED);
+                return -EFSCORRUPTED;
        }
        /*
         * Read the last block in the btree space.
@@ -2209,7 +2209,7 @@ xfs_da3_swap_lastblock(
                    sib_info->magic != dead_info->magic)) {
                        XFS_ERROR_REPORT("xfs_da_swap_lastblock(2)",
                                         XFS_ERRLEVEL_LOW, mp);
-                        error = XFS_ERROR(EFSCORRUPTED);
+                        error = -EFSCORRUPTED;
                        goto done;
                }
                sib_info->forw = cpu_to_be32(dead_blkno);
@@ -2231,7 +2231,7 @@ xfs_da3_swap_lastblock(
                       sib_info->magic != dead_info->magic)) {
                        XFS_ERROR_REPORT("xfs_da_swap_lastblock(3)",
                                         XFS_ERRLEVEL_LOW, mp);
-                        error = XFS_ERROR(EFSCORRUPTED);
+                        error = -EFSCORRUPTED;
                        goto done;
                }
                sib_info->back = cpu_to_be32(dead_blkno);
@@ -2254,7 +2254,7 @@ xfs_da3_swap_lastblock(
                if (level >= 0 && level != par_hdr.level + 1) {
                        XFS_ERROR_REPORT("xfs_da_swap_lastblock(4)",
                                         XFS_ERRLEVEL_LOW, mp);
-                        error = XFS_ERROR(EFSCORRUPTED);
+                        error = -EFSCORRUPTED;
                        goto done;
                }
                level = par_hdr.level;
@@ -2267,7 +2267,7 @@ xfs_da3_swap_lastblock(
                if (entno == par_hdr.count) {
                        XFS_ERROR_REPORT("xfs_da_swap_lastblock(5)",
                                         XFS_ERRLEVEL_LOW, mp);
-                        error = XFS_ERROR(EFSCORRUPTED);
+                        error = -EFSCORRUPTED;
                        goto done;
                }
                par_blkno = be32_to_cpu(btree[entno].before);
@@ -2294,7 +2294,7 @@ xfs_da3_swap_lastblock(
                if (unlikely(par_blkno == 0)) {
                        XFS_ERROR_REPORT("xfs_da_swap_lastblock(6)",
                                         XFS_ERRLEVEL_LOW, mp);
-                        error = XFS_ERROR(EFSCORRUPTED);
+                        error = -EFSCORRUPTED;
                        goto done;
                }
                error = xfs_da3_node_read(tp, dp, par_blkno, -1, &par_buf, w);
@@ -2305,7 +2305,7 @@ xfs_da3_swap_lastblock(
                if (par_hdr.level != level) {
                        XFS_ERROR_REPORT("xfs_da_swap_lastblock(7)",
                                         XFS_ERRLEVEL_LOW, mp);
-                        error = XFS_ERROR(EFSCORRUPTED);
+                        error = -EFSCORRUPTED;
                        goto done;
                }
                btree = dp->d_ops->node_tree_p(par_node);
@@ -2359,7 +2359,7 @@ xfs_da_shrink_inode(
                error = xfs_bunmapi(tp, dp, dead_blkno, count,
                                    xfs_bmapi_aflag(w)|XFS_BMAPI_METADATA,
                                    0, args->firstblock, args->flist, &done);
-                if (error == ENOSPC) {
+                if (error == -ENOSPC) {
                        if (w != XFS_DATA_FORK)
                                break;
                        error = xfs_da3_swap_lastblock(args, &dead_blkno,
@@ -2427,7 +2427,7 @@ xfs_buf_map_from_irec(
                map = kmem_zalloc(nirecs * sizeof(struct xfs_buf_map),
                                  KM_SLEEP | KM_NOFS);
                if (!map)
-                        return ENOMEM;
+                        return -ENOMEM;
                *mapp = map;
        }
@@ -2500,8 +2500,8 @@ xfs_dabuf_map(
        }
        if (!xfs_da_map_covers_blocks(nirecs, irecs, bno, nfsb)) {
-                error = mappedbno == -2 ? -1 : XFS_ERROR(EFSCORRUPTED);
+                error = mappedbno == -2 ? -1 : -EFSCORRUPTED;
-                if (unlikely(error == EFSCORRUPTED)) {
+                if (unlikely(error == -EFSCORRUPTED)) {
                        if (xfs_error_level >= XFS_ERRLEVEL_LOW) {
                                int i;
                                xfs_alert(mp, "%s: bno %lld dir: inode %lld",
@@ -2561,7 +2561,7 @@ xfs_da_get_buf(
        bp = xfs_trans_get_buf_map(trans, dp->i_mount->m_ddev_targp,
                                    mapp, nmap, 0);
-        error = bp ? bp->b_error : XFS_ERROR(EIO);
+        error = bp ? bp->b_error : -EIO;
        if (error) {
                xfs_trans_brelse(trans, bp);
                goto out_free;
diff --git a/fs/xfs/xfs_da_btree.h b/fs/xfs/libxfs/xfs_da_btree.h
index 6e153e399a77..6e153e399a77 100644
--- a/fs/xfs/xfs_da_btree.h
+++ b/fs/xfs/libxfs/xfs_da_btree.h
diff --git a/fs/xfs/xfs_da_format.c b/fs/xfs/libxfs/xfs_da_format.c
index c9aee52a37e2..c9aee52a37e2 100644
--- a/fs/xfs/xfs_da_format.c
+++ b/fs/xfs/libxfs/xfs_da_format.c
diff --git a/fs/xfs/xfs_da_format.h b/fs/xfs/libxfs/xfs_da_format.h
index 0a49b0286372..0a49b0286372 100644
--- a/fs/xfs/xfs_da_format.h
+++ b/fs/xfs/libxfs/xfs_da_format.h
diff --git a/fs/xfs/xfs_dinode.h b/fs/xfs/libxfs/xfs_dinode.h
index 623bbe8fd921..623bbe8fd921 100644
--- a/fs/xfs/xfs_dinode.h
+++ b/fs/xfs/libxfs/xfs_dinode.h
diff --git a/fs/xfs/xfs_dir2.c b/fs/xfs/libxfs/xfs_dir2.c
index 79670cda48ae..6cef22152fd6 100644
--- a/fs/xfs/xfs_dir2.c
+++ b/fs/xfs/libxfs/xfs_dir2.c
@@ -108,7 +108,7 @@ xfs_da_mount(
        if (!mp->m_dir_geo || !mp->m_attr_geo) {
                kmem_free(mp->m_dir_geo);
                kmem_free(mp->m_attr_geo);
-                return ENOMEM;
+                return -ENOMEM;
        }
        /* set up directory geometry */
@@ -202,7 +202,7 @@ xfs_dir_ino_validate(
                xfs_warn(mp, "Invalid inode number 0x%Lx",
                                (unsigned long long) ino);
                XFS_ERROR_REPORT("xfs_dir_ino_validate", XFS_ERRLEVEL_LOW, mp);
-                return XFS_ERROR(EFSCORRUPTED);
+                return -EFSCORRUPTED;
        }
        return 0;
 }
@@ -226,7 +226,7 @@ xfs_dir_init(
        args = kmem_zalloc(sizeof(*args), KM_SLEEP | KM_NOFS);
        if (!args)
-                return ENOMEM;
+                return -ENOMEM;
        args->geo = dp->i_mount->m_dir_geo;
        args->dp = dp;
@@ -261,7 +261,7 @@ xfs_dir_createname(
        args = kmem_zalloc(sizeof(*args), KM_SLEEP | KM_NOFS);
        if (!args)
-                return ENOMEM;
+                return -ENOMEM;
        args->geo = dp->i_mount->m_dir_geo;
        args->name = name->name;
@@ -314,18 +314,18 @@ xfs_dir_cilookup_result(
        int             len)
 {
        if (args->cmpresult == XFS_CMP_DIFFERENT)
-                return ENOENT;
+                return -ENOENT;
        if (args->cmpresult != XFS_CMP_CASE ||
                                        !(args->op_flags & XFS_DA_OP_CILOOKUP))
-                return EEXIST;
+                return -EEXIST;
        args->value = kmem_alloc(len, KM_NOFS | KM_MAYFAIL);
        if (!args->value)
-                return ENOMEM;
+                return -ENOMEM;
        memcpy(args->value, name, len);
        args->valuelen = len;
-        return EEXIST;
+        return -EEXIST;
 }
 /*
@@ -392,7 +392,7 @@ xfs_dir_lookup(
                rval = xfs_dir2_node_lookup(args);
 out_check_rval:
-        if (rval == EEXIST)
+        if (rval == -EEXIST)
                rval = 0;
        if (!rval) {
                *inum = args->inumber;
@@ -428,7 +428,7 @@ xfs_dir_removename(
        args = kmem_zalloc(sizeof(*args), KM_SLEEP | KM_NOFS);
        if (!args)
-                return ENOMEM;
+                return -ENOMEM;
        args->geo = dp->i_mount->m_dir_geo;
        args->name = name->name;
@@ -493,7 +493,7 @@ xfs_dir_replace(
        args = kmem_zalloc(sizeof(*args), KM_SLEEP | KM_NOFS);
        if (!args)
-                return ENOMEM;
+                return -ENOMEM;
        args->geo = dp->i_mount->m_dir_geo;
        args->name = name->name;
@@ -555,7 +555,7 @@ xfs_dir_canenter(
        args = kmem_zalloc(sizeof(*args), KM_SLEEP | KM_NOFS);
        if (!args)
-                return ENOMEM;
+                return -ENOMEM;
        args->geo = dp->i_mount->m_dir_geo;
        args->name = name->name;
diff --git a/fs/xfs/xfs_dir2.h b/fs/xfs/libxfs/xfs_dir2.h
index c8e86b0b5e99..c8e86b0b5e99 100644
--- a/fs/xfs/xfs_dir2.h
+++ b/fs/xfs/libxfs/xfs_dir2.h
diff --git a/fs/xfs/xfs_dir2_block.c b/fs/xfs/libxfs/xfs_dir2_block.c
index c7cd3154026a..9628ceccfa02 100644
--- a/fs/xfs/xfs_dir2_block.c
+++ b/fs/xfs/libxfs/xfs_dir2_block.c
@@ -91,9 +91,9 @@ xfs_dir3_block_read_verify(
        if (xfs_sb_version_hascrc(&mp->m_sb) &&
             !xfs_buf_verify_cksum(bp, XFS_DIR3_DATA_CRC_OFF))
-                xfs_buf_ioerror(bp, EFSBADCRC);
+                xfs_buf_ioerror(bp, -EFSBADCRC);
        else if (!xfs_dir3_block_verify(bp))
-                xfs_buf_ioerror(bp, EFSCORRUPTED);
+                xfs_buf_ioerror(bp, -EFSCORRUPTED);
        if (bp->b_error)
                xfs_verifier_error(bp);
@@ -108,7 +108,7 @@ xfs_dir3_block_write_verify(
        struct xfs_dir3_blk_hdr *hdr3 = bp->b_addr;
        if (!xfs_dir3_block_verify(bp)) {
-                xfs_buf_ioerror(bp, EFSCORRUPTED);
+                xfs_buf_ioerror(bp, -EFSCORRUPTED);
                xfs_verifier_error(bp);
                return;
        }
@@ -392,7 +392,7 @@ xfs_dir2_block_addname(
        if (args->op_flags & XFS_DA_OP_JUSTCHECK) {
                xfs_trans_brelse(tp, bp);
                if (!dup)
-                        return XFS_ERROR(ENOSPC);
+                        return -ENOSPC;
                return 0;
        }
@@ -402,7 +402,7 @@ xfs_dir2_block_addname(
        if (!dup) {
                /* Don't have a space reservation: return no-space.  */
                if (args->total == 0)
-                        return XFS_ERROR(ENOSPC);
+                        return -ENOSPC;
                /*
                 * Convert to the next larger format.
                 * Then add the new entry in that format.
@@ -647,7 +647,7 @@ xfs_dir2_block_lookup(
        args->filetype = dp->d_ops->data_get_ftype(dep);
        error = xfs_dir_cilookup_result(args, dep->name, dep->namelen);
        xfs_trans_brelse(args->trans, bp);
-        return XFS_ERROR(error);
+        return error;
 }
 /*
@@ -703,7 +703,7 @@ xfs_dir2_block_lookup_int(
                if (low > high) {
                        ASSERT(args->op_flags & XFS_DA_OP_OKNOENT);
                        xfs_trans_brelse(tp, bp);
-                        return XFS_ERROR(ENOENT);
+                        return -ENOENT;
                }
        }
        /*
@@ -751,7 +751,7 @@ xfs_dir2_block_lookup_int(
         * No match, release the buffer and return ENOENT.
         */
        xfs_trans_brelse(tp, bp);
-        return XFS_ERROR(ENOENT);
+        return -ENOENT;
 }
 /*
@@ -1091,7 +1091,7 @@ xfs_dir2_sf_to_block(
         */
        if (dp->i_d.di_size < offsetof(xfs_dir2_sf_hdr_t, parent)) {
                ASSERT(XFS_FORCED_SHUTDOWN(mp));
-                return XFS_ERROR(EIO);
+                return -EIO;
        }
        oldsfp = (xfs_dir2_sf_hdr_t *)ifp->if_u1.if_data;
diff --git a/fs/xfs/xfs_dir2_data.c b/fs/xfs/libxfs/xfs_dir2_data.c
index 8c2f6422648e..fdd803fecb8e 100644
--- a/fs/xfs/xfs_dir2_data.c
+++ b/fs/xfs/libxfs/xfs_dir2_data.c
@@ -100,7 +100,7 @@ __xfs_dir3_data_check(
                break;
        default:
                XFS_ERROR_REPORT("Bad Magic", XFS_ERRLEVEL_LOW, mp);
-                return EFSCORRUPTED;
+                return -EFSCORRUPTED;
        }
        /*
@@ -256,7 +256,7 @@ xfs_dir3_data_reada_verify(
                xfs_dir3_data_verify(bp);
                return;
        default:
-                xfs_buf_ioerror(bp, EFSCORRUPTED);
+                xfs_buf_ioerror(bp, -EFSCORRUPTED);
                xfs_verifier_error(bp);
                break;
        }
@@ -270,9 +270,9 @@ xfs_dir3_data_read_verify(
        if (xfs_sb_version_hascrc(&mp->m_sb) &&
             !xfs_buf_verify_cksum(bp, XFS_DIR3_DATA_CRC_OFF))
-                 xfs_buf_ioerror(bp, EFSBADCRC);
+                 xfs_buf_ioerror(bp, -EFSBADCRC);
        else if (!xfs_dir3_data_verify(bp))
-                xfs_buf_ioerror(bp, EFSCORRUPTED);
+                xfs_buf_ioerror(bp, -EFSCORRUPTED);
        if (bp->b_error)
                xfs_verifier_error(bp);
@@ -287,7 +287,7 @@ xfs_dir3_data_write_verify(
        struct xfs_dir3_blk_hdr *hdr3 = bp->b_addr;
        if (!xfs_dir3_data_verify(bp)) {
-                xfs_buf_ioerror(bp, EFSCORRUPTED);
+                xfs_buf_ioerror(bp, -EFSCORRUPTED);
                xfs_verifier_error(bp);
                return;
        }
diff --git a/fs/xfs/xfs_dir2_leaf.c b/fs/xfs/libxfs/xfs_dir2_leaf.c
index fb0aad4440c1..a19174eb3cb2 100644
--- a/fs/xfs/xfs_dir2_leaf.c
+++ b/fs/xfs/libxfs/xfs_dir2_leaf.c
@@ -183,9 +183,9 @@ __read_verify(
        if (xfs_sb_version_hascrc(&mp->m_sb) &&
             !xfs_buf_verify_cksum(bp, XFS_DIR3_LEAF_CRC_OFF))
-                xfs_buf_ioerror(bp, EFSBADCRC);
+                xfs_buf_ioerror(bp, -EFSBADCRC);
        else if (!xfs_dir3_leaf_verify(bp, magic))
-                xfs_buf_ioerror(bp, EFSCORRUPTED);
+                xfs_buf_ioerror(bp, -EFSCORRUPTED);
        if (bp->b_error)
                xfs_verifier_error(bp);
@@ -201,7 +201,7 @@ __write_verify(
        struct xfs_dir3_leaf_hdr *hdr3 = bp->b_addr;
        if (!xfs_dir3_leaf_verify(bp, magic)) {
-                xfs_buf_ioerror(bp, EFSCORRUPTED);
+                xfs_buf_ioerror(bp, -EFSCORRUPTED);
                xfs_verifier_error(bp);
                return;
        }
@@ -731,7 +731,7 @@ xfs_dir2_leaf_addname(
                if ((args->op_flags & XFS_DA_OP_JUSTCHECK) ||
                                                        args->total == 0) {
                        xfs_trans_brelse(tp, lbp);
-                        return XFS_ERROR(ENOSPC);
+                        return -ENOSPC;
                }
                /*
                 * Convert to node form.
@@ -755,7 +755,7 @@ xfs_dir2_leaf_addname(
         */
        if (args->op_flags & XFS_DA_OP_JUSTCHECK) {
                xfs_trans_brelse(tp, lbp);
-                return use_block == -1 ? XFS_ERROR(ENOSPC) : 0;
+                return use_block == -1 ? -ENOSPC : 0;
        }
        /*
         * If no allocations are allowed, return now before we've
@@ -763,7 +763,7 @@ xfs_dir2_leaf_addname(
         */
        if (args->total == 0 && use_block == -1) {
                xfs_trans_brelse(tp, lbp);
-                return XFS_ERROR(ENOSPC);
+                return -ENOSPC;
        }
        /*
         * Need to compact the leaf entries, removing stale ones.
@@ -1198,7 +1198,7 @@ xfs_dir2_leaf_lookup(
        error = xfs_dir_cilookup_result(args, dep->name, dep->namelen);
        xfs_trans_brelse(tp, dbp);
        xfs_trans_brelse(tp, lbp);
-        return XFS_ERROR(error);
+        return error;
 }
 /*
@@ -1327,13 +1327,13 @@ xfs_dir2_leaf_lookup_int(
                return 0;
        }
        /*
-         * No match found, return ENOENT.
+         * No match found, return -ENOENT.
         */
        ASSERT(cidb == -1);
        if (dbp)
                xfs_trans_brelse(tp, dbp);
        xfs_trans_brelse(tp, lbp);
-        return XFS_ERROR(ENOENT);
+        return -ENOENT;
 }
 /*
@@ -1440,7 +1440,7 @@ xfs_dir2_leaf_removename(
                         * Just go on, returning success, leaving the
                         * empty block in place.
                         */
-                        if (error == ENOSPC && args->total == 0)
+                        if (error == -ENOSPC && args->total == 0)
                                error = 0;
                        xfs_dir3_leaf_check(dp, lbp);
                        return error;
@@ -1641,7 +1641,7 @@ xfs_dir2_leaf_trim_data(
         * Get rid of the data block.
         */
        if ((error = xfs_dir2_shrink_inode(args, db, dbp))) {
-                ASSERT(error != ENOSPC);
+                ASSERT(error != -ENOSPC);
                xfs_trans_brelse(tp, dbp);
                return error;
        }
@@ -1815,7 +1815,7 @@ xfs_dir2_node_to_leaf(
                 * punching out the middle of an extent, and this is an
                 * isolated block.
                 */
-                ASSERT(error != ENOSPC);
+                ASSERT(error != -ENOSPC);
                return error;
        }
        fbp = NULL;
diff --git a/fs/xfs/xfs_dir2_node.c b/fs/xfs/libxfs/xfs_dir2_node.c
index da43d304fca2..2ae6ac2c11ae 100644
--- a/fs/xfs/xfs_dir2_node.c
+++ b/fs/xfs/libxfs/xfs_dir2_node.c
@@ -117,9 +117,9 @@ xfs_dir3_free_read_verify(
        if (xfs_sb_version_hascrc(&mp->m_sb) &&
            !xfs_buf_verify_cksum(bp, XFS_DIR3_FREE_CRC_OFF))
-                xfs_buf_ioerror(bp, EFSBADCRC);
+                xfs_buf_ioerror(bp, -EFSBADCRC);
        else if (!xfs_dir3_free_verify(bp))
-                xfs_buf_ioerror(bp, EFSCORRUPTED);
+                xfs_buf_ioerror(bp, -EFSCORRUPTED);
        if (bp->b_error)
                xfs_verifier_error(bp);
@@ -134,7 +134,7 @@ xfs_dir3_free_write_verify(
        struct xfs_dir3_blk_hdr *hdr3 = bp->b_addr;
        if (!xfs_dir3_free_verify(bp)) {
-                xfs_buf_ioerror(bp, EFSCORRUPTED);
+                xfs_buf_ioerror(bp, -EFSCORRUPTED);
                xfs_verifier_error(bp);
                return;
        }
@@ -406,7 +406,7 @@ xfs_dir2_leafn_add(
         * into other peoples memory
         */
        if (index < 0)
-                return XFS_ERROR(EFSCORRUPTED);
+                return -EFSCORRUPTED;
        /*
         * If there are already the maximum number of leaf entries in
@@ -417,7 +417,7 @@ xfs_dir2_leafn_add(
        if (leafhdr.count == dp->d_ops->leaf_max_ents(args->geo)) {
                if (!leafhdr.stale)
-                        return XFS_ERROR(ENOSPC);
+                        return -ENOSPC;
                compact = leafhdr.stale > 1;
        } else
                compact = 0;
@@ -629,7 +629,7 @@ xfs_dir2_leafn_lookup_for_addname(
                                                        XFS_ERRLEVEL_LOW, mp);
                                if (curfdb != newfdb)
                                        xfs_trans_brelse(tp, curbp);
-                                return XFS_ERROR(EFSCORRUPTED);
+                                return -EFSCORRUPTED;
                        }
                        curfdb = newfdb;
                        if (be16_to_cpu(bests[fi]) >= length)
@@ -660,7 +660,7 @@ out:
         * Return the index, that will be the insertion point.
         */
        *indexp = index;
-        return XFS_ERROR(ENOENT);
+        return -ENOENT;
 }
 /*
@@ -789,7 +789,7 @@ xfs_dir2_leafn_lookup_for_entry(
                        curbp->b_ops = &xfs_dir3_data_buf_ops;
                        xfs_trans_buf_set_type(tp, curbp, XFS_BLFT_DIR_DATA_BUF);
                        if (cmp == XFS_CMP_EXACT)
-                                return XFS_ERROR(EEXIST);
+                                return -EEXIST;
                }
        }
        ASSERT(index == leafhdr.count || (args->op_flags & XFS_DA_OP_OKNOENT));
@@ -812,7 +812,7 @@ xfs_dir2_leafn_lookup_for_entry(
                state->extravalid = 0;
        }
        *indexp = index;
-        return XFS_ERROR(ENOENT);
+        return -ENOENT;
 }
 /*
@@ -1133,7 +1133,7 @@ xfs_dir3_data_block_free(
                if (error == 0) {
                        fbp = NULL;
                        logfree = 0;
-                } else if (error != ENOSPC || args->total != 0)
+                } else if (error != -ENOSPC || args->total != 0)
                        return error;
                /*
                 * It's possible to get ENOSPC if there is no
@@ -1287,7 +1287,7 @@ xfs_dir2_leafn_remove(
                         * In this case just drop the buffer and some one else
                         * will eventually get rid of the empty block.
                         */
-                        else if (!(error == ENOSPC && args->total == 0))
+                        else if (!(error == -ENOSPC && args->total == 0))
                                return error;
                }
                /*
@@ -1599,7 +1599,7 @@ xfs_dir2_node_addname(
        error = xfs_da3_node_lookup_int(state, &rval);
        if (error)
                rval = error;
-        if (rval != ENOENT) {
+        if (rval != -ENOENT) {
                goto done;
        }
        /*
@@ -1628,7 +1628,7 @@ xfs_dir2_node_addname(
                 * It didn't work, we need to split the leaf block.
                 */
                if (args->total == 0) {
-                        ASSERT(rval == ENOSPC);
+                        ASSERT(rval == -ENOSPC);
                        goto done;
                }
                /*
@@ -1815,7 +1815,7 @@ xfs_dir2_node_addname_int(
                 * Not allowed to allocate, return failure.
                 */
                if ((args->op_flags & XFS_DA_OP_JUSTCHECK) || args->total == 0)
-                        return XFS_ERROR(ENOSPC);
+                        return -ENOSPC;
                /*
                 * Allocate and initialize the new data block.
@@ -1876,7 +1876,7 @@ xfs_dir2_node_addname_int(
                                }
                                XFS_ERROR_REPORT("xfs_dir2_node_addname_int",
                                                 XFS_ERRLEVEL_LOW, mp);
-                                return XFS_ERROR(EFSCORRUPTED);
+                                return -EFSCORRUPTED;
                        }
                        /*
@@ -2042,8 +2042,8 @@ xfs_dir2_node_lookup(
        error = xfs_da3_node_lookup_int(state, &rval);
        if (error)
                rval = error;
-        else if (rval == ENOENT && args->cmpresult == XFS_CMP_CASE) {
+        else if (rval == -ENOENT && args->cmpresult == XFS_CMP_CASE) {
-                /* If a CI match, dup the actual name and return EEXIST */
+                /* If a CI match, dup the actual name and return -EEXIST */
                xfs_dir2_data_entry_t   *dep;
                dep = (xfs_dir2_data_entry_t *)
@@ -2096,7 +2096,7 @@ xfs_dir2_node_removename(
                goto out_free;
        /* Didn't find it, upper layer screwed up. */
-        if (rval != EEXIST) {
+        if (rval != -EEXIST) {
                error = rval;
                goto out_free;
        }
@@ -2169,7 +2169,7 @@ xfs_dir2_node_replace(
         * It should be found, since the vnodeops layer has looked it up
         * and locked it.  But paranoia is good.
         */
-        if (rval == EEXIST) {
+        if (rval == -EEXIST) {
                struct xfs_dir2_leaf_entry *ents;
                /*
                 * Find the leaf entry.
@@ -2272,7 +2272,7 @@ xfs_dir2_node_trim_free(
                 * space reservation, when breaking up an extent into two
                 * pieces.  This is the last block of an extent.
                 */
-                ASSERT(error != ENOSPC);
+                ASSERT(error != -ENOSPC);
                xfs_trans_brelse(tp, bp);
                return error;
        }
diff --git a/fs/xfs/xfs_dir2_priv.h b/fs/xfs/libxfs/xfs_dir2_priv.h
index 27ce0794d196..27ce0794d196 100644
--- a/fs/xfs/xfs_dir2_priv.h
+++ b/fs/xfs/libxfs/xfs_dir2_priv.h
diff --git a/fs/xfs/xfs_dir2_sf.c b/fs/xfs/libxfs/xfs_dir2_sf.c
index 53c3be619db5..5079e051ef08 100644
--- a/fs/xfs/xfs_dir2_sf.c
+++ b/fs/xfs/libxfs/xfs_dir2_sf.c
@@ -51,10 +51,9 @@ static void xfs_dir2_sf_check(xfs_da_args_t *args);
 #else
 #define xfs_dir2_sf_check(args)
 #endif /* DEBUG */
-#if XFS_BIG_INUMS
 static void xfs_dir2_sf_toino4(xfs_da_args_t *args);
 static void xfs_dir2_sf_toino8(xfs_da_args_t *args);
-#endif /* XFS_BIG_INUMS */
 /*
 * Given a block directory (dp/block), calculate its size as a shortform (sf)
@@ -117,10 +116,10 @@ xfs_dir2_block_sfsize(
                isdotdot =
                        dep->namelen == 2 &&
                        dep->name[0] == '.' && dep->name[1] == '.';
-#if XFS_BIG_INUMS
                if (!isdot)
                        i8count += be64_to_cpu(dep->inumber) > XFS_DIR2_MAX_SHORT_INUM;
-#endif
                /* take into account the file type field */
                if (!isdot && !isdotdot) {
                        count++;
@@ -251,7 +250,7 @@ xfs_dir2_block_to_sf(
        logflags = XFS_ILOG_CORE;
        error = xfs_dir2_shrink_inode(args, args->geo->datablk, bp);
        if (error) {
-                ASSERT(error != ENOSPC);
+                ASSERT(error != -ENOSPC);
                goto out;
        }
@@ -299,7 +298,7 @@ xfs_dir2_sf_addname(
        trace_xfs_dir2_sf_addname(args);
-        ASSERT(xfs_dir2_sf_lookup(args) == ENOENT);
+        ASSERT(xfs_dir2_sf_lookup(args) == -ENOENT);
        dp = args->dp;
        ASSERT(dp->i_df.if_flags & XFS_IFINLINE);
        /*
@@ -307,7 +306,7 @@ xfs_dir2_sf_addname(
         */
        if (dp->i_d.di_size < offsetof(xfs_dir2_sf_hdr_t, parent)) {
                ASSERT(XFS_FORCED_SHUTDOWN(dp->i_mount));
-                return XFS_ERROR(EIO);
+                return -EIO;
        }
        ASSERT(dp->i_df.if_bytes == dp->i_d.di_size);
        ASSERT(dp->i_df.if_u1.if_data != NULL);
@@ -318,7 +317,7 @@ xfs_dir2_sf_addname(
         */
        incr_isize = dp->d_ops->sf_entsize(sfp, args->namelen);
        objchange = 0;
-#if XFS_BIG_INUMS
        /*
         * Do we have to change to 8 byte inodes?
         */
@@ -332,7 +331,7 @@ xfs_dir2_sf_addname(
                         (uint)sizeof(xfs_dir2_ino4_t));
                objchange = 1;
        }
-#endif
        new_isize = (int)dp->i_d.di_size + incr_isize;
        /*
         * Won't fit as shortform any more (due to size),
@@ -345,7 +344,7 @@ xfs_dir2_sf_addname(
                 * Just checking or no space reservation, it doesn't fit.
                 */
                if ((args->op_flags & XFS_DA_OP_JUSTCHECK) || args->total == 0)
-                        return XFS_ERROR(ENOSPC);
+                        return -ENOSPC;
                /*
                 * Convert to block form then add the name.
                 */
@@ -370,10 +369,8 @@ xfs_dir2_sf_addname(
         */
        else {
                ASSERT(pick == 2);
-#if XFS_BIG_INUMS
                if (objchange)
                        xfs_dir2_sf_toino8(args);
-#endif
                xfs_dir2_sf_addname_hard(args, objchange, new_isize);
        }
        xfs_trans_log_inode(args->trans, dp, XFS_ILOG_CORE | XFS_ILOG_DDATA);
@@ -425,10 +422,8 @@ xfs_dir2_sf_addname_easy(
         * Update the header and inode.
         */
        sfp->count++;
-#if XFS_BIG_INUMS
        if (args->inumber > XFS_DIR2_MAX_SHORT_INUM)
                sfp->i8count++;
-#endif
        dp->i_d.di_size = new_isize;
        xfs_dir2_sf_check(args);
 }
@@ -516,10 +511,8 @@ xfs_dir2_sf_addname_hard(
        dp->d_ops->sf_put_ino(sfp, sfep, args->inumber);
        dp->d_ops->sf_put_ftype(sfep, args->filetype);
        sfp->count++;
-#if XFS_BIG_INUMS
        if (args->inumber > XFS_DIR2_MAX_SHORT_INUM && !objchange)
                sfp->i8count++;
-#endif
        /*
         * If there's more left to copy, do that.
         */
@@ -593,13 +586,8 @@ xfs_dir2_sf_addname_pick(
        /*
         * If changing the inode number size, do it the hard way.
         */
-#if XFS_BIG_INUMS
+        if (objchange)
-        if (objchange) {
                return 2;
-        }
-#else
-        ASSERT(objchange == 0);
-#endif
        /*
         * If it won't fit at the end then do it the hard way (use the hole).
         */
@@ -650,7 +638,6 @@ xfs_dir2_sf_check(
                ASSERT(dp->d_ops->sf_get_ftype(sfep) < XFS_DIR3_FT_MAX);
        }
        ASSERT(i8count == sfp->i8count);
-        ASSERT(XFS_BIG_INUMS || i8count == 0);
        ASSERT((char *)sfep - (char *)sfp == dp->i_d.di_size);
        ASSERT(offset +
               (sfp->count + 2) * (uint)sizeof(xfs_dir2_leaf_entry_t) +
@@ -738,7 +725,7 @@ xfs_dir2_sf_lookup(
         */
        if (dp->i_d.di_size < offsetof(xfs_dir2_sf_hdr_t, parent)) {
                ASSERT(XFS_FORCED_SHUTDOWN(dp->i_mount));
-                return XFS_ERROR(EIO);
+                return -EIO;
        }
        ASSERT(dp->i_df.if_bytes == dp->i_d.di_size);
        ASSERT(dp->i_df.if_u1.if_data != NULL);
@@ -751,7 +738,7 @@ xfs_dir2_sf_lookup(
                args->inumber = dp->i_ino;
                args->cmpresult = XFS_CMP_EXACT;
                args->filetype = XFS_DIR3_FT_DIR;
-                return XFS_ERROR(EEXIST);
+                return -EEXIST;
        }
        /*
         * Special case for ..
@@ -761,7 +748,7 @@ xfs_dir2_sf_lookup(
                args->inumber = dp->d_ops->sf_get_parent_ino(sfp);
                args->cmpresult = XFS_CMP_EXACT;
                args->filetype = XFS_DIR3_FT_DIR;
-                return XFS_ERROR(EEXIST);
+                return -EEXIST;
        }
        /*
         * Loop over all the entries trying to match ours.
@@ -781,20 +768,20 @@ xfs_dir2_sf_lookup(
                        args->inumber = dp->d_ops->sf_get_ino(sfp, sfep);
                        args->filetype = dp->d_ops->sf_get_ftype(sfep);
                        if (cmp == XFS_CMP_EXACT)
-                                return XFS_ERROR(EEXIST);
+                                return -EEXIST;
                        ci_sfep = sfep;
                }
        }
        ASSERT(args->op_flags & XFS_DA_OP_OKNOENT);
        /*
         * Here, we can only be doing a lookup (not a rename or replace).
-         * If a case-insensitive match was not found, return ENOENT.
+         * If a case-insensitive match was not found, return -ENOENT.
         */
        if (!ci_sfep)
-                return XFS_ERROR(ENOENT);
+                return -ENOENT;
        /* otherwise process the CI match as required by the caller */
        error = xfs_dir_cilookup_result(args, ci_sfep->name, ci_sfep->namelen);
-        return XFS_ERROR(error);
+        return error;
 }
 /*
@@ -824,7 +811,7 @@ xfs_dir2_sf_removename(
         */
        if (oldsize < offsetof(xfs_dir2_sf_hdr_t, parent)) {
                ASSERT(XFS_FORCED_SHUTDOWN(dp->i_mount));
-                return XFS_ERROR(EIO);
+                return -EIO;
        }
        ASSERT(dp->i_df.if_bytes == oldsize);
        ASSERT(dp->i_df.if_u1.if_data != NULL);
@@ -847,7 +834,7 @@ xfs_dir2_sf_removename(
         * Didn't find it.
         */
        if (i == sfp->count)
-                return XFS_ERROR(ENOENT);
+                return -ENOENT;
        /*
         * Calculate sizes.
         */
@@ -870,7 +857,6 @@ xfs_dir2_sf_removename(
         */
        xfs_idata_realloc(dp, newsize - oldsize, XFS_DATA_FORK);
        sfp = (xfs_dir2_sf_hdr_t *)dp->i_df.if_u1.if_data;
-#if XFS_BIG_INUMS
        /*
         * Are we changing inode number size?
         */
@@ -880,7 +866,6 @@ xfs_dir2_sf_removename(
                else
                        sfp->i8count--;
        }
-#endif
        xfs_dir2_sf_check(args);
        xfs_trans_log_inode(args->trans, dp, XFS_ILOG_CORE | XFS_ILOG_DDATA);
        return 0;
@@ -895,12 +880,8 @@ xfs_dir2_sf_replace(
 {
        xfs_inode_t             *dp;            /* incore directory inode */
        int                     i;              /* entry index */
-#if XFS_BIG_INUMS || defined(DEBUG)
        xfs_ino_t               ino=0;          /* entry old inode number */
-#endif
-#if XFS_BIG_INUMS
        int                     i8elevated;     /* sf_toino8 set i8count=1 */
-#endif
        xfs_dir2_sf_entry_t     *sfep;          /* shortform directory entry */
        xfs_dir2_sf_hdr_t       *sfp;           /* shortform structure */
@@ -914,13 +895,13 @@ xfs_dir2_sf_replace(
         */
        if (dp->i_d.di_size < offsetof(xfs_dir2_sf_hdr_t, parent)) {
                ASSERT(XFS_FORCED_SHUTDOWN(dp->i_mount));
-                return XFS_ERROR(EIO);
+                return -EIO;
        }
        ASSERT(dp->i_df.if_bytes == dp->i_d.di_size);
        ASSERT(dp->i_df.if_u1.if_data != NULL);
        sfp = (xfs_dir2_sf_hdr_t *)dp->i_df.if_u1.if_data;
        ASSERT(dp->i_d.di_size >= xfs_dir2_sf_hdr_size(sfp->i8count));
-#if XFS_BIG_INUMS
        /*
         * New inode number is large, and need to convert to 8-byte inodes.
         */
@@ -951,17 +932,15 @@ xfs_dir2_sf_replace(
                sfp = (xfs_dir2_sf_hdr_t *)dp->i_df.if_u1.if_data;
        } else
                i8elevated = 0;
-#endif
        ASSERT(args->namelen != 1 || args->name[0] != '.');
        /*
         * Replace ..'s entry.
         */
        if (args->namelen == 2 &&
            args->name[0] == '.' && args->name[1] == '.') {
-#if XFS_BIG_INUMS || defined(DEBUG)
                ino = dp->d_ops->sf_get_parent_ino(sfp);
                ASSERT(args->inumber != ino);
-#endif
                dp->d_ops->sf_put_parent_ino(sfp, args->inumber);
        }
        /*
@@ -972,10 +951,8 @@ xfs_dir2_sf_replace(
                     i++, sfep = dp->d_ops->sf_nextentry(sfp, sfep)) {
                        if (xfs_da_compname(args, sfep->name, sfep->namelen) ==
                                                                XFS_CMP_EXACT) {
-#if XFS_BIG_INUMS || defined(DEBUG)
                                ino = dp->d_ops->sf_get_ino(sfp, sfep);
                                ASSERT(args->inumber != ino);
-#endif
                                dp->d_ops->sf_put_ino(sfp, sfep, args->inumber);
                                dp->d_ops->sf_put_ftype(sfep, args->filetype);
                                break;
@@ -986,14 +963,11 @@ xfs_dir2_sf_replace(
                 */
                if (i == sfp->count) {
                        ASSERT(args->op_flags & XFS_DA_OP_OKNOENT);
-#if XFS_BIG_INUMS
                        if (i8elevated)
                                xfs_dir2_sf_toino4(args);
-#endif
+                        return -ENOENT;
-                        return XFS_ERROR(ENOENT);
                }
        }
-#if XFS_BIG_INUMS
        /*
         * See if the old number was large, the new number is small.
         */
@@ -1020,13 +994,11 @@ xfs_dir2_sf_replace(
                if (!i8elevated)
                        sfp->i8count++;
        }
-#endif
        xfs_dir2_sf_check(args);
        xfs_trans_log_inode(args->trans, dp, XFS_ILOG_DDATA);
        return 0;
 }
-#if XFS_BIG_INUMS
 /*
 * Convert from 8-byte inode numbers to 4-byte inode numbers.
 * The last 8-byte inode number is gone, but the count is still 1.
@@ -1181,4 +1153,3 @@ xfs_dir2_sf_toino8(
        dp->i_d.di_size = newsize;
        xfs_trans_log_inode(args->trans, dp, XFS_ILOG_CORE | XFS_ILOG_DDATA);
 }
-#endif  /* XFS_BIG_INUMS */
diff --git a/fs/xfs/xfs_dquot_buf.c b/fs/xfs/libxfs/xfs_dquot_buf.c
index c2ac0c611ad8..bb969337efc8 100644
--- a/fs/xfs/xfs_dquot_buf.c
+++ b/fs/xfs/libxfs/xfs_dquot_buf.c
@@ -257,9 +257,9 @@ xfs_dquot_buf_read_verify(
        struct xfs_mount        *mp = bp->b_target->bt_mount;
        if (!xfs_dquot_buf_verify_crc(mp, bp))
-                xfs_buf_ioerror(bp, EFSBADCRC);
+                xfs_buf_ioerror(bp, -EFSBADCRC);
        else if (!xfs_dquot_buf_verify(mp, bp))
-                xfs_buf_ioerror(bp, EFSCORRUPTED);
+                xfs_buf_ioerror(bp, -EFSCORRUPTED);
        if (bp->b_error)
                xfs_verifier_error(bp);
@@ -277,7 +277,7 @@ xfs_dquot_buf_write_verify(
        struct xfs_mount        *mp = bp->b_target->bt_mount;
        if (!xfs_dquot_buf_verify(mp, bp)) {
-                xfs_buf_ioerror(bp, EFSCORRUPTED);
+                xfs_buf_ioerror(bp, -EFSCORRUPTED);
                xfs_verifier_error(bp);
                return;
        }
diff --git a/fs/xfs/xfs_format.h b/fs/xfs/libxfs/xfs_format.h
index 34d85aca3058..7e42bba9a420 100644
--- a/fs/xfs/xfs_format.h
+++ b/fs/xfs/libxfs/xfs_format.h
@@ -68,11 +68,7 @@ struct xfs_ifork;
 #define XFS_RTLOBIT(w)  xfs_lowbit32(w)
 #define XFS_RTHIBIT(w)  xfs_highbit32(w)
-#if XFS_BIG_BLKNOS
 #define XFS_RTBLOCKLOG(b)       xfs_highbit64(b)
-#else
-#define XFS_RTBLOCKLOG(b)       xfs_highbit32(b)
-#endif
 /*
 * Dquot and dquot block format definitions
@@ -304,23 +300,15 @@ typedef struct xfs_bmbt_rec_host {
 * Values and macros for delayed-allocation startblock fields.
 */
 #define STARTBLOCKVALBITS       17
-#define STARTBLOCKMASKBITS      (15 + XFS_BIG_BLKNOS * 20)
+#define STARTBLOCKMASKBITS      (15 + 20)
-#define DSTARTBLOCKMASKBITS     (15 + 20)
 #define STARTBLOCKMASK          \
        (((((xfs_fsblock_t)1) << STARTBLOCKMASKBITS) - 1) << STARTBLOCKVALBITS)
-#define DSTARTBLOCKMASK         \
-        (((((xfs_dfsbno_t)1) << DSTARTBLOCKMASKBITS) - 1) << STARTBLOCKVALBITS)
 static inline int isnullstartblock(xfs_fsblock_t x)
 {
        return ((x) & STARTBLOCKMASK) == STARTBLOCKMASK;
 }
-static inline int isnulldstartblock(xfs_dfsbno_t x)
-{
-        return ((x) & DSTARTBLOCKMASK) == DSTARTBLOCKMASK;
-}
 static inline xfs_fsblock_t nullstartblock(int k)
 {
        ASSERT(k < (1 << STARTBLOCKVALBITS));
diff --git a/fs/xfs/xfs_ialloc.c b/fs/xfs/libxfs/xfs_ialloc.c
index 5960e5593fe0..b62771f1f4b5 100644
--- a/fs/xfs/xfs_ialloc.c
+++ b/fs/xfs/libxfs/xfs_ialloc.c
@@ -292,7 +292,7 @@ xfs_ialloc_inode_init(
                                         mp->m_bsize * blks_per_cluster,
                                         XBF_UNMAPPED);
                if (!fbuf)
-                        return ENOMEM;
+                        return -ENOMEM;
                /* Initialize the inode buffers and log them appropriately. */
                fbuf->b_ops = &xfs_inode_buf_ops;
@@ -380,7 +380,7 @@ xfs_ialloc_ag_alloc(
        newlen = args.mp->m_ialloc_inos;
        if (args.mp->m_maxicount &&
            args.mp->m_sb.sb_icount + newlen > args.mp->m_maxicount)
-                return XFS_ERROR(ENOSPC);
+                return -ENOSPC;
        args.minlen = args.maxlen = args.mp->m_ialloc_blks;
        /*
         * First try to allocate inodes contiguous with the last-allocated
@@ -1385,7 +1385,7 @@ xfs_dialloc(
                if (error) {
                        xfs_trans_brelse(tp, agbp);
-                        if (error != ENOSPC)
+                        if (error != -ENOSPC)
                                goto out_error;
                        xfs_perag_put(pag);
@@ -1416,7 +1416,7 @@ nextag:
                        agno = 0;
                if (agno == start_agno) {
                        *inop = NULLFSINO;
-                        return noroom ? ENOSPC : 0;
+                        return noroom ? -ENOSPC : 0;
                }
        }
@@ -1425,7 +1425,7 @@ out_alloc:
        return xfs_dialloc_ag(tp, agbp, parent, inop);
 out_error:
        xfs_perag_put(pag);
-        return XFS_ERROR(error);
+        return error;
 }
 STATIC int
@@ -1682,7 +1682,7 @@ xfs_difree(
                xfs_warn(mp, "%s: agno >= mp->m_sb.sb_agcount (%d >= %d).",
                        __func__, agno, mp->m_sb.sb_agcount);
                ASSERT(0);
-                return XFS_ERROR(EINVAL);
+                return -EINVAL;
        }
        agino = XFS_INO_TO_AGINO(mp, inode);
        if (inode != XFS_AGINO_TO_INO(mp, agno, agino))  {
@@ -1690,14 +1690,14 @@ xfs_difree(
                        __func__, (unsigned long long)inode,
                        (unsigned long long)XFS_AGINO_TO_INO(mp, agno, agino));
                ASSERT(0);
-                return XFS_ERROR(EINVAL);
+                return -EINVAL;
        }
        agbno = XFS_AGINO_TO_AGBNO(mp, agino);
        if (agbno >= mp->m_sb.sb_agblocks)  {
                xfs_warn(mp, "%s: agbno >= mp->m_sb.sb_agblocks (%d >= %d).",
                        __func__, agbno, mp->m_sb.sb_agblocks);
                ASSERT(0);
-                return XFS_ERROR(EINVAL);
+                return -EINVAL;
        }
        /*
         * Get the allocation group header.
@@ -1769,7 +1769,7 @@ xfs_imap_lookup(
                if (i)
                        error = xfs_inobt_get_rec(cur, &rec, &i);
                if (!error && i == 0)
-                        error = EINVAL;
+                        error = -EINVAL;
        }
        xfs_trans_brelse(tp, agbp);
@@ -1780,12 +1780,12 @@ xfs_imap_lookup(
        /* check that the returned record contains the required inode */
        if (rec.ir_startino > agino ||
            rec.ir_startino + mp->m_ialloc_inos <= agino)
-                return EINVAL;
+                return -EINVAL;
        /* for untrusted inodes check it is allocated first */
        if ((flags & XFS_IGET_UNTRUSTED) &&
            (rec.ir_free & XFS_INOBT_MASK(agino - rec.ir_startino)))
-                return EINVAL;
+                return -EINVAL;
        *chunk_agbno = XFS_AGINO_TO_AGBNO(mp, rec.ir_startino);
        *offset_agbno = agbno - *chunk_agbno;
@@ -1829,7 +1829,7 @@ xfs_imap(
                 * as they can be invalid without implying corruption.
                 */
                if (flags & XFS_IGET_UNTRUSTED)
-                        return XFS_ERROR(EINVAL);
+                        return -EINVAL;
                if (agno >= mp->m_sb.sb_agcount) {
                        xfs_alert(mp,
                                "%s: agno (%d) >= mp->m_sb.sb_agcount (%d)",
@@ -1849,7 +1849,7 @@ xfs_imap(
                }
                xfs_stack_trace();
 #endif /* DEBUG */
-                return XFS_ERROR(EINVAL);
+                return -EINVAL;
        }
        blks_per_cluster = xfs_icluster_size_fsb(mp);
@@ -1922,7 +1922,7 @@ out_map:
                        __func__, (unsigned long long) imap->im_blkno,
                        (unsigned long long) imap->im_len,
                        XFS_FSB_TO_BB(mp, mp->m_sb.sb_dblocks));
-                return XFS_ERROR(EINVAL);
+                return -EINVAL;
        }
        return 0;
 }
@@ -2072,11 +2072,11 @@ xfs_agi_read_verify(
        if (xfs_sb_version_hascrc(&mp->m_sb) &&
            !xfs_buf_verify_cksum(bp, XFS_AGI_CRC_OFF))
-                xfs_buf_ioerror(bp, EFSBADCRC);
+                xfs_buf_ioerror(bp, -EFSBADCRC);
        else if (XFS_TEST_ERROR(!xfs_agi_verify(bp), mp,
                                XFS_ERRTAG_IALLOC_READ_AGI,
                                XFS_RANDOM_IALLOC_READ_AGI))
-                xfs_buf_ioerror(bp, EFSCORRUPTED);
+                xfs_buf_ioerror(bp, -EFSCORRUPTED);
        if (bp->b_error)
                xfs_verifier_error(bp);
@@ -2090,7 +2090,7 @@ xfs_agi_write_verify(
        struct xfs_buf_log_item *bip = bp->b_fspriv;
        if (!xfs_agi_verify(bp)) {
-                xfs_buf_ioerror(bp, EFSCORRUPTED);
+                xfs_buf_ioerror(bp, -EFSCORRUPTED);
                xfs_verifier_error(bp);
                return;
        }
diff --git a/fs/xfs/xfs_ialloc.h b/fs/xfs/libxfs/xfs_ialloc.h
index 95ad1c002d60..95ad1c002d60 100644
--- a/fs/xfs/xfs_ialloc.h
+++ b/fs/xfs/libxfs/xfs_ialloc.h
diff --git a/fs/xfs/xfs_ialloc_btree.c b/fs/xfs/libxfs/xfs_ialloc_btree.c
index 726f83a681a5..c9b06f30fe86 100644
--- a/fs/xfs/xfs_ialloc_btree.c
+++ b/fs/xfs/libxfs/xfs_ialloc_btree.c
@@ -272,9 +272,9 @@ xfs_inobt_read_verify(
        struct xfs_buf  *bp)
 {
        if (!xfs_btree_sblock_verify_crc(bp))
-                xfs_buf_ioerror(bp, EFSBADCRC);
+                xfs_buf_ioerror(bp, -EFSBADCRC);
        else if (!xfs_inobt_verify(bp))
-                xfs_buf_ioerror(bp, EFSCORRUPTED);
+                xfs_buf_ioerror(bp, -EFSCORRUPTED);
        if (bp->b_error) {
                trace_xfs_btree_corrupt(bp, _RET_IP_);
@@ -288,7 +288,7 @@ xfs_inobt_write_verify(
 {
        if (!xfs_inobt_verify(bp)) {
                trace_xfs_btree_corrupt(bp, _RET_IP_);
-                xfs_buf_ioerror(bp, EFSCORRUPTED);
+                xfs_buf_ioerror(bp, -EFSCORRUPTED);
                xfs_verifier_error(bp);
                return;
        }
diff --git a/fs/xfs/xfs_ialloc_btree.h b/fs/xfs/libxfs/xfs_ialloc_btree.h
index d7ebea72c2d0..d7ebea72c2d0 100644
--- a/fs/xfs/xfs_ialloc_btree.h
+++ b/fs/xfs/libxfs/xfs_ialloc_btree.h
diff --git a/fs/xfs/xfs_inode_buf.c b/fs/xfs/libxfs/xfs_inode_buf.c
index cb35ae41d4a1..f18fd2da49f7 100644
--- a/fs/xfs/xfs_inode_buf.c
+++ b/fs/xfs/libxfs/xfs_inode_buf.c
@@ -101,7 +101,7 @@ xfs_inode_buf_verify(
                                return;
                        }
-                        xfs_buf_ioerror(bp, EFSCORRUPTED);
+                        xfs_buf_ioerror(bp, -EFSCORRUPTED);
                        xfs_verifier_error(bp);
 #ifdef DEBUG
                        xfs_alert(mp,
@@ -174,14 +174,14 @@ xfs_imap_to_bp(
                                   (int)imap->im_len, buf_flags, &bp,
                                   &xfs_inode_buf_ops);
        if (error) {
-                if (error == EAGAIN) {
+                if (error == -EAGAIN) {
                        ASSERT(buf_flags & XBF_TRYLOCK);
                        return error;
                }
-                if (error == EFSCORRUPTED &&
+                if (error == -EFSCORRUPTED &&
                    (iget_flags & XFS_IGET_UNTRUSTED))
-                        return XFS_ERROR(EINVAL);
+                        return -EINVAL;
                xfs_warn(mp, "%s: xfs_trans_read_buf() returned error %d.",
                        __func__, error);
@@ -390,7 +390,7 @@ xfs_iread(
                                __func__, ip->i_ino);
                XFS_CORRUPTION_ERROR(__func__, XFS_ERRLEVEL_LOW, mp, dip);
-                error = XFS_ERROR(EFSCORRUPTED);
+                error = -EFSCORRUPTED;
                goto out_brelse;
        }
diff --git a/fs/xfs/xfs_inode_buf.h b/fs/xfs/libxfs/xfs_inode_buf.h
index 9308c47f2a52..9308c47f2a52 100644
--- a/fs/xfs/xfs_inode_buf.h
+++ b/fs/xfs/libxfs/xfs_inode_buf.h
diff --git a/fs/xfs/xfs_inode_fork.c b/fs/xfs/libxfs/xfs_inode_fork.c
index b031e8d0d928..6a00f7fed69d 100644
--- a/fs/xfs/xfs_inode_fork.c
+++ b/fs/xfs/libxfs/xfs_inode_fork.c
@@ -102,7 +102,7 @@ xfs_iformat_fork(
                                be64_to_cpu(dip->di_nblocks));
                XFS_CORRUPTION_ERROR("xfs_iformat(1)", XFS_ERRLEVEL_LOW,
                                     ip->i_mount, dip);
-                return XFS_ERROR(EFSCORRUPTED);
+                return -EFSCORRUPTED;
        }
        if (unlikely(dip->di_forkoff > ip->i_mount->m_sb.sb_inodesize)) {
@@ -111,7 +111,7 @@ xfs_iformat_fork(
                        dip->di_forkoff);
                XFS_CORRUPTION_ERROR("xfs_iformat(2)", XFS_ERRLEVEL_LOW,
                                     ip->i_mount, dip);
-                return XFS_ERROR(EFSCORRUPTED);
+                return -EFSCORRUPTED;
        }
        if (unlikely((ip->i_d.di_flags & XFS_DIFLAG_REALTIME) &&
@@ -121,7 +121,7 @@ xfs_iformat_fork(
                        ip->i_ino);
                XFS_CORRUPTION_ERROR("xfs_iformat(realtime)",
                                     XFS_ERRLEVEL_LOW, ip->i_mount, dip);
-                return XFS_ERROR(EFSCORRUPTED);
+                return -EFSCORRUPTED;
        }
        switch (ip->i_d.di_mode & S_IFMT) {
@@ -132,7 +132,7 @@ xfs_iformat_fork(
                if (unlikely(dip->di_format != XFS_DINODE_FMT_DEV)) {
                        XFS_CORRUPTION_ERROR("xfs_iformat(3)", XFS_ERRLEVEL_LOW,
                                              ip->i_mount, dip);
-                        return XFS_ERROR(EFSCORRUPTED);
+                        return -EFSCORRUPTED;
                }
                ip->i_d.di_size = 0;
                ip->i_df.if_u2.if_rdev = xfs_dinode_get_rdev(dip);
@@ -153,7 +153,7 @@ xfs_iformat_fork(
                                XFS_CORRUPTION_ERROR("xfs_iformat(4)",
                                                     XFS_ERRLEVEL_LOW,
                                                     ip->i_mount, dip);
-                                return XFS_ERROR(EFSCORRUPTED);
+                                return -EFSCORRUPTED;
                        }
                        di_size = be64_to_cpu(dip->di_size);
@@ -166,7 +166,7 @@ xfs_iformat_fork(
                                XFS_CORRUPTION_ERROR("xfs_iformat(5)",
                                                     XFS_ERRLEVEL_LOW,
                                                     ip->i_mount, dip);
-                                return XFS_ERROR(EFSCORRUPTED);
+                                return -EFSCORRUPTED;
                        }
                        size = (int)di_size;
@@ -181,13 +181,13 @@ xfs_iformat_fork(
                default:
                        XFS_ERROR_REPORT("xfs_iformat(6)", XFS_ERRLEVEL_LOW,
                                         ip->i_mount);
-                        return XFS_ERROR(EFSCORRUPTED);
+                        return -EFSCORRUPTED;
                }
                break;
        default:
                XFS_ERROR_REPORT("xfs_iformat(7)", XFS_ERRLEVEL_LOW, ip->i_mount);
-                return XFS_ERROR(EFSCORRUPTED);
+                return -EFSCORRUPTED;
        }
        if (error) {
                return error;
@@ -211,7 +211,7 @@ xfs_iformat_fork(
                        XFS_CORRUPTION_ERROR("xfs_iformat(8)",
                                             XFS_ERRLEVEL_LOW,
                                             ip->i_mount, dip);
-                        return XFS_ERROR(EFSCORRUPTED);
+                        return -EFSCORRUPTED;
                }
                error = xfs_iformat_local(ip, dip, XFS_ATTR_FORK, size);
@@ -223,7 +223,7 @@ xfs_iformat_fork(
                error = xfs_iformat_btree(ip, dip, XFS_ATTR_FORK);
                break;
        default:
-                error = XFS_ERROR(EFSCORRUPTED);
+                error = -EFSCORRUPTED;
                break;
        }
        if (error) {
@@ -266,7 +266,7 @@ xfs_iformat_local(
                        XFS_DFORK_SIZE(dip, ip->i_mount, whichfork));
                XFS_CORRUPTION_ERROR("xfs_iformat_local", XFS_ERRLEVEL_LOW,
                                     ip->i_mount, dip);
-                return XFS_ERROR(EFSCORRUPTED);
+                return -EFSCORRUPTED;
        }
        ifp = XFS_IFORK_PTR(ip, whichfork);
        real_size = 0;
@@ -322,7 +322,7 @@ xfs_iformat_extents(
                        (unsigned long long) ip->i_ino, nex);
                XFS_CORRUPTION_ERROR("xfs_iformat_extents(1)", XFS_ERRLEVEL_LOW,
                                     ip->i_mount, dip);
-                return XFS_ERROR(EFSCORRUPTED);
+                return -EFSCORRUPTED;
        }
        ifp->if_real_bytes = 0;
@@ -350,7 +350,7 @@ xfs_iformat_extents(
                                        XFS_ERROR_REPORT("xfs_iformat_extents(2)",
                                                         XFS_ERRLEVEL_LOW,
                                                         ip->i_mount);
-                                        return XFS_ERROR(EFSCORRUPTED);
+                                        return -EFSCORRUPTED;
                                }
        }
        ifp->if_flags |= XFS_IFEXTENTS;
@@ -399,7 +399,7 @@ xfs_iformat_btree(
                                        (unsigned long long) ip->i_ino);
                XFS_CORRUPTION_ERROR("xfs_iformat_btree", XFS_ERRLEVEL_LOW,
                                         mp, dip);
-                return XFS_ERROR(EFSCORRUPTED);
+                return -EFSCORRUPTED;
        }
        ifp->if_broot_bytes = size;
@@ -436,7 +436,7 @@ xfs_iread_extents(
        if (unlikely(XFS_IFORK_FORMAT(ip, whichfork) != XFS_DINODE_FMT_BTREE)) {
                XFS_ERROR_REPORT("xfs_iread_extents", XFS_ERRLEVEL_LOW,
                                 ip->i_mount);
-                return XFS_ERROR(EFSCORRUPTED);
+                return -EFSCORRUPTED;
        }
        nextents = XFS_IFORK_NEXTENTS(ip, whichfork);
        ifp = XFS_IFORK_PTR(ip, whichfork);
@@ -528,7 +528,7 @@ xfs_iroot_realloc(
                ifp->if_broot_bytes = (int)new_size;
                ASSERT(XFS_BMAP_BMDR_SPACE(ifp->if_broot) <=
                        XFS_IFORK_SIZE(ip, whichfork));
-                memmove(np, op, cur_max * (uint)sizeof(xfs_dfsbno_t));
+                memmove(np, op, cur_max * (uint)sizeof(xfs_fsblock_t));
                return;
        }
@@ -575,7 +575,7 @@ xfs_iroot_realloc(
                                                     ifp->if_broot_bytes);
                np = (char *)XFS_BMAP_BROOT_PTR_ADDR(mp, new_broot, 1,
                                                     (int)new_size);
-                memcpy(np, op, new_max * (uint)sizeof(xfs_dfsbno_t));
+                memcpy(np, op, new_max * (uint)sizeof(xfs_fsblock_t));
        }
        kmem_free(ifp->if_broot);
        ifp->if_broot = new_broot;
@@ -1692,7 +1692,7 @@ xfs_iext_idx_to_irec(
        }
        *idxp = page_idx;
        *erp_idxp = erp_idx;
-        return(erp);
+        return erp;
 }
 /*
diff --git a/fs/xfs/xfs_inode_fork.h b/fs/xfs/libxfs/xfs_inode_fork.h
index 7d3b1ed6dcbe..7d3b1ed6dcbe 100644
--- a/fs/xfs/xfs_inode_fork.h
+++ b/fs/xfs/libxfs/xfs_inode_fork.h
diff --git a/fs/xfs/xfs_inum.h b/fs/xfs/libxfs/xfs_inum.h
index 90efdaf1706f..4ff2278e147a 100644
--- a/fs/xfs/xfs_inum.h
+++ b/fs/xfs/libxfs/xfs_inum.h
@@ -54,11 +54,7 @@ struct xfs_mount;
 #define XFS_OFFBNO_TO_AGINO(mp,b,o)     \
        ((xfs_agino_t)(((b) << XFS_INO_OFFSET_BITS(mp)) | (o)))
-#if XFS_BIG_INUMS
 #define XFS_MAXINUMBER          ((xfs_ino_t)((1ULL << 56) - 1ULL))
-#else
-#define XFS_MAXINUMBER          ((xfs_ino_t)((1ULL << 32) - 1ULL))
-#endif
 #define XFS_MAXINUMBER_32       ((xfs_ino_t)((1ULL << 32) - 1ULL))
 #endif  /* __XFS_INUM_H__ */
diff --git a/fs/xfs/xfs_log_format.h b/fs/xfs/libxfs/xfs_log_format.h
index f0969c77bdbe..aff12f2d4428 100644
--- a/fs/xfs/xfs_log_format.h
+++ b/fs/xfs/libxfs/xfs_log_format.h
@@ -380,7 +380,7 @@ typedef struct xfs_icdinode {
        xfs_ictimestamp_t di_mtime;     /* time last modified */
        xfs_ictimestamp_t di_ctime;     /* time created/inode modified */
        xfs_fsize_t     di_size;        /* number of bytes in file */
-        xfs_drfsbno_t   di_nblocks;     /* # of direct & btree blocks used */
+        xfs_rfsblock_t  di_nblocks;     /* # of direct & btree blocks used */
        xfs_extlen_t    di_extsize;     /* basic/minimum extent size for file */
        xfs_extnum_t    di_nextents;    /* number of extents in data fork */
        xfs_aextnum_t   di_anextents;   /* number of extents in attribute fork*/
@@ -516,7 +516,7 @@ xfs_blft_from_flags(struct xfs_buf_log_format *blf)
 * EFI/EFD log format definitions
 */
 typedef struct xfs_extent {
-        xfs_dfsbno_t    ext_start;
+        xfs_fsblock_t   ext_start;
        xfs_extlen_t    ext_len;
 } xfs_extent_t;
diff --git a/fs/xfs/xfs_log_recover.h b/fs/xfs/libxfs/xfs_log_recover.h
index 1c55ccbb379d..1c55ccbb379d 100644
--- a/fs/xfs/xfs_log_recover.h
+++ b/fs/xfs/libxfs/xfs_log_recover.h
diff --git a/fs/xfs/xfs_log_rlimit.c b/fs/xfs/libxfs/xfs_log_rlimit.c
index ee7e0e80246b..ee7e0e80246b 100644
--- a/fs/xfs/xfs_log_rlimit.c
+++ b/fs/xfs/libxfs/xfs_log_rlimit.c
diff --git a/fs/xfs/xfs_quota_defs.h b/fs/xfs/libxfs/xfs_quota_defs.h
index 137e20937077..1b0a08379759 100644
--- a/fs/xfs/xfs_quota_defs.h
+++ b/fs/xfs/libxfs/xfs_quota_defs.h
@@ -98,8 +98,6 @@ typedef __uint16_t	xfs_qwarncnt_t;
 #define XFS_IS_QUOTA_ON(mp)     ((mp)->m_qflags & (XFS_UQUOTA_ACTIVE | \
                                                   XFS_GQUOTA_ACTIVE | \
                                                   XFS_PQUOTA_ACTIVE))
-#define XFS_IS_OQUOTA_ON(mp)    ((mp)->m_qflags & (XFS_GQUOTA_ACTIVE | \
-                                                   XFS_PQUOTA_ACTIVE))
 #define XFS_IS_UQUOTA_ON(mp)    ((mp)->m_qflags & XFS_UQUOTA_ACTIVE)
 #define XFS_IS_GQUOTA_ON(mp)    ((mp)->m_qflags & XFS_GQUOTA_ACTIVE)
 #define XFS_IS_PQUOTA_ON(mp)    ((mp)->m_qflags & XFS_PQUOTA_ACTIVE)
diff --git a/fs/xfs/xfs_rtbitmap.c b/fs/xfs/libxfs/xfs_rtbitmap.c
index f4dd697cac08..f4dd697cac08 100644
--- a/fs/xfs/xfs_rtbitmap.c
+++ b/fs/xfs/libxfs/xfs_rtbitmap.c
diff --git a/fs/xfs/xfs_sb.c b/fs/xfs/libxfs/xfs_sb.c
index 7703fa6770ff..ad525a5623a4 100644
--- a/fs/xfs/xfs_sb.c
+++ b/fs/xfs/libxfs/xfs_sb.c
@@ -186,13 +186,13 @@ xfs_mount_validate_sb(
         */
        if (sbp->sb_magicnum != XFS_SB_MAGIC) {
                xfs_warn(mp, "bad magic number");
-                return XFS_ERROR(EWRONGFS);
+                return -EWRONGFS;
        }
        if (!xfs_sb_good_version(sbp)) {
                xfs_warn(mp, "bad version");
-                return XFS_ERROR(EWRONGFS);
+                return -EWRONGFS;
        }
        /*
@@ -220,7 +220,7 @@ xfs_mount_validate_sb(
                                xfs_warn(mp,
 "Attempted to mount read-only compatible filesystem read-write.\n"
 "Filesystem can only be safely mounted read only.");
-                                return XFS_ERROR(EINVAL);
+                                return -EINVAL;
                        }
                }
                if (xfs_sb_has_incompat_feature(sbp,
@@ -230,7 +230,7 @@ xfs_mount_validate_sb(
 "Filesystem can not be safely mounted by this kernel.",
                                (sbp->sb_features_incompat &
                                                XFS_SB_FEAT_INCOMPAT_UNKNOWN));
-                        return XFS_ERROR(EINVAL);
+                        return -EINVAL;
                }
        }
@@ -238,13 +238,13 @@ xfs_mount_validate_sb(
                if (sbp->sb_qflags & (XFS_OQUOTA_ENFD | XFS_OQUOTA_CHKD)) {
                        xfs_notice(mp,
                           "Version 5 of Super block has XFS_OQUOTA bits.");
-                        return XFS_ERROR(EFSCORRUPTED);
+                        return -EFSCORRUPTED;
                }
        } else if (sbp->sb_qflags & (XFS_PQUOTA_ENFD | XFS_GQUOTA_ENFD |
                                XFS_PQUOTA_CHKD | XFS_GQUOTA_CHKD)) {
                        xfs_notice(mp,
 "Superblock earlier than Version 5 has XFS_[PQ]UOTA_{ENFD|CHKD} bits.");
-                        return XFS_ERROR(EFSCORRUPTED);
+                        return -EFSCORRUPTED;
        }
        if (unlikely(
@@ -252,7 +252,7 @@ xfs_mount_validate_sb(
                xfs_warn(mp,
                "filesystem is marked as having an external log; "
                "specify logdev on the mount command line.");
-                return XFS_ERROR(EINVAL);
+                return -EINVAL;
        }
        if (unlikely(
@@ -260,7 +260,7 @@ xfs_mount_validate_sb(
                xfs_warn(mp,
                "filesystem is marked as having an internal log; "
                "do not specify logdev on the mount command line.");
-                return XFS_ERROR(EINVAL);
+                return -EINVAL;
        }
        /*
@@ -294,7 +294,7 @@ xfs_mount_validate_sb(
            sbp->sb_dblocks < XFS_MIN_DBLOCKS(sbp)                      ||
            sbp->sb_shared_vn != 0)) {
                xfs_notice(mp, "SB sanity check failed");
-                return XFS_ERROR(EFSCORRUPTED);
+                return -EFSCORRUPTED;
        }
        /*
@@ -305,7 +305,7 @@ xfs_mount_validate_sb(
                "File system with blocksize %d bytes. "
                "Only pagesize (%ld) or less will currently work.",
                                sbp->sb_blocksize, PAGE_SIZE);
-                return XFS_ERROR(ENOSYS);
+                return -ENOSYS;
        }
        /*
@@ -320,19 +320,19 @@ xfs_mount_validate_sb(
        default:
                xfs_warn(mp, "inode size of %d bytes not supported",
                                sbp->sb_inodesize);
-                return XFS_ERROR(ENOSYS);
+                return -ENOSYS;
        }
        if (xfs_sb_validate_fsb_count(sbp, sbp->sb_dblocks) ||
            xfs_sb_validate_fsb_count(sbp, sbp->sb_rblocks)) {
                xfs_warn(mp,
                "file system too large to be mounted on this system.");
-                return XFS_ERROR(EFBIG);
+                return -EFBIG;
        }
        if (check_inprogress && sbp->sb_inprogress) {
                xfs_warn(mp, "Offline file system operation in progress!");
-                return XFS_ERROR(EFSCORRUPTED);
+                return -EFSCORRUPTED;
        }
        return 0;
 }
@@ -386,10 +386,11 @@ xfs_sb_quota_from_disk(struct xfs_sb *sbp)
        }
 }
-void
+static void
-xfs_sb_from_disk(
+__xfs_sb_from_disk(
        struct xfs_sb   *to,
-        xfs_dsb_t       *from)
+        xfs_dsb_t       *from,
+        bool            convert_xquota)
 {
        to->sb_magicnum = be32_to_cpu(from->sb_magicnum);
        to->sb_blocksize = be32_to_cpu(from->sb_blocksize);
@@ -445,6 +446,17 @@ xfs_sb_from_disk(
        to->sb_pad = 0;
        to->sb_pquotino = be64_to_cpu(from->sb_pquotino);
        to->sb_lsn = be64_to_cpu(from->sb_lsn);
+        /* Convert on-disk flags to in-memory flags? */
+        if (convert_xquota)
+                xfs_sb_quota_from_disk(to);
+}
+void
+xfs_sb_from_disk(
+        struct xfs_sb   *to,
+        xfs_dsb_t       *from)
+{
+        __xfs_sb_from_disk(to, from, true);
 }
 static inline void
@@ -577,7 +589,11 @@ xfs_sb_verify(
        struct xfs_mount *mp = bp->b_target->bt_mount;
        struct xfs_sb   sb;
-        xfs_sb_from_disk(&sb, XFS_BUF_TO_SBP(bp));
+        /*
+         * Use call variant which doesn't convert quota flags from disk 
+         * format, because xfs_mount_validate_sb checks the on-disk flags.
+         */
+        __xfs_sb_from_disk(&sb, XFS_BUF_TO_SBP(bp), false);
        /*
         * Only check the in progress field for the primary superblock as
@@ -620,7 +636,7 @@ xfs_sb_read_verify(
                        /* Only fail bad secondaries on a known V5 filesystem */
                        if (bp->b_bn == XFS_SB_DADDR ||
                            xfs_sb_version_hascrc(&mp->m_sb)) {
-                                error = EFSBADCRC;
+                                error = -EFSBADCRC;
                                goto out_error;
                        }
                }
@@ -630,7 +646,7 @@ xfs_sb_read_verify(
 out_error:
        if (error) {
                xfs_buf_ioerror(bp, error);
-                if (error == EFSCORRUPTED || error == EFSBADCRC)
+                if (error == -EFSCORRUPTED || error == -EFSBADCRC)
                        xfs_verifier_error(bp);
        }
 }
@@ -653,7 +669,7 @@ xfs_sb_quiet_read_verify(
                return;
        }
        /* quietly fail */
-        xfs_buf_ioerror(bp, EWRONGFS);
+        xfs_buf_ioerror(bp, -EWRONGFS);
 }
 static void
diff --git a/fs/xfs/xfs_sb.h b/fs/xfs/libxfs/xfs_sb.h
index c43c2d609a24..2e739708afd3 100644
--- a/fs/xfs/xfs_sb.h
+++ b/fs/xfs/libxfs/xfs_sb.h
@@ -87,11 +87,11 @@ struct xfs_trans;
 typedef struct xfs_sb {
        __uint32_t      sb_magicnum;    /* magic number == XFS_SB_MAGIC */
        __uint32_t      sb_blocksize;   /* logical block size, bytes */
-        xfs_drfsbno_t   sb_dblocks;     /* number of data blocks */
+        xfs_rfsblock_t  sb_dblocks;     /* number of data blocks */
-        xfs_drfsbno_t   sb_rblocks;     /* number of realtime blocks */
+        xfs_rfsblock_t  sb_rblocks;     /* number of realtime blocks */
-        xfs_drtbno_t    sb_rextents;    /* number of realtime extents */
+        xfs_rtblock_t   sb_rextents;    /* number of realtime extents */
        uuid_t          sb_uuid;        /* file system unique id */
-        xfs_dfsbno_t    sb_logstart;    /* starting block of log if internal */
+        xfs_fsblock_t   sb_logstart;    /* starting block of log if internal */
        xfs_ino_t       sb_rootino;     /* root inode number */
        xfs_ino_t       sb_rbmino;      /* bitmap inode for realtime extents */
        xfs_ino_t       sb_rsumino;     /* summary inode for rt bitmap */
diff --git a/fs/xfs/xfs_shared.h b/fs/xfs/libxfs/xfs_shared.h
index 82404da2ca67..82404da2ca67 100644
--- a/fs/xfs/xfs_shared.h
+++ b/fs/xfs/libxfs/xfs_shared.h
diff --git a/fs/xfs/xfs_symlink_remote.c b/fs/xfs/libxfs/xfs_symlink_remote.c
index 23c2f2577c8d..5782f037eab4 100644
--- a/fs/xfs/xfs_symlink_remote.c
+++ b/fs/xfs/libxfs/xfs_symlink_remote.c
@@ -133,9 +133,9 @@ xfs_symlink_read_verify(
                return;
        if (!xfs_buf_verify_cksum(bp, XFS_SYMLINK_CRC_OFF))
-                xfs_buf_ioerror(bp, EFSBADCRC);
+                xfs_buf_ioerror(bp, -EFSBADCRC);
        else if (!xfs_symlink_verify(bp))
-                xfs_buf_ioerror(bp, EFSCORRUPTED);
+                xfs_buf_ioerror(bp, -EFSCORRUPTED);
        if (bp->b_error)
                xfs_verifier_error(bp);
@@ -153,7 +153,7 @@ xfs_symlink_write_verify(
                return;
        if (!xfs_symlink_verify(bp)) {
-                xfs_buf_ioerror(bp, EFSCORRUPTED);
+                xfs_buf_ioerror(bp, -EFSCORRUPTED);
                xfs_verifier_error(bp);
                return;
        }
diff --git a/fs/xfs/xfs_trans_resv.c b/fs/xfs/libxfs/xfs_trans_resv.c
index f2bda7c76b8a..f2bda7c76b8a 100644
--- a/fs/xfs/xfs_trans_resv.c
+++ b/fs/xfs/libxfs/xfs_trans_resv.c
diff --git a/fs/xfs/xfs_trans_resv.h b/fs/xfs/libxfs/xfs_trans_resv.h
index 1097d14cd583..1097d14cd583 100644
--- a/fs/xfs/xfs_trans_resv.h
+++ b/fs/xfs/libxfs/xfs_trans_resv.h
diff --git a/fs/xfs/xfs_trans_space.h b/fs/xfs/libxfs/xfs_trans_space.h
index bf9c4579334d..bf9c4579334d 100644
--- a/fs/xfs/xfs_trans_space.h
+++ b/fs/xfs/libxfs/xfs_trans_space.h
diff --git a/fs/xfs/xfs_acl.c b/fs/xfs/xfs_acl.c
index 6888ad886ff6..a65fa5dde6e9 100644
--- a/fs/xfs/xfs_acl.c
+++ b/fs/xfs/xfs_acl.c
@@ -152,7 +152,7 @@ xfs_get_acl(struct inode *inode, int type)
        if (!xfs_acl)
                return ERR_PTR(-ENOMEM);
-        error = -xfs_attr_get(ip, ea_name, (unsigned char *)xfs_acl,
+        error = xfs_attr_get(ip, ea_name, (unsigned char *)xfs_acl,
                                                        &len, ATTR_ROOT);
        if (error) {
                /*
@@ -210,7 +210,7 @@ __xfs_set_acl(struct inode *inode, int type, struct posix_acl *acl)
                len -= sizeof(struct xfs_acl_entry) *
                         (XFS_ACL_MAX_ENTRIES(ip->i_mount) - acl->a_count);
-                error = -xfs_attr_set(ip, ea_name, (unsigned char *)xfs_acl,
+                error = xfs_attr_set(ip, ea_name, (unsigned char *)xfs_acl,
                                len, ATTR_ROOT);
                kmem_free(xfs_acl);
@@ -218,7 +218,7 @@ __xfs_set_acl(struct inode *inode, int type, struct posix_acl *acl)
                /*
                 * A NULL ACL argument means we want to remove the ACL.
                 */
-                error = -xfs_attr_remove(ip, ea_name, ATTR_ROOT);
+                error = xfs_attr_remove(ip, ea_name, ATTR_ROOT);
                /*
                 * If the attribute didn't exist to start with that's fine.
@@ -244,7 +244,7 @@ xfs_set_mode(struct inode *inode, umode_t mode)
                iattr.ia_mode = mode;
                iattr.ia_ctime = current_fs_time(inode->i_sb);
-                error = -xfs_setattr_nonsize(XFS_I(inode), &iattr, XFS_ATTR_NOACL);
+                error = xfs_setattr_nonsize(XFS_I(inode), &iattr, XFS_ATTR_NOACL);
        }
        return error;
diff --git a/fs/xfs/xfs_aops.c b/fs/xfs/xfs_aops.c
index faaf716e2080..11e9b4caa54f 100644
--- a/fs/xfs/xfs_aops.c
+++ b/fs/xfs/xfs_aops.c
@@ -240,7 +240,7 @@ xfs_end_io(
 done:
        if (error)
-                ioend->io_error = -error;
+                ioend->io_error = error;
        xfs_destroy_ioend(ioend);
 }
@@ -308,14 +308,14 @@ xfs_map_blocks(
        int                     nimaps = 1;
        if (XFS_FORCED_SHUTDOWN(mp))
-                return -XFS_ERROR(EIO);
+                return -EIO;
        if (type == XFS_IO_UNWRITTEN)
                bmapi_flags |= XFS_BMAPI_IGSTATE;
        if (!xfs_ilock_nowait(ip, XFS_ILOCK_SHARED)) {
                if (nonblocking)
-                        return -XFS_ERROR(EAGAIN);
+                        return -EAGAIN;
                xfs_ilock(ip, XFS_ILOCK_SHARED);
        }
@@ -332,14 +332,14 @@ xfs_map_blocks(
        xfs_iunlock(ip, XFS_ILOCK_SHARED);
        if (error)
-                return -XFS_ERROR(error);
+                return error;
        if (type == XFS_IO_DELALLOC &&
            (!nimaps || isnullstartblock(imap->br_startblock))) {
                error = xfs_iomap_write_allocate(ip, offset, imap);
                if (!error)
                        trace_xfs_map_blocks_alloc(ip, offset, count, type, imap);
-                return -XFS_ERROR(error);
+                return error;
        }
 #ifdef DEBUG
@@ -502,7 +502,7 @@ xfs_submit_ioend(
                 * time.
                 */
                if (fail) {
-                        ioend->io_error = -fail;
+                        ioend->io_error = fail;
                        xfs_finish_ioend(ioend);
                        continue;
                }
@@ -1253,7 +1253,7 @@ __xfs_get_blocks(
        int                     new = 0;
        if (XFS_FORCED_SHUTDOWN(mp))
-                return -XFS_ERROR(EIO);
+                return -EIO;
        offset = (xfs_off_t)iblock << inode->i_blkbits;
        ASSERT(bh_result->b_size >= (1 << inode->i_blkbits));
@@ -1302,7 +1302,7 @@ __xfs_get_blocks(
                        error = xfs_iomap_write_direct(ip, offset, size,
                                                       &imap, nimaps);
                        if (error)
-                                return -error;
+                                return error;
                        new = 1;
                } else {
                        /*
@@ -1415,7 +1415,7 @@ __xfs_get_blocks(
 out_unlock:
        xfs_iunlock(ip, lockmode);
-        return -error;
+        return error;
 }
 int
diff --git a/fs/xfs/xfs_attr_inactive.c b/fs/xfs/xfs_attr_inactive.c
index 09480c57f069..aa2a8b1838a2 100644
--- a/fs/xfs/xfs_attr_inactive.c
+++ b/fs/xfs/xfs_attr_inactive.c
@@ -76,7 +76,7 @@ xfs_attr3_leaf_freextent(
                error = xfs_bmapi_read(dp, (xfs_fileoff_t)tblkno, tblkcnt,
                                       &map, &nmap, XFS_BMAPI_ATTRFORK);
                if (error) {
-                        return(error);
+                        return error;
                }
                ASSERT(nmap == 1);
                ASSERT(map.br_startblock != DELAYSTARTBLOCK);
@@ -95,21 +95,21 @@ xfs_attr3_leaf_freextent(
                                        dp->i_mount->m_ddev_targp,
                                        dblkno, dblkcnt, 0);
                        if (!bp)
-                                return ENOMEM;
+                                return -ENOMEM;
                        xfs_trans_binval(*trans, bp);
                        /*
                         * Roll to next transaction.
                         */
                        error = xfs_trans_roll(trans, dp);
                        if (error)
-                                return (error);
+                                return error;
                }
                tblkno += map.br_blockcount;
                tblkcnt -= map.br_blockcount;
        }
-        return(0);
+        return 0;
 }
 /*
@@ -227,7 +227,7 @@ xfs_attr3_node_inactive(
         */
        if (level > XFS_DA_NODE_MAXDEPTH) {
                xfs_trans_brelse(*trans, bp);   /* no locks for later trans */
-                return XFS_ERROR(EIO);
+                return -EIO;
        }
        node = bp->b_addr;
@@ -256,7 +256,7 @@ xfs_attr3_node_inactive(
                error = xfs_da3_node_read(*trans, dp, child_fsb, -2, &child_bp,
                                                XFS_ATTR_FORK);
                if (error)
-                        return(error);
+                        return error;
                if (child_bp) {
                                                /* save for re-read later */
                        child_blkno = XFS_BUF_ADDR(child_bp);
@@ -277,7 +277,7 @@ xfs_attr3_node_inactive(
                                                        child_bp);
                                break;
                        default:
-                                error = XFS_ERROR(EIO);
+                                error = -EIO;
                                xfs_trans_brelse(*trans, child_bp);
                                break;
                        }
@@ -360,7 +360,7 @@ xfs_attr3_root_inactive(
                error = xfs_attr3_leaf_inactive(trans, dp, bp);
                break;
        default:
-                error = XFS_ERROR(EIO);
+                error = -EIO;
                xfs_trans_brelse(*trans, bp);
                break;
        }
@@ -414,7 +414,7 @@ xfs_attr_inactive(xfs_inode_t *dp)
        error = xfs_trans_reserve(trans, &M_RES(mp)->tr_attrinval, 0, 0);
        if (error) {
                xfs_trans_cancel(trans, 0);
-                return(error);
+                return error;
        }
        xfs_ilock(dp, XFS_ILOCK_EXCL);
@@ -443,10 +443,10 @@ xfs_attr_inactive(xfs_inode_t *dp)
        error = xfs_trans_commit(trans, XFS_TRANS_RELEASE_LOG_RES);
        xfs_iunlock(dp, XFS_ILOCK_EXCL);
-        return(error);
+        return error;
 out:
        xfs_trans_cancel(trans, XFS_TRANS_RELEASE_LOG_RES|XFS_TRANS_ABORT);
        xfs_iunlock(dp, XFS_ILOCK_EXCL);
-        return(error);
+        return error;
 }
diff --git a/fs/xfs/xfs_attr_list.c b/fs/xfs/xfs_attr_list.c
index 90e2eeb21207..62db83ab6cbc 100644
--- a/fs/xfs/xfs_attr_list.c
+++ b/fs/xfs/xfs_attr_list.c
@@ -50,11 +50,11 @@ xfs_attr_shortform_compare(const void *a, const void *b)
        sa = (xfs_attr_sf_sort_t *)a;
        sb = (xfs_attr_sf_sort_t *)b;
        if (sa->hash < sb->hash) {
-                return(-1);
+                return -1;
        } else if (sa->hash > sb->hash) {
-                return(1);
+                return 1;
        } else {
-                return(sa->entno - sb->entno);
+                return sa->entno - sb->entno;
        }
 }
@@ -86,7 +86,7 @@ xfs_attr_shortform_list(xfs_attr_list_context_t *context)
        sf = (xfs_attr_shortform_t *)dp->i_afp->if_u1.if_data;
        ASSERT(sf != NULL);
        if (!sf->hdr.count)
-                return(0);
+                return 0;
        cursor = context->cursor;
        ASSERT(cursor != NULL);
@@ -124,7 +124,7 @@ xfs_attr_shortform_list(xfs_attr_list_context_t *context)
                        sfe = XFS_ATTR_SF_NEXTENTRY(sfe);
                }
                trace_xfs_attr_list_sf_all(context);
-                return(0);
+                return 0;
        }
        /* do no more for a search callback */
@@ -150,7 +150,7 @@ xfs_attr_shortform_list(xfs_attr_list_context_t *context)
                                             XFS_ERRLEVEL_LOW,
                                             context->dp->i_mount, sfe);
                        kmem_free(sbuf);
-                        return XFS_ERROR(EFSCORRUPTED);
+                        return -EFSCORRUPTED;
                }
                sbp->entno = i;
@@ -188,7 +188,7 @@ xfs_attr_shortform_list(xfs_attr_list_context_t *context)
        }
        if (i == nsbuf) {
                kmem_free(sbuf);
-                return(0);
+                return 0;
        }
        /*
@@ -213,7 +213,7 @@ xfs_attr_shortform_list(xfs_attr_list_context_t *context)
        }
        kmem_free(sbuf);
-        return(0);
+        return 0;
 }
 STATIC int
@@ -243,8 +243,8 @@ xfs_attr_node_list(xfs_attr_list_context_t *context)
        if (cursor->blkno > 0) {
                error = xfs_da3_node_read(NULL, dp, cursor->blkno, -1,
                                              &bp, XFS_ATTR_FORK);
-                if ((error != 0) && (error != EFSCORRUPTED))
+                if ((error != 0) && (error != -EFSCORRUPTED))
-                        return(error);
+                        return error;
                if (bp) {
                        struct xfs_attr_leaf_entry *entries;
@@ -295,7 +295,7 @@ xfs_attr_node_list(xfs_attr_list_context_t *context)
                                                      cursor->blkno, -1, &bp,
                                                      XFS_ATTR_FORK);
                        if (error)
-                                return(error);
+                                return error;
                        node = bp->b_addr;
                        magic = be16_to_cpu(node->hdr.info.magic);
                        if (magic == XFS_ATTR_LEAF_MAGIC ||
@@ -308,7 +308,7 @@ xfs_attr_node_list(xfs_attr_list_context_t *context)
                                                     context->dp->i_mount,
                                                     node);
                                xfs_trans_brelse(NULL, bp);
-                                return XFS_ERROR(EFSCORRUPTED);
+                                return -EFSCORRUPTED;
                        }
                        dp->d_ops->node_hdr_from_disk(&nodehdr, node);
@@ -496,11 +496,11 @@ xfs_attr_leaf_list(xfs_attr_list_context_t *context)
        context->cursor->blkno = 0;
        error = xfs_attr3_leaf_read(NULL, context->dp, 0, -1, &bp);
        if (error)
-                return XFS_ERROR(error);
+                return error;
        error = xfs_attr3_leaf_list_int(bp, context);
        xfs_trans_brelse(NULL, bp);
-        return XFS_ERROR(error);
+        return error;
 }
 int
@@ -514,7 +514,7 @@ xfs_attr_list_int(
        XFS_STATS_INC(xs_attr_list);
        if (XFS_FORCED_SHUTDOWN(dp->i_mount))
-                return EIO;
+                return -EIO;
        /*
         * Decide on what work routines to call based on the inode size.
@@ -616,16 +616,16 @@ xfs_attr_list(
         * Validate the cursor.
         */
        if (cursor->pad1 || cursor->pad2)
-                return(XFS_ERROR(EINVAL));
+                return -EINVAL;
        if ((cursor->initted == 0) &&
            (cursor->hashval || cursor->blkno || cursor->offset))
-                return XFS_ERROR(EINVAL);
+                return -EINVAL;
        /*
         * Check for a properly aligned buffer.
         */
        if (((long)buffer) & (sizeof(int)-1))
-                return XFS_ERROR(EFAULT);
+                return -EFAULT;
        if (flags & ATTR_KERNOVAL)
                bufsize = 0;
@@ -648,6 +648,6 @@ xfs_attr_list(
        alist->al_offset[0] = context.bufsize;
        error = xfs_attr_list_int(&context);
-        ASSERT(error >= 0);
+        ASSERT(error <= 0);
        return error;
 }
diff --git a/fs/xfs/xfs_bmap_util.c b/fs/xfs/xfs_bmap_util.c
index 64731ef3324d..2f1e30d39a35 100644
--- a/fs/xfs/xfs_bmap_util.c
+++ b/fs/xfs/xfs_bmap_util.c
@@ -133,7 +133,7 @@ xfs_bmap_finish(
                        mp = ntp->t_mountp;
                        if (!XFS_FORCED_SHUTDOWN(mp))
                                xfs_force_shutdown(mp,
-                                                   (error == EFSCORRUPTED) ?
+                                                   (error == -EFSCORRUPTED) ?
                                                   SHUTDOWN_CORRUPT_INCORE :
                                                   SHUTDOWN_META_IO_ERROR);
                        return error;
@@ -365,7 +365,7 @@ xfs_bmap_count_tree(
                        xfs_trans_brelse(tp, bp);
                        XFS_ERROR_REPORT("xfs_bmap_count_tree(1)",
                                         XFS_ERRLEVEL_LOW, mp);
-                        return XFS_ERROR(EFSCORRUPTED);
+                        return -EFSCORRUPTED;
                }
                xfs_trans_brelse(tp, bp);
        } else {
@@ -425,14 +425,14 @@ xfs_bmap_count_blocks(
        ASSERT(level > 0);
        pp = XFS_BMAP_BROOT_PTR_ADDR(mp, block, 1, ifp->if_broot_bytes);
        bno = be64_to_cpu(*pp);
-        ASSERT(bno != NULLDFSBNO);
+        ASSERT(bno != NULLFSBLOCK);
        ASSERT(XFS_FSB_TO_AGNO(mp, bno) < mp->m_sb.sb_agcount);
        ASSERT(XFS_FSB_TO_AGBNO(mp, bno) < mp->m_sb.sb_agblocks);
        if (unlikely(xfs_bmap_count_tree(mp, tp, ifp, bno, level, count) < 0)) {
                XFS_ERROR_REPORT("xfs_bmap_count_blocks(2)", XFS_ERRLEVEL_LOW,
                                 mp);
-                return XFS_ERROR(EFSCORRUPTED);
+                return -EFSCORRUPTED;
        }
        return 0;
@@ -524,13 +524,13 @@ xfs_getbmap(
                        if (ip->i_d.di_aformat != XFS_DINODE_FMT_EXTENTS &&
                            ip->i_d.di_aformat != XFS_DINODE_FMT_BTREE &&
                            ip->i_d.di_aformat != XFS_DINODE_FMT_LOCAL)
-                                return XFS_ERROR(EINVAL);
+                                return -EINVAL;
                } else if (unlikely(
                           ip->i_d.di_aformat != 0 &&
                           ip->i_d.di_aformat != XFS_DINODE_FMT_EXTENTS)) {
                        XFS_ERROR_REPORT("xfs_getbmap", XFS_ERRLEVEL_LOW,
                                         ip->i_mount);
-                        return XFS_ERROR(EFSCORRUPTED);
+                        return -EFSCORRUPTED;
                }
                prealloced = 0;
@@ -539,7 +539,7 @@ xfs_getbmap(
                if (ip->i_d.di_format != XFS_DINODE_FMT_EXTENTS &&
                    ip->i_d.di_format != XFS_DINODE_FMT_BTREE &&
                    ip->i_d.di_format != XFS_DINODE_FMT_LOCAL)
-                        return XFS_ERROR(EINVAL);
+                        return -EINVAL;
                if (xfs_get_extsz_hint(ip) ||
                    ip->i_d.di_flags & (XFS_DIFLAG_PREALLOC|XFS_DIFLAG_APPEND)){
@@ -559,26 +559,26 @@ xfs_getbmap(
                bmv->bmv_entries = 0;
                return 0;
        } else if (bmv->bmv_length < 0) {
-                return XFS_ERROR(EINVAL);
+                return -EINVAL;
        }
        nex = bmv->bmv_count - 1;
        if (nex <= 0)
-                return XFS_ERROR(EINVAL);
+                return -EINVAL;
        bmvend = bmv->bmv_offset + bmv->bmv_length;
        if (bmv->bmv_count > ULONG_MAX / sizeof(struct getbmapx))
-                return XFS_ERROR(ENOMEM);
+                return -ENOMEM;
        out = kmem_zalloc_large(bmv->bmv_count * sizeof(struct getbmapx), 0);
        if (!out)
-                return XFS_ERROR(ENOMEM);
+                return -ENOMEM;
        xfs_ilock(ip, XFS_IOLOCK_SHARED);
        if (whichfork == XFS_DATA_FORK) {
                if (!(iflags & BMV_IF_DELALLOC) &&
                    (ip->i_delayed_blks || XFS_ISIZE(ip) > ip->i_d.di_size)) {
-                        error = -filemap_write_and_wait(VFS_I(ip)->i_mapping);
+                        error = filemap_write_and_wait(VFS_I(ip)->i_mapping);
                        if (error)
                                goto out_unlock_iolock;
@@ -611,7 +611,7 @@ xfs_getbmap(
        /*
         * Allocate enough space to handle "subnex" maps at a time.
         */
-        error = ENOMEM;
+        error = -ENOMEM;
        subnex = 16;
        map = kmem_alloc(subnex * sizeof(*map), KM_MAYFAIL | KM_NOFS);
        if (!map)
@@ -809,7 +809,7 @@ xfs_can_free_eofblocks(struct xfs_inode *ip, bool force)
         * have speculative prealloc/delalloc blocks to remove.
         */
        if (VFS_I(ip)->i_size == 0 &&
-            VN_CACHED(VFS_I(ip)) == 0 &&
+            VFS_I(ip)->i_mapping->nrpages == 0 &&
            ip->i_delayed_blks == 0)
                return false;
@@ -882,7 +882,7 @@ xfs_free_eofblocks(
                if (need_iolock) {
                        if (!xfs_ilock_nowait(ip, XFS_IOLOCK_EXCL)) {
                                xfs_trans_cancel(tp, 0);
-                                return EAGAIN;
+                                return -EAGAIN;
                        }
                }
@@ -955,14 +955,14 @@ xfs_alloc_file_space(
        trace_xfs_alloc_file_space(ip);
        if (XFS_FORCED_SHUTDOWN(mp))
-                return XFS_ERROR(EIO);
+                return -EIO;
        error = xfs_qm_dqattach(ip, 0);
        if (error)
                return error;
        if (len <= 0)
-                return XFS_ERROR(EINVAL);
+                return -EINVAL;
        rt = XFS_IS_REALTIME_INODE(ip);
        extsz = xfs_get_extsz_hint(ip);
@@ -1028,7 +1028,7 @@ xfs_alloc_file_space(
                        /*
                         * Free the transaction structure.
                         */
-                        ASSERT(error == ENOSPC || XFS_FORCED_SHUTDOWN(mp));
+                        ASSERT(error == -ENOSPC || XFS_FORCED_SHUTDOWN(mp));
                        xfs_trans_cancel(tp, 0);
                        break;
                }
@@ -1065,7 +1065,7 @@ xfs_alloc_file_space(
                allocated_fsb = imapp->br_blockcount;
                if (nimaps == 0) {
-                        error = XFS_ERROR(ENOSPC);
+                        error = -ENOSPC;
                        break;
                }
@@ -1126,7 +1126,7 @@ xfs_zero_remaining_bytes(
                                        mp->m_rtdev_targp : mp->m_ddev_targp,
                                  BTOBB(mp->m_sb.sb_blocksize), 0);
        if (!bp)
-                return XFS_ERROR(ENOMEM);
+                return -ENOMEM;
        xfs_buf_unlock(bp);
@@ -1158,7 +1158,7 @@ xfs_zero_remaining_bytes(
                XFS_BUF_SET_ADDR(bp, xfs_fsb_to_db(ip, imap.br_startblock));
                if (XFS_FORCED_SHUTDOWN(mp)) {
-                        error = XFS_ERROR(EIO);
+                        error = -EIO;
                        break;
                }
                xfs_buf_iorequest(bp);
@@ -1176,7 +1176,7 @@ xfs_zero_remaining_bytes(
                XFS_BUF_WRITE(bp);
                if (XFS_FORCED_SHUTDOWN(mp)) {
-                        error = XFS_ERROR(EIO);
+                        error = -EIO;
                        break;
                }
                xfs_buf_iorequest(bp);
@@ -1234,7 +1234,7 @@ xfs_free_file_space(
        rounding = max_t(xfs_off_t, 1 << mp->m_sb.sb_blocklog, PAGE_CACHE_SIZE);
        ioffset = offset & ~(rounding - 1);
-        error = -filemap_write_and_wait_range(VFS_I(ip)->i_mapping,
+        error = filemap_write_and_wait_range(VFS_I(ip)->i_mapping,
                                              ioffset, -1);
        if (error)
                goto out;
@@ -1315,7 +1315,7 @@ xfs_free_file_space(
                        /*
                         * Free the transaction structure.
                         */
-                        ASSERT(error == ENOSPC || XFS_FORCED_SHUTDOWN(mp));
+                        ASSERT(error == -ENOSPC || XFS_FORCED_SHUTDOWN(mp));
                        xfs_trans_cancel(tp, 0);
                        break;
                }
@@ -1557,14 +1557,14 @@ xfs_swap_extents_check_format(
        /* Should never get a local format */
        if (ip->i_d.di_format == XFS_DINODE_FMT_LOCAL ||
            tip->i_d.di_format == XFS_DINODE_FMT_LOCAL)
-                return EINVAL;
+                return -EINVAL;
        /*
         * if the target inode has less extents that then temporary inode then
         * why did userspace call us?
         */
        if (ip->i_d.di_nextents < tip->i_d.di_nextents)
-                return EINVAL;
+                return -EINVAL;
        /*
         * if the target inode is in extent form and the temp inode is in btree
@@ -1573,19 +1573,19 @@ xfs_swap_extents_check_format(
         */
        if (ip->i_d.di_format == XFS_DINODE_FMT_EXTENTS &&
            tip->i_d.di_format == XFS_DINODE_FMT_BTREE)
-                return EINVAL;
+                return -EINVAL;
        /* Check temp in extent form to max in target */
        if (tip->i_d.di_format == XFS_DINODE_FMT_EXTENTS &&
            XFS_IFORK_NEXTENTS(tip, XFS_DATA_FORK) >
                        XFS_IFORK_MAXEXT(ip, XFS_DATA_FORK))
-                return EINVAL;
+                return -EINVAL;
        /* Check target in extent form to max in temp */
        if (ip->i_d.di_format == XFS_DINODE_FMT_EXTENTS &&
            XFS_IFORK_NEXTENTS(ip, XFS_DATA_FORK) >
                        XFS_IFORK_MAXEXT(tip, XFS_DATA_FORK))
-                return EINVAL;
+                return -EINVAL;
        /*
         * If we are in a btree format, check that the temp root block will fit
@@ -1599,26 +1599,50 @@ xfs_swap_extents_check_format(
        if (tip->i_d.di_format == XFS_DINODE_FMT_BTREE) {
                if (XFS_IFORK_BOFF(ip) &&
                    XFS_BMAP_BMDR_SPACE(tip->i_df.if_broot) > XFS_IFORK_BOFF(ip))
-                        return EINVAL;
+                        return -EINVAL;
                if (XFS_IFORK_NEXTENTS(tip, XFS_DATA_FORK) <=
                    XFS_IFORK_MAXEXT(ip, XFS_DATA_FORK))
-                        return EINVAL;
+                        return -EINVAL;
        }
        /* Reciprocal target->temp btree format checks */
        if (ip->i_d.di_format == XFS_DINODE_FMT_BTREE) {
                if (XFS_IFORK_BOFF(tip) &&
                    XFS_BMAP_BMDR_SPACE(ip->i_df.if_broot) > XFS_IFORK_BOFF(tip))
-                        return EINVAL;
+                        return -EINVAL;
                if (XFS_IFORK_NEXTENTS(ip, XFS_DATA_FORK) <=
                    XFS_IFORK_MAXEXT(tip, XFS_DATA_FORK))
-                        return EINVAL;
+                        return -EINVAL;
        }
        return 0;
 }
 int
+xfs_swap_extent_flush(
+        struct xfs_inode        *ip)
+{
+        int     error;
+        error = filemap_write_and_wait(VFS_I(ip)->i_mapping);
+        if (error)
+                return error;
+        truncate_pagecache_range(VFS_I(ip), 0, -1);
+        /* Verify O_DIRECT for ftmp */
+        if (VFS_I(ip)->i_mapping->nrpages)
+                return -EINVAL;
+        /*
+         * Don't try to swap extents on mmap()d files because we can't lock
+         * out races against page faults safely.
+         */
+        if (mapping_mapped(VFS_I(ip)->i_mapping))
+                return -EBUSY;
+        return 0;
+}
+int
 xfs_swap_extents(
        xfs_inode_t     *ip,    /* target inode */
        xfs_inode_t     *tip,   /* tmp inode */
@@ -1633,51 +1657,57 @@ xfs_swap_extents(
        int             aforkblks = 0;
        int             taforkblks = 0;
        __uint64_t      tmp;
+        int             lock_flags;
        tempifp = kmem_alloc(sizeof(xfs_ifork_t), KM_MAYFAIL);
        if (!tempifp) {
-                error = XFS_ERROR(ENOMEM);
+                error = -ENOMEM;
                goto out;
        }
        /*
-         * we have to do two separate lock calls here to keep lockdep
+         * Lock up the inodes against other IO and truncate to begin with.
-         * happy. If we try to get all the locks in one call, lock will
+         * Then we can ensure the inodes are flushed and have no page cache
-         * report false positives when we drop the ILOCK and regain them
+         * safely. Once we have done this we can take the ilocks and do the rest
-         * below.
+         * of the checks.
         */
+        lock_flags = XFS_IOLOCK_EXCL;
        xfs_lock_two_inodes(ip, tip, XFS_IOLOCK_EXCL);
-        xfs_lock_two_inodes(ip, tip, XFS_ILOCK_EXCL);
        /* Verify that both files have the same format */
        if ((ip->i_d.di_mode & S_IFMT) != (tip->i_d.di_mode & S_IFMT)) {
-                error = XFS_ERROR(EINVAL);
+                error = -EINVAL;
                goto out_unlock;
        }
        /* Verify both files are either real-time or non-realtime */
        if (XFS_IS_REALTIME_INODE(ip) != XFS_IS_REALTIME_INODE(tip)) {
-                error = XFS_ERROR(EINVAL);
+                error = -EINVAL;
                goto out_unlock;
        }
-        error = -filemap_write_and_wait(VFS_I(tip)->i_mapping);
+        error = xfs_swap_extent_flush(ip);
+        if (error)
+                goto out_unlock;
+        error = xfs_swap_extent_flush(tip);
        if (error)
                goto out_unlock;
-        truncate_pagecache_range(VFS_I(tip), 0, -1);
-        /* Verify O_DIRECT for ftmp */
+        tp = xfs_trans_alloc(mp, XFS_TRANS_SWAPEXT);
-        if (VN_CACHED(VFS_I(tip)) != 0) {
+        error = xfs_trans_reserve(tp, &M_RES(mp)->tr_ichange, 0, 0);
-                error = XFS_ERROR(EINVAL);
+        if (error) {
+                xfs_trans_cancel(tp, 0);
                goto out_unlock;
        }
+        xfs_lock_two_inodes(ip, tip, XFS_ILOCK_EXCL);
+        lock_flags |= XFS_ILOCK_EXCL;
        /* Verify all data are being swapped */
        if (sxp->sx_offset != 0 ||
            sxp->sx_length != ip->i_d.di_size ||
            sxp->sx_length != tip->i_d.di_size) {
-                error = XFS_ERROR(EFAULT);
+                error = -EFAULT;
-                goto out_unlock;
+                goto out_trans_cancel;
        }
        trace_xfs_swap_extent_before(ip, 0);
@@ -1689,7 +1719,7 @@ xfs_swap_extents(
                xfs_notice(mp,
                    "%s: inode 0x%llx format is incompatible for exchanging.",
                                __func__, ip->i_ino);
-                goto out_unlock;
+                goto out_trans_cancel;
        }
        /*
@@ -1703,43 +1733,9 @@ xfs_swap_extents(
            (sbp->bs_ctime.tv_nsec != VFS_I(ip)->i_ctime.tv_nsec) ||
            (sbp->bs_mtime.tv_sec != VFS_I(ip)->i_mtime.tv_sec) ||
            (sbp->bs_mtime.tv_nsec != VFS_I(ip)->i_mtime.tv_nsec)) {
-                error = XFS_ERROR(EBUSY);
+                error = -EBUSY;
-                goto out_unlock;
+                goto out_trans_cancel;
-        }
-        /* We need to fail if the file is memory mapped.  Once we have tossed
-         * all existing pages, the page fault will have no option
-         * but to go to the filesystem for pages. By making the page fault call
-         * vop_read (or write in the case of autogrow) they block on the iolock
-         * until we have switched the extents.
-         */
-        if (VN_MAPPED(VFS_I(ip))) {
-                error = XFS_ERROR(EBUSY);
-                goto out_unlock;
-        }
-        xfs_iunlock(ip, XFS_ILOCK_EXCL);
-        xfs_iunlock(tip, XFS_ILOCK_EXCL);
-        /*
-         * There is a race condition here since we gave up the
-         * ilock.  However, the data fork will not change since
-         * we have the iolock (locked for truncation too) so we
-         * are safe.  We don't really care if non-io related
-         * fields change.
-         */
-        truncate_pagecache_range(VFS_I(ip), 0, -1);
-        tp = xfs_trans_alloc(mp, XFS_TRANS_SWAPEXT);
-        error = xfs_trans_reserve(tp, &M_RES(mp)->tr_ichange, 0, 0);
-        if (error) {
-                xfs_iunlock(ip,  XFS_IOLOCK_EXCL);
-                xfs_iunlock(tip, XFS_IOLOCK_EXCL);
-                xfs_trans_cancel(tp, 0);
-                goto out;
        }
-        xfs_lock_two_inodes(ip, tip, XFS_ILOCK_EXCL);
        /*
         * Count the number of extended attribute blocks
         */
@@ -1757,8 +1753,8 @@ xfs_swap_extents(
                        goto out_trans_cancel;
        }
-        xfs_trans_ijoin(tp, ip, XFS_ILOCK_EXCL | XFS_IOLOCK_EXCL);
+        xfs_trans_ijoin(tp, ip, lock_flags);
-        xfs_trans_ijoin(tp, tip, XFS_ILOCK_EXCL | XFS_IOLOCK_EXCL);
+        xfs_trans_ijoin(tp, tip, lock_flags);
        /*
         * Before we've swapped the forks, lets set the owners of the forks
@@ -1887,8 +1883,8 @@ out:
        return error;
 out_unlock:
-        xfs_iunlock(ip,  XFS_ILOCK_EXCL | XFS_IOLOCK_EXCL);
+        xfs_iunlock(ip, lock_flags);
-        xfs_iunlock(tip, XFS_ILOCK_EXCL | XFS_IOLOCK_EXCL);
+        xfs_iunlock(tip, lock_flags);
        goto out;
 out_trans_cancel:
diff --git a/fs/xfs/xfs_buf.c b/fs/xfs/xfs_buf.c
index 7a34a1ae6552..cd7b8ca9b064 100644
--- a/fs/xfs/xfs_buf.c
+++ b/fs/xfs/xfs_buf.c
@@ -130,7 +130,7 @@ xfs_buf_get_maps(
        bp->b_maps = kmem_zalloc(map_count * sizeof(struct xfs_buf_map),
                                KM_NOFS);
        if (!bp->b_maps)
-                return ENOMEM;
+                return -ENOMEM;
        return 0;
 }
@@ -344,7 +344,7 @@ retry:
                if (unlikely(page == NULL)) {
                        if (flags & XBF_READ_AHEAD) {
                                bp->b_page_count = i;
-                                error = ENOMEM;
+                                error = -ENOMEM;
                                goto out_free_pages;
                        }
@@ -465,7 +465,7 @@ _xfs_buf_find(
        eofs = XFS_FSB_TO_BB(btp->bt_mount, btp->bt_mount->m_sb.sb_dblocks);
        if (blkno >= eofs) {
                /*
-                 * XXX (dgc): we should really be returning EFSCORRUPTED here,
+                 * XXX (dgc): we should really be returning -EFSCORRUPTED here,
                 * but none of the higher level infrastructure supports
                 * returning a specific error on buffer lookup failures.
                 */
@@ -1052,8 +1052,8 @@ xfs_buf_ioerror(
        xfs_buf_t               *bp,
        int                     error)
 {
-        ASSERT(error >= 0 && error <= 0xffff);
+        ASSERT(error <= 0 && error >= -1000);
-        bp->b_error = (unsigned short)error;
+        bp->b_error = error;
        trace_xfs_buf_ioerror(bp, error, _RET_IP_);
 }
@@ -1064,7 +1064,7 @@ xfs_buf_ioerror_alert(
 {
        xfs_alert(bp->b_target->bt_mount,
 "metadata I/O error: block 0x%llx (\"%s\") error %d numblks %d",
-                (__uint64_t)XFS_BUF_ADDR(bp), func, bp->b_error, bp->b_length);
+                (__uint64_t)XFS_BUF_ADDR(bp), func, -bp->b_error, bp->b_length);
 }
 /*
@@ -1083,7 +1083,7 @@ xfs_bioerror(
        /*
         * No need to wait until the buffer is unpinned, we aren't flushing it.
         */
-        xfs_buf_ioerror(bp, EIO);
+        xfs_buf_ioerror(bp, -EIO);
        /*
         * We're calling xfs_buf_ioend, so delete XBF_DONE flag.
@@ -1094,7 +1094,7 @@ xfs_bioerror(
        xfs_buf_ioend(bp, 0);
-        return EIO;
+        return -EIO;
 }
 /*
@@ -1127,13 +1127,13 @@ xfs_bioerror_relse(
                 * There's no reason to mark error for
                 * ASYNC buffers.
                 */
-                xfs_buf_ioerror(bp, EIO);
+                xfs_buf_ioerror(bp, -EIO);
                complete(&bp->b_iowait);
        } else {
                xfs_buf_relse(bp);
        }
-        return EIO;
+        return -EIO;
 }
 STATIC int
@@ -1199,7 +1199,7 @@ xfs_buf_bio_end_io(
         * buffers that require multiple bios to complete.
         */
        if (!bp->b_error)
-                xfs_buf_ioerror(bp, -error);
+                xfs_buf_ioerror(bp, error);
        if (!bp->b_error && xfs_buf_is_vmapped(bp) && (bp->b_flags & XBF_READ))
                invalidate_kernel_vmap_range(bp->b_addr, xfs_buf_vmap_len(bp));
@@ -1286,7 +1286,7 @@ next_chunk:
                 * because the caller (xfs_buf_iorequest) holds a count itself.
                 */
                atomic_dec(&bp->b_io_remaining);
-                xfs_buf_ioerror(bp, EIO);
+                xfs_buf_ioerror(bp, -EIO);
                bio_put(bio);
        }
@@ -1330,6 +1330,20 @@ _xfs_buf_ioapply(
                                                   SHUTDOWN_CORRUPT_INCORE);
                                return;
                        }
+                } else if (bp->b_bn != XFS_BUF_DADDR_NULL) {
+                        struct xfs_mount *mp = bp->b_target->bt_mount;
+                        /*
+                         * non-crc filesystems don't attach verifiers during
+                         * log recovery, so don't warn for such filesystems.
+                         */
+                        if (xfs_sb_version_hascrc(&mp->m_sb)) {
+                                xfs_warn(mp,
+                                        "%s: no ops on block 0x%llx/0x%x",
+                                        __func__, bp->b_bn, bp->b_length);
+                                xfs_hex_dump(bp->b_addr, 64);
+                                dump_stack();
+                        }
                }
        } else if (bp->b_flags & XBF_READ_AHEAD) {
                rw = READA;
@@ -1628,7 +1642,7 @@ xfs_setsize_buftarg(
                xfs_warn(btp->bt_mount,
                        "Cannot set_blocksize to %u on device %s",
                        sectorsize, name);
-                return EINVAL;
+                return -EINVAL;
        }
        /* Set up device logical sector size mask */
diff --git a/fs/xfs/xfs_buf.h b/fs/xfs/xfs_buf.h
index 3a7a5523d3dc..c753183900b3 100644
--- a/fs/xfs/xfs_buf.h
+++ b/fs/xfs/xfs_buf.h
@@ -178,7 +178,7 @@ typedef struct xfs_buf {
        atomic_t                b_io_remaining; /* #outstanding I/O requests */
        unsigned int            b_page_count;   /* size of page array */
        unsigned int            b_offset;       /* page offset in first page */
-        unsigned short          b_error;        /* error code on I/O */
+        int                     b_error;        /* error code on I/O */
        const struct xfs_buf_ops        *b_ops;
 #ifdef XFS_BUF_LOCK_TRACKING
diff --git a/fs/xfs/xfs_buf_item.c b/fs/xfs/xfs_buf_item.c
index 4654338b03fc..76007deed31f 100644
--- a/fs/xfs/xfs_buf_item.c
+++ b/fs/xfs/xfs_buf_item.c
@@ -488,7 +488,7 @@ xfs_buf_item_unpin(
                xfs_buf_lock(bp);
                xfs_buf_hold(bp);
                bp->b_flags |= XBF_ASYNC;
-                xfs_buf_ioerror(bp, EIO);
+                xfs_buf_ioerror(bp, -EIO);
                XFS_BUF_UNDONE(bp);
                xfs_buf_stale(bp);
                xfs_buf_ioend(bp, 0);
@@ -725,7 +725,7 @@ xfs_buf_item_get_format(
        bip->bli_formats = kmem_zalloc(count * sizeof(struct xfs_buf_log_format),
                                KM_SLEEP);
        if (!bip->bli_formats)
-                return ENOMEM;
+                return -ENOMEM;
        return 0;
 }
diff --git a/fs/xfs/xfs_dir2_readdir.c b/fs/xfs/xfs_dir2_readdir.c
index 48e99afb9cb0..f1b69edcdf31 100644
--- a/fs/xfs/xfs_dir2_readdir.c
+++ b/fs/xfs/xfs_dir2_readdir.c
@@ -95,7 +95,7 @@ xfs_dir2_sf_getdents(
         */
        if (dp->i_d.di_size < offsetof(xfs_dir2_sf_hdr_t, parent)) {
                ASSERT(XFS_FORCED_SHUTDOWN(dp->i_mount));
-                return XFS_ERROR(EIO);
+                return -EIO;
        }
        ASSERT(dp->i_df.if_bytes == dp->i_d.di_size);
@@ -677,7 +677,7 @@ xfs_readdir(
        trace_xfs_readdir(dp);
        if (XFS_FORCED_SHUTDOWN(dp->i_mount))
-                return XFS_ERROR(EIO);
+                return -EIO;
        ASSERT(S_ISDIR(dp->i_d.di_mode));
        XFS_STATS_INC(xs_dir_getdents);
diff --git a/fs/xfs/xfs_discard.c b/fs/xfs/xfs_discard.c
index 4f11ef011139..13d08a1b390e 100644
--- a/fs/xfs/xfs_discard.c
+++ b/fs/xfs/xfs_discard.c
@@ -124,7 +124,7 @@ xfs_trim_extents(
                }
                trace_xfs_discard_extent(mp, agno, fbno, flen);
-                error = -blkdev_issue_discard(bdev, dbno, dlen, GFP_NOFS, 0);
+                error = blkdev_issue_discard(bdev, dbno, dlen, GFP_NOFS, 0);
                if (error)
                        goto out_del_cursor;
                *blocks_trimmed += flen;
@@ -166,11 +166,11 @@ xfs_ioc_trim(
        int                     error, last_error = 0;
        if (!capable(CAP_SYS_ADMIN))
-                return -XFS_ERROR(EPERM);
+                return -EPERM;
        if (!blk_queue_discard(q))
-                return -XFS_ERROR(EOPNOTSUPP);
+                return -EOPNOTSUPP;
        if (copy_from_user(&range, urange, sizeof(range)))
-                return -XFS_ERROR(EFAULT);
+                return -EFAULT;
        /*
         * Truncating down the len isn't actually quite correct, but using
@@ -182,7 +182,7 @@ xfs_ioc_trim(
        if (range.start >= XFS_FSB_TO_B(mp, mp->m_sb.sb_dblocks) ||
            range.minlen > XFS_FSB_TO_B(mp, XFS_ALLOC_AG_MAX_USABLE(mp)) ||
            range.len < mp->m_sb.sb_blocksize)
-                return -XFS_ERROR(EINVAL);
+                return -EINVAL;
        start = BTOBB(range.start);
        end = start + BTOBBT(range.len) - 1;
@@ -195,7 +195,7 @@ xfs_ioc_trim(
        end_agno = xfs_daddr_to_agno(mp, end);
        for (agno = start_agno; agno <= end_agno; agno++) {
-                error = -xfs_trim_extents(mp, agno, start, end, minlen,
+                error = xfs_trim_extents(mp, agno, start, end, minlen,
                                          &blocks_trimmed);
                if (error)
                        last_error = error;
@@ -206,7 +206,7 @@ xfs_ioc_trim(
        range.len = XFS_FSB_TO_B(mp, blocks_trimmed);
        if (copy_to_user(urange, &range, sizeof(range)))
-                return -XFS_ERROR(EFAULT);
+                return -EFAULT;
        return 0;
 }
@@ -222,11 +222,11 @@ xfs_discard_extents(
                trace_xfs_discard_extent(mp, busyp->agno, busyp->bno,
                                         busyp->length);
-                error = -blkdev_issue_discard(mp->m_ddev_targp->bt_bdev,
+                error = blkdev_issue_discard(mp->m_ddev_targp->bt_bdev,
                                XFS_AGB_TO_DADDR(mp, busyp->agno, busyp->bno),
                                XFS_FSB_TO_BB(mp, busyp->length),
                                GFP_NOFS, 0);
-                if (error && error != EOPNOTSUPP) {
+                if (error && error != -EOPNOTSUPP) {
                        xfs_info(mp,
         "discard failed for extent [0x%llu,%u], error %d",
                                 (unsigned long long)busyp->bno,
diff --git a/fs/xfs/xfs_dquot.c b/fs/xfs/xfs_dquot.c
index 3ee0cd43edc0..63c2de49f61d 100644
--- a/fs/xfs/xfs_dquot.c
+++ b/fs/xfs/xfs_dquot.c
@@ -327,7 +327,7 @@ xfs_qm_dqalloc(
         */
        if (!xfs_this_quota_on(dqp->q_mount, dqp->dq_flags)) {
                xfs_iunlock(quotip, XFS_ILOCK_EXCL);
-                return (ESRCH);
+                return -ESRCH;
        }
        xfs_trans_ijoin(tp, quotip, XFS_ILOCK_EXCL);
@@ -354,7 +354,7 @@ xfs_qm_dqalloc(
                               mp->m_quotainfo->qi_dqchunklen,
                               0);
        if (!bp) {
-                error = ENOMEM;
+                error = -ENOMEM;
                goto error1;
        }
        bp->b_ops = &xfs_dquot_buf_ops;
@@ -400,7 +400,7 @@ xfs_qm_dqalloc(
      error0:
        xfs_iunlock(quotip, XFS_ILOCK_EXCL);
-        return (error);
+        return error;
 }
 STATIC int
@@ -426,7 +426,7 @@ xfs_qm_dqrepair(
        if (error) {
                ASSERT(*bpp == NULL);
-                return XFS_ERROR(error);
+                return error;
        }
        (*bpp)->b_ops = &xfs_dquot_buf_ops;
@@ -442,7 +442,7 @@ xfs_qm_dqrepair(
                if (error) {
                        /* repair failed, we're screwed */
                        xfs_trans_brelse(tp, *bpp);
-                        return XFS_ERROR(EIO);
+                        return -EIO;
                }
        }
@@ -480,7 +480,7 @@ xfs_qm_dqtobp(
                 * didn't have the quota inode lock.
                 */
                xfs_iunlock(quotip, lock_mode);
-                return ESRCH;
+                return -ESRCH;
        }
        /*
@@ -508,7 +508,7 @@ xfs_qm_dqtobp(
                 * We don't allocate unless we're asked to
                 */
                if (!(flags & XFS_QMOPT_DQALLOC))
-                        return ENOENT;
+                        return -ENOENT;
                ASSERT(tp);
                error = xfs_qm_dqalloc(tpp, mp, dqp, quotip,
@@ -530,7 +530,7 @@ xfs_qm_dqtobp(
                                           mp->m_quotainfo->qi_dqchunklen,
                                           0, &bp, &xfs_dquot_buf_ops);
-                if (error == EFSCORRUPTED && (flags & XFS_QMOPT_DQREPAIR)) {
+                if (error == -EFSCORRUPTED && (flags & XFS_QMOPT_DQREPAIR)) {
                        xfs_dqid_t firstid = (xfs_dqid_t)map.br_startoff *
                                                mp->m_quotainfo->qi_dqperchunk;
                        ASSERT(bp == NULL);
@@ -539,7 +539,7 @@ xfs_qm_dqtobp(
                if (error) {
                        ASSERT(bp == NULL);
-                        return XFS_ERROR(error);
+                        return error;
                }
        }
@@ -547,7 +547,7 @@ xfs_qm_dqtobp(
        *O_bpp = bp;
        *O_ddpp = bp->b_addr + dqp->q_bufoffset;
-        return (0);
+        return 0;
 }
@@ -715,7 +715,7 @@ xfs_qm_dqget(
        if ((! XFS_IS_UQUOTA_ON(mp) && type == XFS_DQ_USER) ||
            (! XFS_IS_PQUOTA_ON(mp) && type == XFS_DQ_PROJ) ||
            (! XFS_IS_GQUOTA_ON(mp) && type == XFS_DQ_GROUP)) {
-                return (ESRCH);
+                return -ESRCH;
        }
 #ifdef DEBUG
@@ -723,7 +723,7 @@ xfs_qm_dqget(
                if ((xfs_dqerror_target == mp->m_ddev_targp) &&
                    (xfs_dqreq_num++ % xfs_dqerror_mod) == 0) {
                        xfs_debug(mp, "Returning error in dqget");
-                        return (EIO);
+                        return -EIO;
                }
        }
@@ -796,14 +796,14 @@ restart:
                } else {
                        /* inode stays locked on return */
                        xfs_qm_dqdestroy(dqp);
-                        return XFS_ERROR(ESRCH);
+                        return -ESRCH;
                }
        }
        mutex_lock(&qi->qi_tree_lock);
-        error = -radix_tree_insert(tree, id, dqp);
+        error = radix_tree_insert(tree, id, dqp);
        if (unlikely(error)) {
-                WARN_ON(error != EEXIST);
+                WARN_ON(error != -EEXIST);
                /*
                 * Duplicate found. Just throw away the new dquot and start
@@ -829,7 +829,7 @@ restart:
        ASSERT((ip == NULL) || xfs_isilocked(ip, XFS_ILOCK_EXCL));
        trace_xfs_dqget_miss(dqp);
        *O_dqpp = dqp;
-        return (0);
+        return 0;
 }
 /*
@@ -966,7 +966,7 @@ xfs_qm_dqflush(
                                             SHUTDOWN_CORRUPT_INCORE);
                else
                        spin_unlock(&mp->m_ail->xa_lock);
-                error = XFS_ERROR(EIO);
+                error = -EIO;
                goto out_unlock;
        }
@@ -974,7 +974,8 @@ xfs_qm_dqflush(
         * Get the buffer containing the on-disk dquot
         */
        error = xfs_trans_read_buf(mp, NULL, mp->m_ddev_targp, dqp->q_blkno,
-                                   mp->m_quotainfo->qi_dqchunklen, 0, &bp, NULL);
+                                   mp->m_quotainfo->qi_dqchunklen, 0, &bp,
+                                   &xfs_dquot_buf_ops);
        if (error)
                goto out_unlock;
@@ -992,7 +993,7 @@ xfs_qm_dqflush(
                xfs_buf_relse(bp);
                xfs_dqfunlock(dqp);
                xfs_force_shutdown(mp, SHUTDOWN_CORRUPT_INCORE);
-                return XFS_ERROR(EIO);
+                return -EIO;
        }
        /* This is the only portion of data that needs to persist */
@@ -1045,7 +1046,7 @@ xfs_qm_dqflush(
 out_unlock:
        xfs_dqfunlock(dqp);
-        return XFS_ERROR(EIO);
+        return -EIO;
 }
 /*
diff --git a/fs/xfs/xfs_dquot.h b/fs/xfs/xfs_dquot.h
index 68a68f704837..c24c67e22a2a 100644
--- a/fs/xfs/xfs_dquot.h
+++ b/fs/xfs/xfs_dquot.h
@@ -139,6 +139,21 @@ static inline xfs_dquot_t *xfs_inode_dquot(struct xfs_inode *ip, int type)
        }
 }
+/*
+ * Check whether a dquot is under low free space conditions. We assume the quota
+ * is enabled and enforced.
+ */
+static inline bool xfs_dquot_lowsp(struct xfs_dquot *dqp)
+{
+        int64_t freesp;
+        freesp = be64_to_cpu(dqp->q_core.d_blk_hardlimit) - dqp->q_res_bcount;
+        if (freesp < dqp->q_low_space[XFS_QLOWSP_1_PCNT])
+                return true;
+        return false;
+}
 #define XFS_DQ_IS_LOCKED(dqp)   (mutex_is_locked(&((dqp)->q_qlock)))
 #define XFS_DQ_IS_DIRTY(dqp)    ((dqp)->dq_flags & XFS_DQ_DIRTY)
 #define XFS_QM_ISUDQ(dqp)       ((dqp)->dq_flags & XFS_DQ_USER)
diff --git a/fs/xfs/xfs_error.c b/fs/xfs/xfs_error.c
index edac5b057d28..b92fd7bc49e3 100644
--- a/fs/xfs/xfs_error.c
+++ b/fs/xfs/xfs_error.c
@@ -27,29 +27,6 @@
 #ifdef DEBUG
-int     xfs_etrap[XFS_ERROR_NTRAP] = {
-        0,
-};
-int
-xfs_error_trap(int e)
-{
-        int i;
-        if (!e)
-                return 0;
-        for (i = 0; i < XFS_ERROR_NTRAP; i++) {
-                if (xfs_etrap[i] == 0)
-                        break;
-                if (e != xfs_etrap[i])
-                        continue;
-                xfs_notice(NULL, "%s: error %d", __func__, e);
-                BUG();
-                break;
-        }
-        return e;
-}
 int     xfs_etest[XFS_NUM_INJECT_ERROR];
 int64_t xfs_etest_fsid[XFS_NUM_INJECT_ERROR];
 char *  xfs_etest_fsname[XFS_NUM_INJECT_ERROR];
@@ -190,7 +167,7 @@ xfs_verifier_error(
        struct xfs_mount *mp = bp->b_target->bt_mount;
        xfs_alert(mp, "Metadata %s detected at %pF, block 0x%llx",
-                  bp->b_error == EFSBADCRC ? "CRC error" : "corruption",
+                  bp->b_error == -EFSBADCRC ? "CRC error" : "corruption",
                  __return_address, bp->b_bn);
        xfs_alert(mp, "Unmount and run xfs_repair");
diff --git a/fs/xfs/xfs_error.h b/fs/xfs/xfs_error.h
index c1c57d4a4b5d..279a76e52791 100644
--- a/fs/xfs/xfs_error.h
+++ b/fs/xfs/xfs_error.h
@@ -18,15 +18,6 @@
 #ifndef __XFS_ERROR_H__
 #define __XFS_ERROR_H__
-#ifdef DEBUG
-#define XFS_ERROR_NTRAP 10
-extern int      xfs_etrap[XFS_ERROR_NTRAP];
-extern int      xfs_error_trap(int);
-#define XFS_ERROR(e)    xfs_error_trap(e)
-#else
-#define XFS_ERROR(e)    (e)
-#endif
 struct xfs_mount;
 extern void xfs_error_report(const char *tag, int level, struct xfs_mount *mp,
@@ -56,7 +47,7 @@ extern void xfs_verifier_error(struct xfs_buf *bp);
                if (unlikely(!fs_is_ok)) { \
                        XFS_ERROR_REPORT("XFS_WANT_CORRUPTED_GOTO", \
                                         XFS_ERRLEVEL_LOW, NULL); \
-                        error = XFS_ERROR(EFSCORRUPTED); \
+                        error = -EFSCORRUPTED; \
                        goto l; \
                } \
        }
@@ -68,7 +59,7 @@ extern void xfs_verifier_error(struct xfs_buf *bp);
                if (unlikely(!fs_is_ok)) { \
                        XFS_ERROR_REPORT("XFS_WANT_CORRUPTED_RETURN", \
                                         XFS_ERRLEVEL_LOW, NULL); \
-                        return XFS_ERROR(EFSCORRUPTED); \
+                        return -EFSCORRUPTED; \
                } \
        }
diff --git a/fs/xfs/xfs_export.c b/fs/xfs/xfs_export.c
index 753e467aa1a5..5a6bd5d8779a 100644
--- a/fs/xfs/xfs_export.c
+++ b/fs/xfs/xfs_export.c
@@ -147,9 +147,9 @@ xfs_nfs_get_inode(
                 * We don't use ESTALE directly down the chain to not
                 * confuse applications using bulkstat that expect EINVAL.
                 */
-                if (error == EINVAL || error == ENOENT)
+                if (error == -EINVAL || error == -ENOENT)
-                        error = ESTALE;
+                        error = -ESTALE;
-                return ERR_PTR(-error);
+                return ERR_PTR(error);
        }
        if (ip->i_d.di_gen != generation) {
@@ -217,7 +217,7 @@ xfs_fs_get_parent(
        error = xfs_lookup(XFS_I(child->d_inode), &xfs_name_dotdot, &cip, NULL);
        if (unlikely(error))
-                return ERR_PTR(-error);
+                return ERR_PTR(error);
        return d_obtain_alias(VFS_I(cip));
 }
@@ -237,7 +237,7 @@ xfs_fs_nfs_commit_metadata(
        if (!lsn)
                return 0;
-        return -_xfs_log_force_lsn(mp, lsn, XFS_LOG_SYNC, NULL);
+        return _xfs_log_force_lsn(mp, lsn, XFS_LOG_SYNC, NULL);
 }
 const struct export_operations xfs_export_operations = {
diff --git a/fs/xfs/xfs_extfree_item.c b/fs/xfs/xfs_extfree_item.c
index fb7a4c1ce1c5..c4327419dc5c 100644
--- a/fs/xfs/xfs_extfree_item.c
+++ b/fs/xfs/xfs_extfree_item.c
@@ -298,7 +298,7 @@ xfs_efi_copy_format(xfs_log_iovec_t *buf, xfs_efi_log_format_t *dst_efi_fmt)
                }
                return 0;
        }
-        return EFSCORRUPTED;
+        return -EFSCORRUPTED;
 }
 /*
diff --git a/fs/xfs/xfs_file.c b/fs/xfs/xfs_file.c
index 1f66779d7a46..076b1708d134 100644
--- a/fs/xfs/xfs_file.c
+++ b/fs/xfs/xfs_file.c
@@ -38,6 +38,7 @@
 #include "xfs_trace.h"
 #include "xfs_log.h"
 #include "xfs_dinode.h"
+#include "xfs_icache.h"
 #include <linux/aio.h>
 #include <linux/dcache.h>
@@ -155,7 +156,7 @@ xfs_dir_fsync(
        if (!lsn)
                return 0;
-        return -_xfs_log_force_lsn(mp, lsn, XFS_LOG_SYNC, NULL);
+        return _xfs_log_force_lsn(mp, lsn, XFS_LOG_SYNC, NULL);
 }
 STATIC int
@@ -179,7 +180,7 @@ xfs_file_fsync(
                return error;
        if (XFS_FORCED_SHUTDOWN(mp))
-                return -XFS_ERROR(EIO);
+                return -EIO;
        xfs_iflags_clear(ip, XFS_ITRUNCATED);
@@ -225,7 +226,7 @@ xfs_file_fsync(
            !log_flushed)
                xfs_blkdev_issue_flush(mp->m_ddev_targp);
-        return -error;
+        return error;
 }
 STATIC ssize_t
@@ -246,11 +247,11 @@ xfs_file_read_iter(
        XFS_STATS_INC(xs_read_calls);
        if (unlikely(file->f_flags & O_DIRECT))
-                ioflags |= IO_ISDIRECT;
+                ioflags |= XFS_IO_ISDIRECT;
        if (file->f_mode & FMODE_NOCMTIME)
-                ioflags |= IO_INVIS;
+                ioflags |= XFS_IO_INVIS;
-        if (unlikely(ioflags & IO_ISDIRECT)) {
+        if (unlikely(ioflags & XFS_IO_ISDIRECT)) {
                xfs_buftarg_t   *target =
                        XFS_IS_REALTIME_INODE(ip) ?
                                mp->m_rtdev_targp : mp->m_ddev_targp;
@@ -258,7 +259,7 @@ xfs_file_read_iter(
                if ((pos | size) & target->bt_logical_sectormask) {
                        if (pos == i_size_read(inode))
                                return 0;
-                        return -XFS_ERROR(EINVAL);
+                        return -EINVAL;
                }
        }
@@ -283,7 +284,7 @@ xfs_file_read_iter(
         * proceeed concurrently without serialisation.
         */
        xfs_rw_ilock(ip, XFS_IOLOCK_SHARED);
-        if ((ioflags & IO_ISDIRECT) && inode->i_mapping->nrpages) {
+        if ((ioflags & XFS_IO_ISDIRECT) && inode->i_mapping->nrpages) {
                xfs_rw_iunlock(ip, XFS_IOLOCK_SHARED);
                xfs_rw_ilock(ip, XFS_IOLOCK_EXCL);
@@ -325,7 +326,7 @@ xfs_file_splice_read(
        XFS_STATS_INC(xs_read_calls);
        if (infilp->f_mode & FMODE_NOCMTIME)
-                ioflags |= IO_INVIS;
+                ioflags |= XFS_IO_INVIS;
        if (XFS_FORCED_SHUTDOWN(ip->i_mount))
                return -EIO;
@@ -524,7 +525,7 @@ restart:
                        xfs_rw_ilock(ip, *iolock);
                        goto restart;
                }
-                error = -xfs_zero_eof(ip, *pos, i_size_read(inode));
+                error = xfs_zero_eof(ip, *pos, i_size_read(inode));
                if (error)
                        return error;
        }
@@ -594,7 +595,7 @@ xfs_file_dio_aio_write(
        /* DIO must be aligned to device logical sector size */
        if ((pos | count) & target->bt_logical_sectormask)
-                return -XFS_ERROR(EINVAL);
+                return -EINVAL;
        /* "unaligned" here means not aligned to a filesystem block */
        if ((pos & mp->m_blockmask) || ((pos + count) & mp->m_blockmask))
@@ -689,14 +690,28 @@ write_retry:
        ret = generic_perform_write(file, from, pos);
        if (likely(ret >= 0))
                iocb->ki_pos = pos + ret;
        /*
-         * If we just got an ENOSPC, try to write back all dirty inodes to
+         * If we hit a space limit, try to free up some lingering preallocated
-         * convert delalloc space to free up some of the excess reserved
+         * space before returning an error. In the case of ENOSPC, first try to
-         * metadata space.
+         * write back all dirty inodes to free up some of the excess reserved
+         * metadata space. This reduces the chances that the eofblocks scan
+         * waits on dirty mappings. Since xfs_flush_inodes() is serialized, this
+         * also behaves as a filter to prevent too many eofblocks scans from
+         * running at the same time.
         */
-        if (ret == -ENOSPC && !enospc) {
+        if (ret == -EDQUOT && !enospc) {
+                enospc = xfs_inode_free_quota_eofblocks(ip);
+                if (enospc)
+                        goto write_retry;
+        } else if (ret == -ENOSPC && !enospc) {
+                struct xfs_eofblocks eofb = {0};
                enospc = 1;
                xfs_flush_inodes(ip->i_mount);
+                eofb.eof_scan_owner = ip->i_ino; /* for locking */
+                eofb.eof_flags = XFS_EOF_FLAGS_SYNC;
+                xfs_icache_free_eofblocks(ip->i_mount, &eofb);
                goto write_retry;
        }
@@ -772,7 +787,7 @@ xfs_file_fallocate(
                unsigned blksize_mask = (1 << inode->i_blkbits) - 1;
                if (offset & blksize_mask || len & blksize_mask) {
-                        error = EINVAL;
+                        error = -EINVAL;
                        goto out_unlock;
                }
@@ -781,7 +796,7 @@ xfs_file_fallocate(
                 * in which case it is effectively a truncate operation
                 */
                if (offset + len >= i_size_read(inode)) {
-                        error = EINVAL;
+                        error = -EINVAL;
                        goto out_unlock;
                }
@@ -794,7 +809,7 @@ xfs_file_fallocate(
                if (!(mode & FALLOC_FL_KEEP_SIZE) &&
                    offset + len > i_size_read(inode)) {
                        new_size = offset + len;
-                        error = -inode_newsize_ok(inode, new_size);
+                        error = inode_newsize_ok(inode, new_size);
                        if (error)
                                goto out_unlock;
                }
@@ -844,7 +859,7 @@ xfs_file_fallocate(
 out_unlock:
        xfs_iunlock(ip, XFS_IOLOCK_EXCL);
-        return -error;
+        return error;
 }
@@ -889,7 +904,7 @@ xfs_file_release(
        struct inode    *inode,
        struct file     *filp)
 {
-        return -xfs_release(XFS_I(inode));
+        return xfs_release(XFS_I(inode));
 }
 STATIC int
@@ -918,7 +933,7 @@ xfs_file_readdir(
        error = xfs_readdir(ip, ctx, bufsize);
        if (error)
-                return -error;
+                return error;
        return 0;
 }
@@ -1184,7 +1199,7 @@ xfs_seek_data(
        isize = i_size_read(inode);
        if (start >= isize) {
-                error = ENXIO;
+                error = -ENXIO;
                goto out_unlock;
        }
@@ -1206,7 +1221,7 @@ xfs_seek_data(
                /* No extents at given offset, must be beyond EOF */
                if (nmap == 0) {
-                        error = ENXIO;
+                        error = -ENXIO;
                        goto out_unlock;
                }
@@ -1237,7 +1252,7 @@ xfs_seek_data(
                 * we are reading after EOF if nothing in map[1].
                 */
                if (nmap == 1) {
-                        error = ENXIO;
+                        error = -ENXIO;
                        goto out_unlock;
                }
@@ -1250,7 +1265,7 @@ xfs_seek_data(
                fsbno = map[i - 1].br_startoff + map[i - 1].br_blockcount;
                start = XFS_FSB_TO_B(mp, fsbno);
                if (start >= isize) {
-                        error = ENXIO;
+                        error = -ENXIO;
                        goto out_unlock;
                }
        }
@@ -1262,7 +1277,7 @@ out_unlock:
        xfs_iunlock(ip, lock);
        if (error)
-                return -error;
+                return error;
        return offset;
 }
@@ -1282,13 +1297,13 @@ xfs_seek_hole(
        int                     error;
        if (XFS_FORCED_SHUTDOWN(mp))
-                return -XFS_ERROR(EIO);
+                return -EIO;
        lock = xfs_ilock_data_map_shared(ip);
        isize = i_size_read(inode);
        if (start >= isize) {
-                error = ENXIO;
+                error = -ENXIO;
                goto out_unlock;
        }
@@ -1307,7 +1322,7 @@ xfs_seek_hole(
                /* No extents at given offset, must be beyond EOF */
                if (nmap == 0) {
-                        error = ENXIO;
+                        error = -ENXIO;
                        goto out_unlock;
                }
@@ -1370,7 +1385,7 @@ out_unlock:
        xfs_iunlock(ip, lock);
        if (error)
-                return -error;
+                return error;
        return offset;
 }
diff --git a/fs/xfs/xfs_filestream.c b/fs/xfs/xfs_filestream.c
index 8ec81bed7992..e92730c1d3ca 100644
--- a/fs/xfs/xfs_filestream.c
+++ b/fs/xfs/xfs_filestream.c
@@ -258,7 +258,7 @@ next_ag:
        if (*agp == NULLAGNUMBER)
                return 0;
-        err = ENOMEM;
+        err = -ENOMEM;
        item = kmem_alloc(sizeof(*item), KM_MAYFAIL);
        if (!item)
                goto out_put_ag;
@@ -268,7 +268,7 @@ next_ag:
        err = xfs_mru_cache_insert(mp->m_filestream, ip->i_ino, &item->mru);
        if (err) {
-                if (err == EEXIST)
+                if (err == -EEXIST)
                        err = 0;
                goto out_free_item;
        }
diff --git a/fs/xfs/xfs_fs.h b/fs/xfs/xfs_fs.h
index d34703dbcb42..18dc721ca19f 100644
--- a/fs/xfs/xfs_fs.h
+++ b/fs/xfs/xfs_fs.h
@@ -255,8 +255,8 @@ typedef struct xfs_fsop_resblks {
        ((2 * 1024 * 1024 * 1024ULL) - XFS_MIN_LOG_BYTES)
 /* Used for sanity checks on superblock */
-#define XFS_MAX_DBLOCKS(s) ((xfs_drfsbno_t)(s)->sb_agcount * (s)->sb_agblocks)
+#define XFS_MAX_DBLOCKS(s) ((xfs_rfsblock_t)(s)->sb_agcount * (s)->sb_agblocks)
-#define XFS_MIN_DBLOCKS(s) ((xfs_drfsbno_t)((s)->sb_agcount - 1) *      \
+#define XFS_MIN_DBLOCKS(s) ((xfs_rfsblock_t)((s)->sb_agcount - 1) *     \
                         (s)->sb_agblocks + XFS_MIN_AG_BLOCKS)
 /*
@@ -375,6 +375,9 @@ struct xfs_fs_eofblocks {
 #define XFS_EOF_FLAGS_GID               (1 << 2) /* filter by gid */
 #define XFS_EOF_FLAGS_PRID              (1 << 3) /* filter by project id */
 #define XFS_EOF_FLAGS_MINFILESIZE       (1 << 4) /* filter by min file size */
+#define XFS_EOF_FLAGS_UNION             (1 << 5) /* union filter algorithm;
+                                                  * kernel only, not included in
+                                                  * valid mask */
 #define XFS_EOF_FLAGS_VALID     \
        (XFS_EOF_FLAGS_SYNC |   \
         XFS_EOF_FLAGS_UID |    \
diff --git a/fs/xfs/xfs_fsops.c b/fs/xfs/xfs_fsops.c
index d2295561570a..f91de1ef05e1 100644
--- a/fs/xfs/xfs_fsops.c
+++ b/fs/xfs/xfs_fsops.c
@@ -168,7 +168,7 @@ xfs_growfs_data_private(
        nb = in->newblocks;
        pct = in->imaxpct;
        if (nb < mp->m_sb.sb_dblocks || pct < 0 || pct > 100)
-                return XFS_ERROR(EINVAL);
+                return -EINVAL;
        if ((error = xfs_sb_validate_fsb_count(&mp->m_sb, nb)))
                return error;
        dpct = pct - mp->m_sb.sb_imax_pct;
@@ -176,7 +176,7 @@ xfs_growfs_data_private(
                                XFS_FSB_TO_BB(mp, nb) - XFS_FSS_TO_BB(mp, 1),
                                XFS_FSS_TO_BB(mp, 1), 0, NULL);
        if (!bp)
-                return EIO;
+                return -EIO;
        if (bp->b_error) {
                error = bp->b_error;
                xfs_buf_relse(bp);
@@ -191,7 +191,7 @@ xfs_growfs_data_private(
                nagcount--;
                nb = (xfs_rfsblock_t)nagcount * mp->m_sb.sb_agblocks;
                if (nb < mp->m_sb.sb_dblocks)
-                        return XFS_ERROR(EINVAL);
+                        return -EINVAL;
        }
        new = nb - mp->m_sb.sb_dblocks;
        oagcount = mp->m_sb.sb_agcount;
@@ -229,7 +229,7 @@ xfs_growfs_data_private(
                                XFS_FSS_TO_BB(mp, 1), 0,
                                &xfs_agf_buf_ops);
                if (!bp) {
-                        error = ENOMEM;
+                        error = -ENOMEM;
                        goto error0;
                }
@@ -270,7 +270,7 @@ xfs_growfs_data_private(
                                XFS_FSS_TO_BB(mp, 1), 0,
                                &xfs_agfl_buf_ops);
                if (!bp) {
-                        error = ENOMEM;
+                        error = -ENOMEM;
                        goto error0;
                }
@@ -298,7 +298,7 @@ xfs_growfs_data_private(
                                XFS_FSS_TO_BB(mp, 1), 0,
                                &xfs_agi_buf_ops);
                if (!bp) {
-                        error = ENOMEM;
+                        error = -ENOMEM;
                        goto error0;
                }
@@ -336,7 +336,7 @@ xfs_growfs_data_private(
                                &xfs_allocbt_buf_ops);
                if (!bp) {
-                        error = ENOMEM;
+                        error = -ENOMEM;
                        goto error0;
                }
@@ -365,7 +365,7 @@ xfs_growfs_data_private(
                                BTOBB(mp->m_sb.sb_blocksize), 0,
                                &xfs_allocbt_buf_ops);
                if (!bp) {
-                        error = ENOMEM;
+                        error = -ENOMEM;
                        goto error0;
                }
@@ -395,7 +395,7 @@ xfs_growfs_data_private(
                                BTOBB(mp->m_sb.sb_blocksize), 0,
                                &xfs_inobt_buf_ops);
                if (!bp) {
-                        error = ENOMEM;
+                        error = -ENOMEM;
                        goto error0;
                }
@@ -420,7 +420,7 @@ xfs_growfs_data_private(
                                BTOBB(mp->m_sb.sb_blocksize), 0,
                                &xfs_inobt_buf_ops);
                        if (!bp) {
-                                error = ENOMEM;
+                                error = -ENOMEM;
                                goto error0;
                        }
@@ -531,7 +531,7 @@ xfs_growfs_data_private(
                                bp->b_ops = &xfs_sb_buf_ops;
                                xfs_buf_zero(bp, 0, BBTOB(bp->b_length));
                        } else
-                                error = ENOMEM;
+                                error = -ENOMEM;
                }
                /*
@@ -576,17 +576,17 @@ xfs_growfs_log_private(
        nb = in->newblocks;
        if (nb < XFS_MIN_LOG_BLOCKS || nb < XFS_B_TO_FSB(mp, XFS_MIN_LOG_BYTES))
-                return XFS_ERROR(EINVAL);
+                return -EINVAL;
        if (nb == mp->m_sb.sb_logblocks &&
            in->isint == (mp->m_sb.sb_logstart != 0))
-                return XFS_ERROR(EINVAL);
+                return -EINVAL;
        /*
         * Moving the log is hard, need new interfaces to sync
         * the log first, hold off all activity while moving it.
         * Can have shorter or longer log in the same space,
         * or transform internal to external log or vice versa.
         */
-        return XFS_ERROR(ENOSYS);
+        return -ENOSYS;
 }
 /*
@@ -604,9 +604,9 @@ xfs_growfs_data(
        int error;
        if (!capable(CAP_SYS_ADMIN))
-                return XFS_ERROR(EPERM);
+                return -EPERM;
        if (!mutex_trylock(&mp->m_growlock))
-                return XFS_ERROR(EWOULDBLOCK);
+                return -EWOULDBLOCK;
        error = xfs_growfs_data_private(mp, in);
        mutex_unlock(&mp->m_growlock);
        return error;
@@ -620,9 +620,9 @@ xfs_growfs_log(
        int error;
        if (!capable(CAP_SYS_ADMIN))
-                return XFS_ERROR(EPERM);
+                return -EPERM;
        if (!mutex_trylock(&mp->m_growlock))
-                return XFS_ERROR(EWOULDBLOCK);
+                return -EWOULDBLOCK;
        error = xfs_growfs_log_private(mp, in);
        mutex_unlock(&mp->m_growlock);
        return error;
@@ -674,7 +674,7 @@ xfs_reserve_blocks(
        /* If inval is null, report current values and return */
        if (inval == (__uint64_t *)NULL) {
                if (!outval)
-                        return EINVAL;
+                        return -EINVAL;
                outval->resblks = mp->m_resblks;
                outval->resblks_avail = mp->m_resblks_avail;
                return 0;
@@ -757,7 +757,7 @@ out:
                int error;
                error = xfs_icsb_modify_counters(mp, XFS_SBS_FDBLOCKS,
                                                 fdblks_delta, 0);
-                if (error == ENOSPC)
+                if (error == -ENOSPC)
                        goto retry;
        }
        return 0;
@@ -818,7 +818,7 @@ xfs_fs_goingdown(
                                SHUTDOWN_FORCE_UMOUNT | SHUTDOWN_LOG_IO_ERROR);
                break;
        default:
-                return XFS_ERROR(EINVAL);
+                return -EINVAL;
        }
        return 0;
diff --git a/fs/xfs/xfs_icache.c b/fs/xfs/xfs_icache.c
index c48df5f25b9f..981b2cf51985 100644
--- a/fs/xfs/xfs_icache.c
+++ b/fs/xfs/xfs_icache.c
@@ -33,6 +33,9 @@
 #include "xfs_trace.h"
 #include "xfs_icache.h"
 #include "xfs_bmap_util.h"
+#include "xfs_quota.h"
+#include "xfs_dquot_item.h"
+#include "xfs_dquot.h"
 #include <linux/kthread.h>
 #include <linux/freezer.h>
@@ -158,7 +161,7 @@ xfs_iget_cache_hit(
        if (ip->i_ino != ino) {
                trace_xfs_iget_skip(ip);
                XFS_STATS_INC(xs_ig_frecycle);
-                error = EAGAIN;
+                error = -EAGAIN;
                goto out_error;
        }
@@ -176,7 +179,7 @@ xfs_iget_cache_hit(
        if (ip->i_flags & (XFS_INEW|XFS_IRECLAIM)) {
                trace_xfs_iget_skip(ip);
                XFS_STATS_INC(xs_ig_frecycle);
-                error = EAGAIN;
+                error = -EAGAIN;
                goto out_error;
        }
@@ -184,7 +187,7 @@ xfs_iget_cache_hit(
         * If lookup is racing with unlink return an error immediately.
         */
        if (ip->i_d.di_mode == 0 && !(flags & XFS_IGET_CREATE)) {
-                error = ENOENT;
+                error = -ENOENT;
                goto out_error;
        }
@@ -206,7 +209,7 @@ xfs_iget_cache_hit(
                spin_unlock(&ip->i_flags_lock);
                rcu_read_unlock();
-                error = -inode_init_always(mp->m_super, inode);
+                error = inode_init_always(mp->m_super, inode);
                if (error) {
                        /*
                         * Re-initializing the inode failed, and we are in deep
@@ -243,7 +246,7 @@ xfs_iget_cache_hit(
                /* If the VFS inode is being torn down, pause and try again. */
                if (!igrab(inode)) {
                        trace_xfs_iget_skip(ip);
-                        error = EAGAIN;
+                        error = -EAGAIN;
                        goto out_error;
                }
@@ -285,7 +288,7 @@ xfs_iget_cache_miss(
        ip = xfs_inode_alloc(mp, ino);
        if (!ip)
-                return ENOMEM;
+                return -ENOMEM;
        error = xfs_iread(mp, tp, ip, flags);
        if (error)
@@ -294,7 +297,7 @@ xfs_iget_cache_miss(
        trace_xfs_iget_miss(ip);
        if ((ip->i_d.di_mode == 0) && !(flags & XFS_IGET_CREATE)) {
-                error = ENOENT;
+                error = -ENOENT;
                goto out_destroy;
        }
@@ -305,7 +308,7 @@ xfs_iget_cache_miss(
         * recurse into the file system.
         */
        if (radix_tree_preload(GFP_NOFS)) {
-                error = EAGAIN;
+                error = -EAGAIN;
                goto out_destroy;
        }
@@ -341,7 +344,7 @@ xfs_iget_cache_miss(
        if (unlikely(error)) {
                WARN_ON(error != -EEXIST);
                XFS_STATS_INC(xs_ig_dup);
-                error = EAGAIN;
+                error = -EAGAIN;
                goto out_preload_end;
        }
        spin_unlock(&pag->pag_ici_lock);
@@ -408,7 +411,7 @@ xfs_iget(
        /* reject inode numbers outside existing AGs */
        if (!ino || XFS_INO_TO_AGNO(mp, ino) >= mp->m_sb.sb_agcount)
-                return EINVAL;
+                return -EINVAL;
        /* get the perag structure and ensure that it's inode capable */
        pag = xfs_perag_get(mp, XFS_INO_TO_AGNO(mp, ino));
@@ -445,7 +448,7 @@ again:
        return 0;
 out_error_or_again:
-        if (error == EAGAIN) {
+        if (error == -EAGAIN) {
                delay(1);
                goto again;
        }
@@ -489,18 +492,18 @@ xfs_inode_ag_walk_grab(
        /* nothing to sync during shutdown */
        if (XFS_FORCED_SHUTDOWN(ip->i_mount))
-                return EFSCORRUPTED;
+                return -EFSCORRUPTED;
        /* If we can't grab the inode, it must on it's way to reclaim. */
        if (!igrab(inode))
-                return ENOENT;
+                return -ENOENT;
        /* inode is valid */
        return 0;
 out_unlock_noent:
        spin_unlock(&ip->i_flags_lock);
-        return ENOENT;
+        return -ENOENT;
 }
 STATIC int
@@ -583,16 +586,16 @@ restart:
                                continue;
                        error = execute(batch[i], flags, args);
                        IRELE(batch[i]);
-                        if (error == EAGAIN) {
+                        if (error == -EAGAIN) {
                                skipped++;
                                continue;
                        }
-                        if (error && last_error != EFSCORRUPTED)
+                        if (error && last_error != -EFSCORRUPTED)
                                last_error = error;
                }
                /* bail out if the filesystem is corrupted.  */
-                if (error == EFSCORRUPTED)
+                if (error == -EFSCORRUPTED)
                        break;
                cond_resched();
@@ -652,11 +655,11 @@ xfs_inode_ag_iterator(
                xfs_perag_put(pag);
                if (error) {
                        last_error = error;
-                        if (error == EFSCORRUPTED)
+                        if (error == -EFSCORRUPTED)
                                break;
                }
        }
-        return XFS_ERROR(last_error);
+        return last_error;
 }
 int
@@ -680,11 +683,11 @@ xfs_inode_ag_iterator_tag(
                xfs_perag_put(pag);
                if (error) {
                        last_error = error;
-                        if (error == EFSCORRUPTED)
+                        if (error == -EFSCORRUPTED)
                                break;
                }
        }
-        return XFS_ERROR(last_error);
+        return last_error;
 }
 /*
@@ -944,7 +947,7 @@ restart:
         * see the stale flag set on the inode.
         */
        error = xfs_iflush(ip, &bp);
-        if (error == EAGAIN) {
+        if (error == -EAGAIN) {
                xfs_iunlock(ip, XFS_ILOCK_EXCL);
                /* backoff longer than in xfs_ifree_cluster */
                delay(2);
@@ -997,7 +1000,7 @@ out:
        xfs_iflags_clear(ip, XFS_IRECLAIM);
        xfs_iunlock(ip, XFS_ILOCK_EXCL);
        /*
-         * We could return EAGAIN here to make reclaim rescan the inode tree in
+         * We could return -EAGAIN here to make reclaim rescan the inode tree in
         * a short while. However, this just burns CPU time scanning the tree
         * waiting for IO to complete and the reclaim work never goes back to
         * the idle state. Instead, return 0 to let the next scheduled
@@ -1100,7 +1103,7 @@ restart:
                                if (!batch[i])
                                        continue;
                                error = xfs_reclaim_inode(batch[i], pag, flags);
-                                if (error && last_error != EFSCORRUPTED)
+                                if (error && last_error != -EFSCORRUPTED)
                                        last_error = error;
                        }
@@ -1129,7 +1132,7 @@ restart:
                trylock = 0;
                goto restart;
        }
-        return XFS_ERROR(last_error);
+        return last_error;
 }
 int
@@ -1203,6 +1206,30 @@ xfs_inode_match_id(
        return 1;
 }
+/*
+ * A union-based inode filtering algorithm. Process the inode if any of the
+ * criteria match. This is for global/internal scans only.
+ */
+STATIC int
+xfs_inode_match_id_union(
+        struct xfs_inode        *ip,
+        struct xfs_eofblocks    *eofb)
+{
+        if ((eofb->eof_flags & XFS_EOF_FLAGS_UID) &&
+            uid_eq(VFS_I(ip)->i_uid, eofb->eof_uid))
+                return 1;
+        if ((eofb->eof_flags & XFS_EOF_FLAGS_GID) &&
+            gid_eq(VFS_I(ip)->i_gid, eofb->eof_gid))
+                return 1;
+        if ((eofb->eof_flags & XFS_EOF_FLAGS_PRID) &&
+            xfs_get_projid(ip) == eofb->eof_prid)
+                return 1;
+        return 0;
+}
 STATIC int
 xfs_inode_free_eofblocks(
        struct xfs_inode        *ip,
@@ -1211,6 +1238,10 @@ xfs_inode_free_eofblocks(
 {
        int ret;
        struct xfs_eofblocks *eofb = args;
+        bool need_iolock = true;
+        int match;
+        ASSERT(!eofb || (eofb && eofb->eof_scan_owner != 0));
        if (!xfs_can_free_eofblocks(ip, false)) {
                /* inode could be preallocated or append-only */
@@ -1228,19 +1259,31 @@ xfs_inode_free_eofblocks(
                return 0;
        if (eofb) {
-                if (!xfs_inode_match_id(ip, eofb))
+                if (eofb->eof_flags & XFS_EOF_FLAGS_UNION)
+                        match = xfs_inode_match_id_union(ip, eofb);
+                else
+                        match = xfs_inode_match_id(ip, eofb);
+                if (!match)
                        return 0;
                /* skip the inode if the file size is too small */
                if (eofb->eof_flags & XFS_EOF_FLAGS_MINFILESIZE &&
                    XFS_ISIZE(ip) < eofb->eof_min_file_size)
                        return 0;
+                /*
+                 * A scan owner implies we already hold the iolock. Skip it in
+                 * xfs_free_eofblocks() to avoid deadlock. This also eliminates
+                 * the possibility of EAGAIN being returned.
+                 */
+                if (eofb->eof_scan_owner == ip->i_ino)
+                        need_iolock = false;
        }
-        ret = xfs_free_eofblocks(ip->i_mount, ip, true);
+        ret = xfs_free_eofblocks(ip->i_mount, ip, need_iolock);
        /* don't revisit the inode if we're not waiting */
-        if (ret == EAGAIN && !(flags & SYNC_WAIT))
+        if (ret == -EAGAIN && !(flags & SYNC_WAIT))
                ret = 0;
        return ret;
@@ -1260,6 +1303,55 @@ xfs_icache_free_eofblocks(
                                         eofb, XFS_ICI_EOFBLOCKS_TAG);
 }
+/*
+ * Run eofblocks scans on the quotas applicable to the inode. For inodes with
+ * multiple quotas, we don't know exactly which quota caused an allocation
+ * failure. We make a best effort by including each quota under low free space
+ * conditions (less than 1% free space) in the scan.
+ */
+int
+xfs_inode_free_quota_eofblocks(
+        struct xfs_inode *ip)
+{
+        int scan = 0;
+        struct xfs_eofblocks eofb = {0};
+        struct xfs_dquot *dq;
+        ASSERT(xfs_isilocked(ip, XFS_IOLOCK_EXCL));
+        /*
+         * Set the scan owner to avoid a potential livelock. Otherwise, the scan
+         * can repeatedly trylock on the inode we're currently processing. We
+         * run a sync scan to increase effectiveness and use the union filter to
+         * cover all applicable quotas in a single scan.
+         */
+        eofb.eof_scan_owner = ip->i_ino;
+        eofb.eof_flags = XFS_EOF_FLAGS_UNION|XFS_EOF_FLAGS_SYNC;
+        if (XFS_IS_UQUOTA_ENFORCED(ip->i_mount)) {
+                dq = xfs_inode_dquot(ip, XFS_DQ_USER);
+                if (dq && xfs_dquot_lowsp(dq)) {
+                        eofb.eof_uid = VFS_I(ip)->i_uid;
+                        eofb.eof_flags |= XFS_EOF_FLAGS_UID;
+                        scan = 1;
+                }
+        }
+        if (XFS_IS_GQUOTA_ENFORCED(ip->i_mount)) {
+                dq = xfs_inode_dquot(ip, XFS_DQ_GROUP);
+                if (dq && xfs_dquot_lowsp(dq)) {
+                        eofb.eof_gid = VFS_I(ip)->i_gid;
+                        eofb.eof_flags |= XFS_EOF_FLAGS_GID;
+                        scan = 1;
+                }
+        }
+        if (scan)
+                xfs_icache_free_eofblocks(ip->i_mount, &eofb);
+        return scan;
+}
 void
 xfs_inode_set_eofblocks_tag(
        xfs_inode_t     *ip)
diff --git a/fs/xfs/xfs_icache.h b/fs/xfs/xfs_icache.h
index 9cf017b899be..46748b86b12f 100644
--- a/fs/xfs/xfs_icache.h
+++ b/fs/xfs/xfs_icache.h
@@ -27,6 +27,7 @@ struct xfs_eofblocks {
        kgid_t          eof_gid;
        prid_t          eof_prid;
        __u64           eof_min_file_size;
+        xfs_ino_t       eof_scan_owner;
 };
 #define SYNC_WAIT               0x0001  /* wait for i/o to complete */
@@ -57,6 +58,7 @@ void xfs_inode_set_reclaim_tag(struct xfs_inode *ip);
 void xfs_inode_set_eofblocks_tag(struct xfs_inode *ip);
 void xfs_inode_clear_eofblocks_tag(struct xfs_inode *ip);
 int xfs_icache_free_eofblocks(struct xfs_mount *, struct xfs_eofblocks *);
+int xfs_inode_free_quota_eofblocks(struct xfs_inode *ip);
 void xfs_eofblocks_worker(struct work_struct *);
 int xfs_inode_ag_iterator(struct xfs_mount *mp,
@@ -72,31 +74,32 @@ xfs_fs_eofblocks_from_user(
        struct xfs_eofblocks            *dst)
 {
        if (src->eof_version != XFS_EOFBLOCKS_VERSION)
-                return EINVAL;
+                return -EINVAL;
        if (src->eof_flags & ~XFS_EOF_FLAGS_VALID)
-                return EINVAL;
+                return -EINVAL;
        if (memchr_inv(&src->pad32, 0, sizeof(src->pad32)) ||
            memchr_inv(src->pad64, 0, sizeof(src->pad64)))
-                return EINVAL;
+                return -EINVAL;
        dst->eof_flags = src->eof_flags;
        dst->eof_prid = src->eof_prid;
        dst->eof_min_file_size = src->eof_min_file_size;
+        dst->eof_scan_owner = NULLFSINO;
        dst->eof_uid = INVALID_UID;
        if (src->eof_flags & XFS_EOF_FLAGS_UID) {
                dst->eof_uid = make_kuid(current_user_ns(), src->eof_uid);
                if (!uid_valid(dst->eof_uid))
-                        return EINVAL;
+                        return -EINVAL;
        }
        dst->eof_gid = INVALID_GID;
        if (src->eof_flags & XFS_EOF_FLAGS_GID) {
                dst->eof_gid = make_kgid(current_user_ns(), src->eof_gid);
                if (!gid_valid(dst->eof_gid))
-                        return EINVAL;
+                        return -EINVAL;
        }
        return 0;
 }
diff --git a/fs/xfs/xfs_inode.c b/fs/xfs/xfs_inode.c
index a6115fe1ac94..fea3c92fb3f0 100644
--- a/fs/xfs/xfs_inode.c
+++ b/fs/xfs/xfs_inode.c
@@ -583,7 +583,7 @@ xfs_lookup(
        trace_xfs_lookup(dp, name);
        if (XFS_FORCED_SHUTDOWN(dp->i_mount))
-                return XFS_ERROR(EIO);
+                return -EIO;
        lock_mode = xfs_ilock_data_map_shared(dp);
        error = xfs_dir_lookup(NULL, dp, name, &inum, ci_name);
@@ -893,7 +893,7 @@ xfs_dir_ialloc(
        }
        if (!ialloc_context && !ip) {
                *ipp = NULL;
-                return XFS_ERROR(ENOSPC);
+                return -ENOSPC;
        }
        /*
@@ -1088,7 +1088,7 @@ xfs_create(
        trace_xfs_create(dp, name);
        if (XFS_FORCED_SHUTDOWN(mp))
-                return XFS_ERROR(EIO);
+                return -EIO;
        prid = xfs_get_initial_prid(dp);
@@ -1125,12 +1125,12 @@ xfs_create(
         */
        tres.tr_logflags = XFS_TRANS_PERM_LOG_RES;
        error = xfs_trans_reserve(tp, &tres, resblks, 0);
-        if (error == ENOSPC) {
+        if (error == -ENOSPC) {
                /* flush outstanding delalloc blocks and retry */
                xfs_flush_inodes(mp);
                error = xfs_trans_reserve(tp, &tres, resblks, 0);
        }
-        if (error == ENOSPC) {
+        if (error == -ENOSPC) {
                /* No space at all so try a "no-allocation" reservation */
                resblks = 0;
                error = xfs_trans_reserve(tp, &tres, 0, 0);
@@ -1165,7 +1165,7 @@ xfs_create(
        error = xfs_dir_ialloc(&tp, dp, mode, is_dir ? 2 : 1, rdev,
                               prid, resblks > 0, &ip, &committed);
        if (error) {
-                if (error == ENOSPC)
+                if (error == -ENOSPC)
                        goto out_trans_cancel;
                goto out_trans_abort;
        }
@@ -1184,7 +1184,7 @@ xfs_create(
                                        &first_block, &free_list, resblks ?
                                        resblks - XFS_IALLOC_SPACE_RES(mp) : 0);
        if (error) {
-                ASSERT(error != ENOSPC);
+                ASSERT(error != -ENOSPC);
                goto out_trans_abort;
        }
        xfs_trans_ichgtime(tp, dp, XFS_ICHGTIME_MOD | XFS_ICHGTIME_CHG);
@@ -1274,7 +1274,7 @@ xfs_create_tmpfile(
        uint                    resblks;
        if (XFS_FORCED_SHUTDOWN(mp))
-                return XFS_ERROR(EIO);
+                return -EIO;
        prid = xfs_get_initial_prid(dp);
@@ -1293,7 +1293,7 @@ xfs_create_tmpfile(
        tres = &M_RES(mp)->tr_create_tmpfile;
        error = xfs_trans_reserve(tp, tres, resblks, 0);
-        if (error == ENOSPC) {
+        if (error == -ENOSPC) {
                /* No space at all so try a "no-allocation" reservation */
                resblks = 0;
                error = xfs_trans_reserve(tp, tres, 0, 0);
@@ -1311,7 +1311,7 @@ xfs_create_tmpfile(
        error = xfs_dir_ialloc(&tp, dp, mode, 1, 0,
                                prid, resblks > 0, &ip, NULL);
        if (error) {
-                if (error == ENOSPC)
+                if (error == -ENOSPC)
                        goto out_trans_cancel;
                goto out_trans_abort;
        }
@@ -1382,7 +1382,7 @@ xfs_link(
        ASSERT(!S_ISDIR(sip->i_d.di_mode));
        if (XFS_FORCED_SHUTDOWN(mp))
-                return XFS_ERROR(EIO);
+                return -EIO;
        error = xfs_qm_dqattach(sip, 0);
        if (error)
@@ -1396,7 +1396,7 @@ xfs_link(
        cancel_flags = XFS_TRANS_RELEASE_LOG_RES;
        resblks = XFS_LINK_SPACE_RES(mp, target_name->len);
        error = xfs_trans_reserve(tp, &M_RES(mp)->tr_link, resblks, 0);
-        if (error == ENOSPC) {
+        if (error == -ENOSPC) {
                resblks = 0;
                error = xfs_trans_reserve(tp, &M_RES(mp)->tr_link, 0, 0);
        }
@@ -1417,7 +1417,7 @@ xfs_link(
         */
        if (unlikely((tdp->i_d.di_flags & XFS_DIFLAG_PROJINHERIT) &&
                     (xfs_get_projid(tdp) != xfs_get_projid(sip)))) {
-                error = XFS_ERROR(EXDEV);
+                error = -EXDEV;
                goto error_return;
        }
@@ -1635,8 +1635,8 @@ xfs_release(
                truncated = xfs_iflags_test_and_clear(ip, XFS_ITRUNCATED);
                if (truncated) {
                        xfs_iflags_clear(ip, XFS_IDIRTY_RELEASE);
-                        if (VN_DIRTY(VFS_I(ip)) && ip->i_delayed_blks > 0) {
+                        if (ip->i_delayed_blks > 0) {
-                                error = -filemap_flush(VFS_I(ip)->i_mapping);
+                                error = filemap_flush(VFS_I(ip)->i_mapping);
                                if (error)
                                        return error;
                        }
@@ -1673,7 +1673,7 @@ xfs_release(
                        return 0;
                error = xfs_free_eofblocks(mp, ip, true);
-                if (error && error != EAGAIN)
+                if (error && error != -EAGAIN)
                        return error;
                /* delalloc blocks after truncation means it really is dirty */
@@ -1772,7 +1772,7 @@ xfs_inactive_ifree(
        error = xfs_trans_reserve(tp, &M_RES(mp)->tr_ifree,
                                  XFS_IFREE_SPACE_RES(mp), 0);
        if (error) {
-                if (error == ENOSPC) {
+                if (error == -ENOSPC) {
                        xfs_warn_ratelimited(mp,
                        "Failed to remove inode(s) from unlinked list. "
                        "Please free space, unmount and run xfs_repair.");
@@ -2219,7 +2219,7 @@ xfs_ifree_cluster(
                                        XBF_UNMAPPED);
                if (!bp)
-                        return ENOMEM;
+                        return -ENOMEM;
                /*
                 * This buffer may not have been correctly initialised as we
@@ -2491,7 +2491,7 @@ xfs_remove(
        trace_xfs_remove(dp, name);
        if (XFS_FORCED_SHUTDOWN(mp))
-                return XFS_ERROR(EIO);
+                return -EIO;
        error = xfs_qm_dqattach(dp, 0);
        if (error)
@@ -2521,12 +2521,12 @@ xfs_remove(
         */
        resblks = XFS_REMOVE_SPACE_RES(mp);
        error = xfs_trans_reserve(tp, &M_RES(mp)->tr_remove, resblks, 0);
-        if (error == ENOSPC) {
+        if (error == -ENOSPC) {
                resblks = 0;
                error = xfs_trans_reserve(tp, &M_RES(mp)->tr_remove, 0, 0);
        }
        if (error) {
-                ASSERT(error != ENOSPC);
+                ASSERT(error != -ENOSPC);
                cancel_flags = 0;
                goto out_trans_cancel;
        }
@@ -2543,11 +2543,11 @@ xfs_remove(
        if (is_dir) {
                ASSERT(ip->i_d.di_nlink >= 2);
                if (ip->i_d.di_nlink != 2) {
-                        error = XFS_ERROR(ENOTEMPTY);
+                        error = -ENOTEMPTY;
                        goto out_trans_cancel;
                }
                if (!xfs_dir_isempty(ip)) {
-                        error = XFS_ERROR(ENOTEMPTY);
+                        error = -ENOTEMPTY;
                        goto out_trans_cancel;
                }
@@ -2582,7 +2582,7 @@ xfs_remove(
        error = xfs_dir_removename(tp, dp, name, ip->i_ino,
                                        &first_block, &free_list, resblks);
        if (error) {
-                ASSERT(error != ENOENT);
+                ASSERT(error != -ENOENT);
                goto out_bmap_cancel;
        }
@@ -2702,7 +2702,7 @@ xfs_rename(
        cancel_flags = XFS_TRANS_RELEASE_LOG_RES;
        spaceres = XFS_RENAME_SPACE_RES(mp, target_name->len);
        error = xfs_trans_reserve(tp, &M_RES(mp)->tr_rename, spaceres, 0);
-        if (error == ENOSPC) {
+        if (error == -ENOSPC) {
                spaceres = 0;
                error = xfs_trans_reserve(tp, &M_RES(mp)->tr_rename, 0, 0);
        }
@@ -2747,7 +2747,7 @@ xfs_rename(
         */
        if (unlikely((target_dp->i_d.di_flags & XFS_DIFLAG_PROJINHERIT) &&
                     (xfs_get_projid(target_dp) != xfs_get_projid(src_ip)))) {
-                error = XFS_ERROR(EXDEV);
+                error = -EXDEV;
                goto error_return;
        }
@@ -2770,7 +2770,7 @@ xfs_rename(
                error = xfs_dir_createname(tp, target_dp, target_name,
                                                src_ip->i_ino, &first_block,
                                                &free_list, spaceres);
-                if (error == ENOSPC)
+                if (error == -ENOSPC)
                        goto error_return;
                if (error)
                        goto abort_return;
@@ -2795,7 +2795,7 @@ xfs_rename(
                         */
                        if (!(xfs_dir_isempty(target_ip)) ||
                            (target_ip->i_d.di_nlink > 2)) {
-                                error = XFS_ERROR(EEXIST);
+                                error = -EEXIST;
                                goto error_return;
                        }
                }
@@ -2847,7 +2847,7 @@ xfs_rename(
                error = xfs_dir_replace(tp, src_ip, &xfs_name_dotdot,
                                        target_dp->i_ino,
                                        &first_block, &free_list, spaceres);
-                ASSERT(error != EEXIST);
+                ASSERT(error != -EEXIST);
                if (error)
                        goto abort_return;
        }
@@ -3055,7 +3055,7 @@ cluster_corrupt_out:
                if (bp->b_iodone) {
                        XFS_BUF_UNDONE(bp);
                        xfs_buf_stale(bp);
-                        xfs_buf_ioerror(bp, EIO);
+                        xfs_buf_ioerror(bp, -EIO);
                        xfs_buf_ioend(bp, 0);
                } else {
                        xfs_buf_stale(bp);
@@ -3069,7 +3069,7 @@ cluster_corrupt_out:
        xfs_iflush_abort(iq, false);
        kmem_free(ilist);
        xfs_perag_put(pag);
-        return XFS_ERROR(EFSCORRUPTED);
+        return -EFSCORRUPTED;
 }
 /*
@@ -3124,7 +3124,7 @@ xfs_iflush(
         * as we wait for an empty AIL as part of the unmount process.
         */
        if (XFS_FORCED_SHUTDOWN(mp)) {
-                error = XFS_ERROR(EIO);
+                error = -EIO;
                goto abort_out;
        }
@@ -3167,7 +3167,7 @@ corrupt_out:
        xfs_buf_relse(bp);
        xfs_force_shutdown(mp, SHUTDOWN_CORRUPT_INCORE);
 cluster_corrupt_out:
-        error = XFS_ERROR(EFSCORRUPTED);
+        error = -EFSCORRUPTED;
 abort_out:
        /*
         * Unlocks the flush lock
@@ -3331,5 +3331,5 @@ xfs_iflush_int(
        return 0;
 corrupt_out:
-        return XFS_ERROR(EFSCORRUPTED);
+        return -EFSCORRUPTED;
 }
diff --git a/fs/xfs/xfs_inode.h b/fs/xfs/xfs_inode.h
index f72bffa67266..c10e3fadd9af 100644
--- a/fs/xfs/xfs_inode.h
+++ b/fs/xfs/xfs_inode.h
@@ -398,4 +398,14 @@ do { \
 extern struct kmem_zone *xfs_inode_zone;
+/*
+ * Flags for read/write calls
+ */
+#define XFS_IO_ISDIRECT 0x00001         /* bypass page cache */
+#define XFS_IO_INVIS    0x00002         /* don't update inode timestamps */
+#define XFS_IO_FLAGS \
+        { XFS_IO_ISDIRECT,      "DIRECT" }, \
+        { XFS_IO_INVIS,         "INVIS"}
 #endif  /* __XFS_INODE_H__ */
diff --git a/fs/xfs/xfs_inode_item.c b/fs/xfs/xfs_inode_item.c
index a640137b3573..de5a7be36e60 100644
--- a/fs/xfs/xfs_inode_item.c
+++ b/fs/xfs/xfs_inode_item.c
@@ -788,5 +788,5 @@ xfs_inode_item_format_convert(
                in_f->ilf_boffset = in_f64->ilf_boffset;
                return 0;
        }
-        return EFSCORRUPTED;
+        return -EFSCORRUPTED;
 }
diff --git a/fs/xfs/xfs_ioctl.c b/fs/xfs/xfs_ioctl.c
index 8bc1bbce7451..3799695b9249 100644
--- a/fs/xfs/xfs_ioctl.c
+++ b/fs/xfs/xfs_ioctl.c
@@ -207,7 +207,7 @@ xfs_open_by_handle(
        struct path             path;
        if (!capable(CAP_SYS_ADMIN))
-                return -XFS_ERROR(EPERM);
+                return -EPERM;
        dentry = xfs_handlereq_to_dentry(parfilp, hreq);
        if (IS_ERR(dentry))
@@ -216,7 +216,7 @@ xfs_open_by_handle(
        /* Restrict xfs_open_by_handle to directories & regular files. */
        if (!(S_ISREG(inode->i_mode) || S_ISDIR(inode->i_mode))) {
-                error = -XFS_ERROR(EPERM);
+                error = -EPERM;
                goto out_dput;
        }
@@ -228,18 +228,18 @@ xfs_open_by_handle(
        fmode = OPEN_FMODE(permflag);
        if ((!(permflag & O_APPEND) || (permflag & O_TRUNC)) &&
            (fmode & FMODE_WRITE) && IS_APPEND(inode)) {
-                error = -XFS_ERROR(EPERM);
+                error = -EPERM;
                goto out_dput;
        }
        if ((fmode & FMODE_WRITE) && IS_IMMUTABLE(inode)) {
-                error = -XFS_ERROR(EACCES);
+                error = -EACCES;
                goto out_dput;
        }
        /* Can't write directories. */
        if (S_ISDIR(inode->i_mode) && (fmode & FMODE_WRITE)) {
-                error = -XFS_ERROR(EISDIR);
+                error = -EISDIR;
                goto out_dput;
        }
@@ -282,7 +282,7 @@ xfs_readlink_by_handle(
        int                     error;
        if (!capable(CAP_SYS_ADMIN))
-                return -XFS_ERROR(EPERM);
+                return -EPERM;
        dentry = xfs_handlereq_to_dentry(parfilp, hreq);
        if (IS_ERR(dentry))
@@ -290,22 +290,22 @@ xfs_readlink_by_handle(
        /* Restrict this handle operation to symlinks only. */
        if (!S_ISLNK(dentry->d_inode->i_mode)) {
-                error = -XFS_ERROR(EINVAL);
+                error = -EINVAL;
                goto out_dput;
        }
        if (copy_from_user(&olen, hreq->ohandlen, sizeof(__u32))) {
-                error = -XFS_ERROR(EFAULT);
+                error = -EFAULT;
                goto out_dput;
        }
        link = kmalloc(MAXPATHLEN+1, GFP_KERNEL);
        if (!link) {
-                error = -XFS_ERROR(ENOMEM);
+                error = -ENOMEM;
                goto out_dput;
        }
-        error = -xfs_readlink(XFS_I(dentry->d_inode), link);
+        error = xfs_readlink(XFS_I(dentry->d_inode), link);
        if (error)
                goto out_kfree;
        error = readlink_copy(hreq->ohandle, olen, link);
@@ -330,10 +330,10 @@ xfs_set_dmattrs(
        int             error;
        if (!capable(CAP_SYS_ADMIN))
-                return XFS_ERROR(EPERM);
+                return -EPERM;
        if (XFS_FORCED_SHUTDOWN(mp))
-                return XFS_ERROR(EIO);
+                return -EIO;
        tp = xfs_trans_alloc(mp, XFS_TRANS_SET_DMATTRS);
        error = xfs_trans_reserve(tp, &M_RES(mp)->tr_ichange, 0, 0);
@@ -364,9 +364,9 @@ xfs_fssetdm_by_handle(
        struct dentry           *dentry;
        if (!capable(CAP_MKNOD))
-                return -XFS_ERROR(EPERM);
+                return -EPERM;
        if (copy_from_user(&dmhreq, arg, sizeof(xfs_fsop_setdm_handlereq_t)))
-                return -XFS_ERROR(EFAULT);
+                return -EFAULT;
        error = mnt_want_write_file(parfilp);
        if (error)
@@ -379,16 +379,16 @@ xfs_fssetdm_by_handle(
        }
        if (IS_IMMUTABLE(dentry->d_inode) || IS_APPEND(dentry->d_inode)) {
-                error = -XFS_ERROR(EPERM);
+                error = -EPERM;
                goto out;
        }
        if (copy_from_user(&fsd, dmhreq.data, sizeof(fsd))) {
-                error = -XFS_ERROR(EFAULT);
+                error = -EFAULT;
                goto out;
        }
-        error = -xfs_set_dmattrs(XFS_I(dentry->d_inode), fsd.fsd_dmevmask,
+        error = xfs_set_dmattrs(XFS_I(dentry->d_inode), fsd.fsd_dmevmask,
                                 fsd.fsd_dmstate);
 out:
@@ -409,18 +409,18 @@ xfs_attrlist_by_handle(
        char                    *kbuf;
        if (!capable(CAP_SYS_ADMIN))
-                return -XFS_ERROR(EPERM);
+                return -EPERM;
        if (copy_from_user(&al_hreq, arg, sizeof(xfs_fsop_attrlist_handlereq_t)))
-                return -XFS_ERROR(EFAULT);
+                return -EFAULT;
        if (al_hreq.buflen < sizeof(struct attrlist) ||
            al_hreq.buflen > XATTR_LIST_MAX)
-                return -XFS_ERROR(EINVAL);
+                return -EINVAL;
        /*
         * Reject flags, only allow namespaces.
         */
        if (al_hreq.flags & ~(ATTR_ROOT | ATTR_SECURE))
-                return -XFS_ERROR(EINVAL);
+                return -EINVAL;
        dentry = xfs_handlereq_to_dentry(parfilp, &al_hreq.hreq);
        if (IS_ERR(dentry))
@@ -431,7 +431,7 @@ xfs_attrlist_by_handle(
                goto out_dput;
        cursor = (attrlist_cursor_kern_t *)&al_hreq.pos;
-        error = -xfs_attr_list(XFS_I(dentry->d_inode), kbuf, al_hreq.buflen,
+        error = xfs_attr_list(XFS_I(dentry->d_inode), kbuf, al_hreq.buflen,
                                        al_hreq.flags, cursor);
        if (error)
                goto out_kfree;
@@ -455,20 +455,20 @@ xfs_attrmulti_attr_get(
        __uint32_t              flags)
 {
        unsigned char           *kbuf;
-        int                     error = EFAULT;
+        int                     error = -EFAULT;
        if (*len > XATTR_SIZE_MAX)
-                return EINVAL;
+                return -EINVAL;
        kbuf = kmem_zalloc_large(*len, KM_SLEEP);
        if (!kbuf)
-                return ENOMEM;
+                return -ENOMEM;
        error = xfs_attr_get(XFS_I(inode), name, kbuf, (int *)len, flags);
        if (error)
                goto out_kfree;
        if (copy_to_user(ubuf, kbuf, *len))
-                error = EFAULT;
+                error = -EFAULT;
 out_kfree:
        kmem_free(kbuf);
@@ -484,20 +484,17 @@ xfs_attrmulti_attr_set(
        __uint32_t              flags)
 {
        unsigned char           *kbuf;
-        int                     error = EFAULT;
        if (IS_IMMUTABLE(inode) || IS_APPEND(inode))
-                return EPERM;
+                return -EPERM;
        if (len > XATTR_SIZE_MAX)
-                return EINVAL;
+                return -EINVAL;
        kbuf = memdup_user(ubuf, len);
        if (IS_ERR(kbuf))
                return PTR_ERR(kbuf);
-        error = xfs_attr_set(XFS_I(inode), name, kbuf, len, flags);
+        return xfs_attr_set(XFS_I(inode), name, kbuf, len, flags);
-        return error;
 }
 int
@@ -507,7 +504,7 @@ xfs_attrmulti_attr_remove(
        __uint32_t              flags)
 {
        if (IS_IMMUTABLE(inode) || IS_APPEND(inode))
-                return EPERM;
+                return -EPERM;
        return xfs_attr_remove(XFS_I(inode), name, flags);
 }
@@ -524,9 +521,9 @@ xfs_attrmulti_by_handle(
        unsigned char           *attr_name;
        if (!capable(CAP_SYS_ADMIN))
-                return -XFS_ERROR(EPERM);
+                return -EPERM;
        if (copy_from_user(&am_hreq, arg, sizeof(xfs_fsop_attrmulti_handlereq_t)))
-                return -XFS_ERROR(EFAULT);
+                return -EFAULT;
        /* overflow check */
        if (am_hreq.opcount >= INT_MAX / sizeof(xfs_attr_multiop_t))
@@ -536,18 +533,18 @@ xfs_attrmulti_by_handle(
        if (IS_ERR(dentry))
                return PTR_ERR(dentry);
-        error = E2BIG;
+        error = -E2BIG;
        size = am_hreq.opcount * sizeof(xfs_attr_multiop_t);
        if (!size || size > 16 * PAGE_SIZE)
                goto out_dput;
        ops = memdup_user(am_hreq.ops, size);
        if (IS_ERR(ops)) {
-                error = -PTR_ERR(ops);
+                error = PTR_ERR(ops);
                goto out_dput;
        }
-        error = ENOMEM;
+        error = -ENOMEM;
        attr_name = kmalloc(MAXNAMELEN, GFP_KERNEL);
        if (!attr_name)
                goto out_kfree_ops;
@@ -557,7 +554,7 @@ xfs_attrmulti_by_handle(
                ops[i].am_error = strncpy_from_user((char *)attr_name,
                                ops[i].am_attrname, MAXNAMELEN);
                if (ops[i].am_error == 0 || ops[i].am_error == MAXNAMELEN)
-                        error = ERANGE;
+                        error = -ERANGE;
                if (ops[i].am_error < 0)
                        break;
@@ -588,19 +585,19 @@ xfs_attrmulti_by_handle(
                        mnt_drop_write_file(parfilp);
                        break;
                default:
-                        ops[i].am_error = EINVAL;
+                        ops[i].am_error = -EINVAL;
                }
        }
        if (copy_to_user(am_hreq.ops, ops, size))
-                error = XFS_ERROR(EFAULT);
+                error = -EFAULT;
        kfree(attr_name);
 out_kfree_ops:
        kfree(ops);
 out_dput:
        dput(dentry);
-        return -error;
+        return error;
 }
 int
@@ -625,16 +622,16 @@ xfs_ioc_space(
         */
        if (!xfs_sb_version_hasextflgbit(&ip->i_mount->m_sb) &&
            !capable(CAP_SYS_ADMIN))
-                return -XFS_ERROR(EPERM);
+                return -EPERM;
        if (inode->i_flags & (S_IMMUTABLE|S_APPEND))
-                return -XFS_ERROR(EPERM);
+                return -EPERM;
        if (!(filp->f_mode & FMODE_WRITE))
-                return -XFS_ERROR(EBADF);
+                return -EBADF;
        if (!S_ISREG(inode->i_mode))
-                return -XFS_ERROR(EINVAL);
+                return -EINVAL;
        error = mnt_want_write_file(filp);
        if (error)
@@ -652,7 +649,7 @@ xfs_ioc_space(
                bf->l_start += XFS_ISIZE(ip);
                break;
        default:
-                error = XFS_ERROR(EINVAL);
+                error = -EINVAL;
                goto out_unlock;
        }
@@ -669,7 +666,7 @@ xfs_ioc_space(
        case XFS_IOC_UNRESVSP:
        case XFS_IOC_UNRESVSP64:
                if (bf->l_len <= 0) {
-                        error = XFS_ERROR(EINVAL);
+                        error = -EINVAL;
                        goto out_unlock;
                }
                break;
@@ -682,7 +679,7 @@ xfs_ioc_space(
            bf->l_start > mp->m_super->s_maxbytes ||
            bf->l_start + bf->l_len < 0 ||
            bf->l_start + bf->l_len >= mp->m_super->s_maxbytes) {
-                error = XFS_ERROR(EINVAL);
+                error = -EINVAL;
                goto out_unlock;
        }
@@ -723,7 +720,7 @@ xfs_ioc_space(
                break;
        default:
                ASSERT(0);
-                error = XFS_ERROR(EINVAL);
+                error = -EINVAL;
        }
        if (error)
@@ -739,7 +736,7 @@ xfs_ioc_space(
        xfs_ilock(ip, XFS_ILOCK_EXCL);
        xfs_trans_ijoin(tp, ip, XFS_ILOCK_EXCL);
-        if (!(ioflags & IO_INVIS)) {
+        if (!(ioflags & XFS_IO_INVIS)) {
                ip->i_d.di_mode &= ~S_ISUID;
                if (ip->i_d.di_mode & S_IXGRP)
                        ip->i_d.di_mode &= ~S_ISGID;
@@ -759,7 +756,7 @@ xfs_ioc_space(
 out_unlock:
        xfs_iunlock(ip, XFS_IOLOCK_EXCL);
        mnt_drop_write_file(filp);
-        return -error;
+        return error;
 }
 STATIC int
@@ -781,41 +778,41 @@ xfs_ioc_bulkstat(
                return -EPERM;
        if (XFS_FORCED_SHUTDOWN(mp))
-                return -XFS_ERROR(EIO);
+                return -EIO;
        if (copy_from_user(&bulkreq, arg, sizeof(xfs_fsop_bulkreq_t)))
-                return -XFS_ERROR(EFAULT);
+                return -EFAULT;
        if (copy_from_user(&inlast, bulkreq.lastip, sizeof(__s64)))
-                return -XFS_ERROR(EFAULT);
+                return -EFAULT;
        if ((count = bulkreq.icount) <= 0)
-                return -XFS_ERROR(EINVAL);
+                return -EINVAL;
        if (bulkreq.ubuffer == NULL)
-                return -XFS_ERROR(EINVAL);
+                return -EINVAL;
        if (cmd == XFS_IOC_FSINUMBERS)
                error = xfs_inumbers(mp, &inlast, &count,
                                        bulkreq.ubuffer, xfs_inumbers_fmt);
        else if (cmd == XFS_IOC_FSBULKSTAT_SINGLE)
-                error = xfs_bulkstat_single(mp, &inlast,
+                error = xfs_bulkstat_one(mp, inlast, bulkreq.ubuffer,
-                                                bulkreq.ubuffer, &done);
+                                        sizeof(xfs_bstat_t), NULL, &done);
        else    /* XFS_IOC_FSBULKSTAT */
                error = xfs_bulkstat(mp, &inlast, &count, xfs_bulkstat_one,
                                     sizeof(xfs_bstat_t), bulkreq.ubuffer,
                                     &done);
        if (error)
-                return -error;
+                return error;
        if (bulkreq.ocount != NULL) {
                if (copy_to_user(bulkreq.lastip, &inlast,
                                                sizeof(xfs_ino_t)))
-                        return -XFS_ERROR(EFAULT);
+                        return -EFAULT;
                if (copy_to_user(bulkreq.ocount, &count, sizeof(count)))
-                        return -XFS_ERROR(EFAULT);
+                        return -EFAULT;
        }
        return 0;
@@ -831,7 +828,7 @@ xfs_ioc_fsgeometry_v1(
        error = xfs_fs_geometry(mp, &fsgeo, 3);
        if (error)
-                return -error;
+                return error;
        /*
         * Caller should have passed an argument of type
@@ -839,7 +836,7 @@ xfs_ioc_fsgeometry_v1(
         * xfs_fsop_geom_t that xfs_fs_geometry() fills in.
         */
        if (copy_to_user(arg, &fsgeo, sizeof(xfs_fsop_geom_v1_t)))
-                return -XFS_ERROR(EFAULT);
+                return -EFAULT;
        return 0;
 }
@@ -853,10 +850,10 @@ xfs_ioc_fsgeometry(
        error = xfs_fs_geometry(mp, &fsgeo, 4);
        if (error)
-                return -error;
+                return error;
        if (copy_to_user(arg, &fsgeo, sizeof(fsgeo)))
-                return -XFS_ERROR(EFAULT);
+                return -EFAULT;
        return 0;
 }
@@ -1041,16 +1038,16 @@ xfs_ioctl_setattr(
        trace_xfs_ioctl_setattr(ip);
        if (mp->m_flags & XFS_MOUNT_RDONLY)
-                return XFS_ERROR(EROFS);
+                return -EROFS;
        if (XFS_FORCED_SHUTDOWN(mp))
-                return XFS_ERROR(EIO);
+                return -EIO;
        /*
         * Disallow 32bit project ids when projid32bit feature is not enabled.
         */
        if ((mask & FSX_PROJID) && (fa->fsx_projid > (__uint16_t)-1) &&
                        !xfs_sb_version_hasprojid32bit(&ip->i_mount->m_sb))
-                return XFS_ERROR(EINVAL);
+                return -EINVAL;
        /*
         * If disk quotas is on, we make sure that the dquots do exist on disk,
@@ -1088,7 +1085,7 @@ xfs_ioctl_setattr(
         * CAP_FSETID capability is applicable.
         */
        if (!inode_owner_or_capable(VFS_I(ip))) {
-                code = XFS_ERROR(EPERM);
+                code = -EPERM;
                goto error_return;
        }
@@ -1099,7 +1096,7 @@ xfs_ioctl_setattr(
         */
        if (mask & FSX_PROJID) {
                if (current_user_ns() != &init_user_ns) {
-                        code = XFS_ERROR(EINVAL);
+                        code = -EINVAL;
                        goto error_return;
                }
@@ -1122,7 +1119,7 @@ xfs_ioctl_setattr(
                if (ip->i_d.di_nextents &&
                    ((ip->i_d.di_extsize << mp->m_sb.sb_blocklog) !=
                     fa->fsx_extsize)) {
-                        code = XFS_ERROR(EINVAL);       /* EFBIG? */
+                        code = -EINVAL; /* EFBIG? */
                        goto error_return;
                }
@@ -1141,7 +1138,7 @@ xfs_ioctl_setattr(
                        extsize_fsb = XFS_B_TO_FSB(mp, fa->fsx_extsize);
                        if (extsize_fsb > MAXEXTLEN) {
-                                code = XFS_ERROR(EINVAL);
+                                code = -EINVAL;
                                goto error_return;
                        }
@@ -1153,13 +1150,13 @@ xfs_ioctl_setattr(
                        } else {
                                size = mp->m_sb.sb_blocksize;
                                if (extsize_fsb > mp->m_sb.sb_agblocks / 2) {
-                                        code = XFS_ERROR(EINVAL);
+                                        code = -EINVAL;
                                        goto error_return;
                                }
                        }
                        if (fa->fsx_extsize % size) {
-                                code = XFS_ERROR(EINVAL);
+                                code = -EINVAL;
                                goto error_return;
                        }
                }
@@ -1173,7 +1170,7 @@ xfs_ioctl_setattr(
                if ((ip->i_d.di_nextents || ip->i_delayed_blks) &&
                    (XFS_IS_REALTIME_INODE(ip)) !=
                    (fa->fsx_xflags & XFS_XFLAG_REALTIME)) {
-                        code = XFS_ERROR(EINVAL);       /* EFBIG? */
+                        code = -EINVAL; /* EFBIG? */
                        goto error_return;
                }
@@ -1184,7 +1181,7 @@ xfs_ioctl_setattr(
                        if ((mp->m_sb.sb_rblocks == 0) ||
                            (mp->m_sb.sb_rextsize == 0) ||
                            (ip->i_d.di_extsize % mp->m_sb.sb_rextsize)) {
-                                code = XFS_ERROR(EINVAL);
+                                code = -EINVAL;
                                goto error_return;
                        }
                }
@@ -1198,7 +1195,7 @@ xfs_ioctl_setattr(
                     (fa->fsx_xflags &
                                (XFS_XFLAG_IMMUTABLE | XFS_XFLAG_APPEND))) &&
                    !capable(CAP_LINUX_IMMUTABLE)) {
-                        code = XFS_ERROR(EPERM);
+                        code = -EPERM;
                        goto error_return;
                }
        }
@@ -1301,7 +1298,7 @@ xfs_ioc_fssetxattr(
                return error;
        error = xfs_ioctl_setattr(ip, &fa, mask);
        mnt_drop_write_file(filp);
-        return -error;
+        return error;
 }
 STATIC int
@@ -1346,7 +1343,7 @@ xfs_ioc_setxflags(
                return error;
        error = xfs_ioctl_setattr(ip, &fa, mask);
        mnt_drop_write_file(filp);
-        return -error;
+        return error;
 }
 STATIC int
@@ -1356,7 +1353,7 @@ xfs_getbmap_format(void **ap, struct getbmapx *bmv, int *full)
        /* copy only getbmap portion (not getbmapx) */
        if (copy_to_user(base, bmv, sizeof(struct getbmap)))
-                return XFS_ERROR(EFAULT);
+                return -EFAULT;
        *ap += sizeof(struct getbmap);
        return 0;
@@ -1373,23 +1370,23 @@ xfs_ioc_getbmap(
        int                     error;
        if (copy_from_user(&bmx, arg, sizeof(struct getbmapx)))
-                return -XFS_ERROR(EFAULT);
+                return -EFAULT;
        if (bmx.bmv_count < 2)
-                return -XFS_ERROR(EINVAL);
+                return -EINVAL;
        bmx.bmv_iflags = (cmd == XFS_IOC_GETBMAPA ? BMV_IF_ATTRFORK : 0);
-        if (ioflags & IO_INVIS)
+        if (ioflags & XFS_IO_INVIS)
                bmx.bmv_iflags |= BMV_IF_NO_DMAPI_READ;
        error = xfs_getbmap(ip, &bmx, xfs_getbmap_format,
                            (struct getbmap *)arg+1);
        if (error)
-                return -error;
+                return error;
        /* copy back header - only size of getbmap */
        if (copy_to_user(arg, &bmx, sizeof(struct getbmap)))
-                return -XFS_ERROR(EFAULT);
+                return -EFAULT;
        return 0;
 }
@@ -1399,7 +1396,7 @@ xfs_getbmapx_format(void **ap, struct getbmapx *bmv, int *full)
        struct getbmapx __user  *base = *ap;
        if (copy_to_user(base, bmv, sizeof(struct getbmapx)))
-                return XFS_ERROR(EFAULT);
+                return -EFAULT;
        *ap += sizeof(struct getbmapx);
        return 0;
@@ -1414,22 +1411,22 @@ xfs_ioc_getbmapx(
        int                     error;
        if (copy_from_user(&bmx, arg, sizeof(bmx)))
-                return -XFS_ERROR(EFAULT);
+                return -EFAULT;
        if (bmx.bmv_count < 2)
-                return -XFS_ERROR(EINVAL);
+                return -EINVAL;
        if (bmx.bmv_iflags & (~BMV_IF_VALID))
-                return -XFS_ERROR(EINVAL);
+                return -EINVAL;
        error = xfs_getbmap(ip, &bmx, xfs_getbmapx_format,
                            (struct getbmapx *)arg+1);
        if (error)
-                return -error;
+                return error;
        /* copy back header */
        if (copy_to_user(arg, &bmx, sizeof(struct getbmapx)))
-                return -XFS_ERROR(EFAULT);
+                return -EFAULT;
        return 0;
 }
@@ -1445,33 +1442,33 @@ xfs_ioc_swapext(
        /* Pull information for the target fd */
        f = fdget((int)sxp->sx_fdtarget);
        if (!f.file) {
-                error = XFS_ERROR(EINVAL);
+                error = -EINVAL;
                goto out;
        }
        if (!(f.file->f_mode & FMODE_WRITE) ||
            !(f.file->f_mode & FMODE_READ) ||
            (f.file->f_flags & O_APPEND)) {
-                error = XFS_ERROR(EBADF);
+                error = -EBADF;
                goto out_put_file;
        }
        tmp = fdget((int)sxp->sx_fdtmp);
        if (!tmp.file) {
-                error = XFS_ERROR(EINVAL);
+                error = -EINVAL;
                goto out_put_file;
        }
        if (!(tmp.file->f_mode & FMODE_WRITE) ||
            !(tmp.file->f_mode & FMODE_READ) ||
            (tmp.file->f_flags & O_APPEND)) {
-                error = XFS_ERROR(EBADF);
+                error = -EBADF;
                goto out_put_tmp_file;
        }
        if (IS_SWAPFILE(file_inode(f.file)) ||
            IS_SWAPFILE(file_inode(tmp.file))) {
-                error = XFS_ERROR(EINVAL);
+                error = -EINVAL;
                goto out_put_tmp_file;
        }
@@ -1479,17 +1476,17 @@ xfs_ioc_swapext(
        tip = XFS_I(file_inode(tmp.file));
        if (ip->i_mount != tip->i_mount) {
-                error = XFS_ERROR(EINVAL);
+                error = -EINVAL;
                goto out_put_tmp_file;
        }
        if (ip->i_ino == tip->i_ino) {
-                error = XFS_ERROR(EINVAL);
+                error = -EINVAL;
                goto out_put_tmp_file;
        }
        if (XFS_FORCED_SHUTDOWN(ip->i_mount)) {
-                error = XFS_ERROR(EIO);
+                error = -EIO;
                goto out_put_tmp_file;
        }
@@ -1523,7 +1520,7 @@ xfs_file_ioctl(
        int                     error;
        if (filp->f_mode & FMODE_NOCMTIME)
-                ioflags |= IO_INVIS;
+                ioflags |= XFS_IO_INVIS;
        trace_xfs_file_ioctl(ip);
@@ -1542,7 +1539,7 @@ xfs_file_ioctl(
                xfs_flock64_t           bf;
                if (copy_from_user(&bf, arg, sizeof(bf)))
-                        return -XFS_ERROR(EFAULT);
+                        return -EFAULT;
                return xfs_ioc_space(ip, inode, filp, ioflags, cmd, &bf);
        }
        case XFS_IOC_DIOINFO: {
@@ -1555,7 +1552,7 @@ xfs_file_ioctl(
                da.d_maxiosz = INT_MAX & ~(da.d_miniosz - 1);
                if (copy_to_user(arg, &da, sizeof(da)))
-                        return -XFS_ERROR(EFAULT);
+                        return -EFAULT;
                return 0;
        }
@@ -1588,7 +1585,7 @@ xfs_file_ioctl(
                struct fsdmidata        dmi;
                if (copy_from_user(&dmi, arg, sizeof(dmi)))
-                        return -XFS_ERROR(EFAULT);
+                        return -EFAULT;
                error = mnt_want_write_file(filp);
                if (error)
@@ -1597,7 +1594,7 @@ xfs_file_ioctl(
                error = xfs_set_dmattrs(ip, dmi.fsd_dmevmask,
                                dmi.fsd_dmstate);
                mnt_drop_write_file(filp);
-                return -error;
+                return error;
        }
        case XFS_IOC_GETBMAP:
@@ -1613,14 +1610,14 @@ xfs_file_ioctl(
                xfs_fsop_handlereq_t    hreq;
                if (copy_from_user(&hreq, arg, sizeof(hreq)))
-                        return -XFS_ERROR(EFAULT);
+                        return -EFAULT;
                return xfs_find_handle(cmd, &hreq);
        }
        case XFS_IOC_OPEN_BY_HANDLE: {
                xfs_fsop_handlereq_t    hreq;
                if (copy_from_user(&hreq, arg, sizeof(xfs_fsop_handlereq_t)))
-                        return -XFS_ERROR(EFAULT);
+                        return -EFAULT;
                return xfs_open_by_handle(filp, &hreq);
        }
        case XFS_IOC_FSSETDM_BY_HANDLE:
@@ -1630,7 +1627,7 @@ xfs_file_ioctl(
                xfs_fsop_handlereq_t    hreq;
                if (copy_from_user(&hreq, arg, sizeof(xfs_fsop_handlereq_t)))
-                        return -XFS_ERROR(EFAULT);
+                        return -EFAULT;
                return xfs_readlink_by_handle(filp, &hreq);
        }
        case XFS_IOC_ATTRLIST_BY_HANDLE:
@@ -1643,13 +1640,13 @@ xfs_file_ioctl(
                struct xfs_swapext      sxp;
                if (copy_from_user(&sxp, arg, sizeof(xfs_swapext_t)))
-                        return -XFS_ERROR(EFAULT);
+                        return -EFAULT;
                error = mnt_want_write_file(filp);
                if (error)
                        return error;
                error = xfs_ioc_swapext(&sxp);
                mnt_drop_write_file(filp);
-                return -error;
+                return error;
        }
        case XFS_IOC_FSCOUNTS: {
@@ -1657,10 +1654,10 @@ xfs_file_ioctl(
                error = xfs_fs_counts(mp, &out);
                if (error)
-                        return -error;
+                        return error;
                if (copy_to_user(arg, &out, sizeof(out)))
-                        return -XFS_ERROR(EFAULT);
+                        return -EFAULT;
                return 0;
        }
@@ -1672,10 +1669,10 @@ xfs_file_ioctl(
                        return -EPERM;
                if (mp->m_flags & XFS_MOUNT_RDONLY)
-                        return -XFS_ERROR(EROFS);
+                        return -EROFS;
                if (copy_from_user(&inout, arg, sizeof(inout)))
-                        return -XFS_ERROR(EFAULT);
+                        return -EFAULT;
                error = mnt_want_write_file(filp);
                if (error)
@@ -1686,10 +1683,10 @@ xfs_file_ioctl(
                error = xfs_reserve_blocks(mp, &in, &inout);
                mnt_drop_write_file(filp);
                if (error)
-                        return -error;
+                        return error;
                if (copy_to_user(arg, &inout, sizeof(inout)))
-                        return -XFS_ERROR(EFAULT);
+                        return -EFAULT;
                return 0;
        }
@@ -1701,10 +1698,10 @@ xfs_file_ioctl(
                error = xfs_reserve_blocks(mp, NULL, &out);
                if (error)
-                        return -error;
+                        return error;
                if (copy_to_user(arg, &out, sizeof(out)))
-                        return -XFS_ERROR(EFAULT);
+                        return -EFAULT;
                return 0;
        }
@@ -1713,42 +1710,42 @@ xfs_file_ioctl(
                xfs_growfs_data_t in;
                if (copy_from_user(&in, arg, sizeof(in)))
-                        return -XFS_ERROR(EFAULT);
+                        return -EFAULT;
                error = mnt_want_write_file(filp);
                if (error)
                        return error;
                error = xfs_growfs_data(mp, &in);
                mnt_drop_write_file(filp);
-                return -error;
+                return error;
        }
        case XFS_IOC_FSGROWFSLOG: {
                xfs_growfs_log_t in;
                if (copy_from_user(&in, arg, sizeof(in)))
-                        return -XFS_ERROR(EFAULT);
+                        return -EFAULT;
                error = mnt_want_write_file(filp);
                if (error)
                        return error;
                error = xfs_growfs_log(mp, &in);
                mnt_drop_write_file(filp);
-                return -error;
+                return error;
        }
        case XFS_IOC_FSGROWFSRT: {
                xfs_growfs_rt_t in;
                if (copy_from_user(&in, arg, sizeof(in)))
-                        return -XFS_ERROR(EFAULT);
+                        return -EFAULT;
                error = mnt_want_write_file(filp);
                if (error)
                        return error;
                error = xfs_growfs_rt(mp, &in);
                mnt_drop_write_file(filp);
-                return -error;
+                return error;
        }
        case XFS_IOC_GOINGDOWN: {
@@ -1758,10 +1755,9 @@ xfs_file_ioctl(
                        return -EPERM;
                if (get_user(in, (__uint32_t __user *)arg))
-                        return -XFS_ERROR(EFAULT);
+                        return -EFAULT;
-                error = xfs_fs_goingdown(mp, in);
+                return xfs_fs_goingdown(mp, in);
-                return -error;
        }
        case XFS_IOC_ERROR_INJECTION: {
@@ -1771,18 +1767,16 @@ xfs_file_ioctl(
                        return -EPERM;
                if (copy_from_user(&in, arg, sizeof(in)))
-                        return -XFS_ERROR(EFAULT);
+                        return -EFAULT;
-                error = xfs_errortag_add(in.errtag, mp);
+                return xfs_errortag_add(in.errtag, mp);
-                return -error;
        }
        case XFS_IOC_ERROR_CLEARALL:
                if (!capable(CAP_SYS_ADMIN))
                        return -EPERM;
-                error = xfs_errortag_clearall(mp, 1);
+                return xfs_errortag_clearall(mp, 1);
-                return -error;
        case XFS_IOC_FREE_EOFBLOCKS: {
                struct xfs_fs_eofblocks eofb;
@@ -1792,16 +1786,16 @@ xfs_file_ioctl(
                        return -EPERM;
                if (mp->m_flags & XFS_MOUNT_RDONLY)
-                        return -XFS_ERROR(EROFS);
+                        return -EROFS;
                if (copy_from_user(&eofb, arg, sizeof(eofb)))
-                        return -XFS_ERROR(EFAULT);
+                        return -EFAULT;
                error = xfs_fs_eofblocks_from_user(&eofb, &keofb);
                if (error)
-                        return -error;
+                        return error;
-                return -xfs_icache_free_eofblocks(mp, &keofb);
+                return xfs_icache_free_eofblocks(mp, &keofb);
        }
        default:
diff --git a/fs/xfs/xfs_ioctl32.c b/fs/xfs/xfs_ioctl32.c
index 944d5baa710a..a554646ff141 100644
--- a/fs/xfs/xfs_ioctl32.c
+++ b/fs/xfs/xfs_ioctl32.c
@@ -28,7 +28,6 @@
 #include "xfs_sb.h"
 #include "xfs_ag.h"
 #include "xfs_mount.h"
-#include "xfs_vnode.h"
 #include "xfs_inode.h"
 #include "xfs_itable.h"
 #include "xfs_error.h"
@@ -56,7 +55,7 @@ xfs_compat_flock64_copyin(
            get_user(bf->l_sysid,       &arg32->l_sysid) ||
            get_user(bf->l_pid,         &arg32->l_pid) ||
            copy_from_user(bf->l_pad,   &arg32->l_pad,  4*sizeof(u32)))
-                return -XFS_ERROR(EFAULT);
+                return -EFAULT;
        return 0;
 }
@@ -70,10 +69,10 @@ xfs_compat_ioc_fsgeometry_v1(
        error = xfs_fs_geometry(mp, &fsgeo, 3);
        if (error)
-                return -error;
+                return error;
        /* The 32-bit variant simply has some padding at the end */
        if (copy_to_user(arg32, &fsgeo, sizeof(struct compat_xfs_fsop_geom_v1)))
-                return -XFS_ERROR(EFAULT);
+                return -EFAULT;
        return 0;
 }
@@ -84,7 +83,7 @@ xfs_compat_growfs_data_copyin(
 {
        if (get_user(in->newblocks, &arg32->newblocks) ||
            get_user(in->imaxpct,   &arg32->imaxpct))
-                return -XFS_ERROR(EFAULT);
+                return -EFAULT;
        return 0;
 }
@@ -95,14 +94,14 @@ xfs_compat_growfs_rt_copyin(
 {
        if (get_user(in->newblocks, &arg32->newblocks) ||
            get_user(in->extsize,   &arg32->extsize))
-                return -XFS_ERROR(EFAULT);
+                return -EFAULT;
        return 0;
 }
 STATIC int
 xfs_inumbers_fmt_compat(
        void                    __user *ubuffer,
-        const xfs_inogrp_t      *buffer,
+        const struct xfs_inogrp *buffer,
        long                    count,
        long                    *written)
 {
@@ -113,7 +112,7 @@ xfs_inumbers_fmt_compat(
                if (put_user(buffer[i].xi_startino,   &p32[i].xi_startino) ||
                    put_user(buffer[i].xi_alloccount, &p32[i].xi_alloccount) ||
                    put_user(buffer[i].xi_allocmask,  &p32[i].xi_allocmask))
-                        return -XFS_ERROR(EFAULT);
+                        return -EFAULT;
        }
        *written = count * sizeof(*p32);
        return 0;
@@ -132,7 +131,7 @@ xfs_ioctl32_bstime_copyin(
        if (get_user(sec32,             &bstime32->tv_sec)      ||
            get_user(bstime->tv_nsec,   &bstime32->tv_nsec))
-                return -XFS_ERROR(EFAULT);
+                return -EFAULT;
        bstime->tv_sec = sec32;
        return 0;
 }
@@ -164,7 +163,7 @@ xfs_ioctl32_bstat_copyin(
            get_user(bstat->bs_dmevmask, &bstat32->bs_dmevmask) ||
            get_user(bstat->bs_dmstate, &bstat32->bs_dmstate)   ||
            get_user(bstat->bs_aextents, &bstat32->bs_aextents))
-                return -XFS_ERROR(EFAULT);
+                return -EFAULT;
        return 0;
 }
@@ -180,7 +179,7 @@ xfs_bstime_store_compat(
        sec32 = p->tv_sec;
        if (put_user(sec32, &p32->tv_sec) ||
            put_user(p->tv_nsec, &p32->tv_nsec))
-                return -XFS_ERROR(EFAULT);
+                return -EFAULT;
        return 0;
 }
@@ -195,7 +194,7 @@ xfs_bulkstat_one_fmt_compat(
        compat_xfs_bstat_t      __user *p32 = ubuffer;
        if (ubsize < sizeof(*p32))
-                return XFS_ERROR(ENOMEM);
+                return -ENOMEM;
        if (put_user(buffer->bs_ino,      &p32->bs_ino)         ||
            put_user(buffer->bs_mode,     &p32->bs_mode)        ||
@@ -218,7 +217,7 @@ xfs_bulkstat_one_fmt_compat(
            put_user(buffer->bs_dmevmask, &p32->bs_dmevmask)    ||
            put_user(buffer->bs_dmstate,  &p32->bs_dmstate)     ||
            put_user(buffer->bs_aextents, &p32->bs_aextents))
-                return XFS_ERROR(EFAULT);
+                return -EFAULT;
        if (ubused)
                *ubused = sizeof(*p32);
        return 0;
@@ -256,30 +255,30 @@ xfs_compat_ioc_bulkstat(
        /* should be called again (unused here, but used in dmapi) */
        if (!capable(CAP_SYS_ADMIN))
-                return -XFS_ERROR(EPERM);
+                return -EPERM;
        if (XFS_FORCED_SHUTDOWN(mp))
-                return -XFS_ERROR(EIO);
+                return -EIO;
        if (get_user(addr, &p32->lastip))
-                return -XFS_ERROR(EFAULT);
+                return -EFAULT;
        bulkreq.lastip = compat_ptr(addr);
        if (get_user(bulkreq.icount, &p32->icount) ||
            get_user(addr, &p32->ubuffer))
-                return -XFS_ERROR(EFAULT);
+                return -EFAULT;
        bulkreq.ubuffer = compat_ptr(addr);
        if (get_user(addr, &p32->ocount))
-                return -XFS_ERROR(EFAULT);
+                return -EFAULT;
        bulkreq.ocount = compat_ptr(addr);
        if (copy_from_user(&inlast, bulkreq.lastip, sizeof(__s64)))
-                return -XFS_ERROR(EFAULT);
+                return -EFAULT;
        if ((count = bulkreq.icount) <= 0)
-                return -XFS_ERROR(EINVAL);
+                return -EINVAL;
        if (bulkreq.ubuffer == NULL)
-                return -XFS_ERROR(EINVAL);
+                return -EINVAL;
        if (cmd == XFS_IOC_FSINUMBERS_32) {
                error = xfs_inumbers(mp, &inlast, &count,
@@ -294,17 +293,17 @@ xfs_compat_ioc_bulkstat(
                        xfs_bulkstat_one_compat, sizeof(compat_xfs_bstat_t),
                        bulkreq.ubuffer, &done);
        } else
-                error = XFS_ERROR(EINVAL);
+                error = -EINVAL;
        if (error)
-                return -error;
+                return error;
        if (bulkreq.ocount != NULL) {
                if (copy_to_user(bulkreq.lastip, &inlast,
                                                sizeof(xfs_ino_t)))
-                        return -XFS_ERROR(EFAULT);
+                        return -EFAULT;
                if (copy_to_user(bulkreq.ocount, &count, sizeof(count)))
-                        return -XFS_ERROR(EFAULT);
+                        return -EFAULT;
        }
        return 0;
@@ -318,7 +317,7 @@ xfs_compat_handlereq_copyin(
        compat_xfs_fsop_handlereq_t     hreq32;
        if (copy_from_user(&hreq32, arg32, sizeof(compat_xfs_fsop_handlereq_t)))
-                return -XFS_ERROR(EFAULT);
+                return -EFAULT;
        hreq->fd = hreq32.fd;
        hreq->path = compat_ptr(hreq32.path);
@@ -352,19 +351,19 @@ xfs_compat_attrlist_by_handle(
        char                    *kbuf;
        if (!capable(CAP_SYS_ADMIN))
-                return -XFS_ERROR(EPERM);
+                return -EPERM;
        if (copy_from_user(&al_hreq, arg,
                           sizeof(compat_xfs_fsop_attrlist_handlereq_t)))
-                return -XFS_ERROR(EFAULT);
+                return -EFAULT;
        if (al_hreq.buflen < sizeof(struct attrlist) ||
            al_hreq.buflen > XATTR_LIST_MAX)
-                return -XFS_ERROR(EINVAL);
+                return -EINVAL;
        /*
         * Reject flags, only allow namespaces.
         */
        if (al_hreq.flags & ~(ATTR_ROOT | ATTR_SECURE))
-                return -XFS_ERROR(EINVAL);
+                return -EINVAL;
        dentry = xfs_compat_handlereq_to_dentry(parfilp, &al_hreq.hreq);
        if (IS_ERR(dentry))
@@ -376,7 +375,7 @@ xfs_compat_attrlist_by_handle(
                goto out_dput;
        cursor = (attrlist_cursor_kern_t *)&al_hreq.pos;
-        error = -xfs_attr_list(XFS_I(dentry->d_inode), kbuf, al_hreq.buflen,
+        error = xfs_attr_list(XFS_I(dentry->d_inode), kbuf, al_hreq.buflen,
                                        al_hreq.flags, cursor);
        if (error)
                goto out_kfree;
@@ -404,10 +403,10 @@ xfs_compat_attrmulti_by_handle(
        unsigned char                           *attr_name;
        if (!capable(CAP_SYS_ADMIN))
-                return -XFS_ERROR(EPERM);
+                return -EPERM;
        if (copy_from_user(&am_hreq, arg,
                           sizeof(compat_xfs_fsop_attrmulti_handlereq_t)))
-                return -XFS_ERROR(EFAULT);
+                return -EFAULT;
        /* overflow check */
        if (am_hreq.opcount >= INT_MAX / sizeof(compat_xfs_attr_multiop_t))
@@ -417,7 +416,7 @@ xfs_compat_attrmulti_by_handle(
        if (IS_ERR(dentry))
                return PTR_ERR(dentry);
-        error = E2BIG;
+        error = -E2BIG;
        size = am_hreq.opcount * sizeof(compat_xfs_attr_multiop_t);
        if (!size || size > 16 * PAGE_SIZE)
                goto out_dput;
@@ -428,7 +427,7 @@ xfs_compat_attrmulti_by_handle(
                goto out_dput;
        }
-        error = ENOMEM;
+        error = -ENOMEM;
        attr_name = kmalloc(MAXNAMELEN, GFP_KERNEL);
        if (!attr_name)
                goto out_kfree_ops;
@@ -439,7 +438,7 @@ xfs_compat_attrmulti_by_handle(
                                compat_ptr(ops[i].am_attrname),
                                MAXNAMELEN);
                if (ops[i].am_error == 0 || ops[i].am_error == MAXNAMELEN)
-                        error = ERANGE;
+                        error = -ERANGE;
                if (ops[i].am_error < 0)
                        break;
@@ -470,19 +469,19 @@ xfs_compat_attrmulti_by_handle(
                        mnt_drop_write_file(parfilp);
                        break;
                default:
-                        ops[i].am_error = EINVAL;
+                        ops[i].am_error = -EINVAL;
                }
        }
        if (copy_to_user(compat_ptr(am_hreq.ops), ops, size))
-                error = XFS_ERROR(EFAULT);
+                error = -EFAULT;
        kfree(attr_name);
 out_kfree_ops:
        kfree(ops);
 out_dput:
        dput(dentry);
-        return -error;
+        return error;
 }
 STATIC int
@@ -496,26 +495,26 @@ xfs_compat_fssetdm_by_handle(
        struct dentry           *dentry;
        if (!capable(CAP_MKNOD))
-                return -XFS_ERROR(EPERM);
+                return -EPERM;
        if (copy_from_user(&dmhreq, arg,
                           sizeof(compat_xfs_fsop_setdm_handlereq_t)))
-                return -XFS_ERROR(EFAULT);
+                return -EFAULT;
        dentry = xfs_compat_handlereq_to_dentry(parfilp, &dmhreq.hreq);
        if (IS_ERR(dentry))
                return PTR_ERR(dentry);
        if (IS_IMMUTABLE(dentry->d_inode) || IS_APPEND(dentry->d_inode)) {
-                error = -XFS_ERROR(EPERM);
+                error = -EPERM;
                goto out;
        }
        if (copy_from_user(&fsd, compat_ptr(dmhreq.data), sizeof(fsd))) {
-                error = -XFS_ERROR(EFAULT);
+                error = -EFAULT;
                goto out;
        }
-        error = -xfs_set_dmattrs(XFS_I(dentry->d_inode), fsd.fsd_dmevmask,
+        error = xfs_set_dmattrs(XFS_I(dentry->d_inode), fsd.fsd_dmevmask,
                                 fsd.fsd_dmstate);
 out:
@@ -537,7 +536,7 @@ xfs_file_compat_ioctl(
        int                     error;
        if (filp->f_mode & FMODE_NOCMTIME)
-                ioflags |= IO_INVIS;
+                ioflags |= XFS_IO_INVIS;
        trace_xfs_file_compat_ioctl(ip);
@@ -588,7 +587,7 @@ xfs_file_compat_ioctl(
                struct xfs_flock64      bf;
                if (xfs_compat_flock64_copyin(&bf, arg))
-                        return -XFS_ERROR(EFAULT);
+                        return -EFAULT;
                cmd = _NATIVE_IOC(cmd, struct xfs_flock64);
                return xfs_ioc_space(ip, inode, filp, ioflags, cmd, &bf);
        }
@@ -598,25 +597,25 @@ xfs_file_compat_ioctl(
                struct xfs_growfs_data  in;
                if (xfs_compat_growfs_data_copyin(&in, arg))
-                        return -XFS_ERROR(EFAULT);
+                        return -EFAULT;
                error = mnt_want_write_file(filp);
                if (error)
                        return error;
                error = xfs_growfs_data(mp, &in);
                mnt_drop_write_file(filp);
-                return -error;
+                return error;
        }
        case XFS_IOC_FSGROWFSRT_32: {
                struct xfs_growfs_rt    in;
                if (xfs_compat_growfs_rt_copyin(&in, arg))
-                        return -XFS_ERROR(EFAULT);
+                        return -EFAULT;
                error = mnt_want_write_file(filp);
                if (error)
                        return error;
                error = xfs_growfs_rt(mp, &in);
                mnt_drop_write_file(filp);
-                return -error;
+                return error;
        }
 #endif
        /* long changes size, but xfs only copiese out 32 bits */
@@ -633,13 +632,13 @@ xfs_file_compat_ioctl(
                if (copy_from_user(&sxp, sxu,
                                   offsetof(struct xfs_swapext, sx_stat)) ||
                    xfs_ioctl32_bstat_copyin(&sxp.sx_stat, &sxu->sx_stat))
-                        return -XFS_ERROR(EFAULT);
+                        return -EFAULT;
                error = mnt_want_write_file(filp);
                if (error)
                        return error;
                error = xfs_ioc_swapext(&sxp);
                mnt_drop_write_file(filp);
-                return -error;
+                return error;
        }
        case XFS_IOC_FSBULKSTAT_32:
        case XFS_IOC_FSBULKSTAT_SINGLE_32:
@@ -651,7 +650,7 @@ xfs_file_compat_ioctl(
                struct xfs_fsop_handlereq       hreq;
                if (xfs_compat_handlereq_copyin(&hreq, arg))
-                        return -XFS_ERROR(EFAULT);
+                        return -EFAULT;
                cmd = _NATIVE_IOC(cmd, struct xfs_fsop_handlereq);
                return xfs_find_handle(cmd, &hreq);
        }
@@ -659,14 +658,14 @@ xfs_file_compat_ioctl(
                struct xfs_fsop_handlereq       hreq;
                if (xfs_compat_handlereq_copyin(&hreq, arg))
-                        return -XFS_ERROR(EFAULT);
+                        return -EFAULT;
                return xfs_open_by_handle(filp, &hreq);
        }
        case XFS_IOC_READLINK_BY_HANDLE_32: {
                struct xfs_fsop_handlereq       hreq;
                if (xfs_compat_handlereq_copyin(&hreq, arg))
-                        return -XFS_ERROR(EFAULT);
+                        return -EFAULT;
                return xfs_readlink_by_handle(filp, &hreq);
        }
        case XFS_IOC_ATTRLIST_BY_HANDLE_32:
@@ -676,6 +675,6 @@ xfs_file_compat_ioctl(
        case XFS_IOC_FSSETDM_BY_HANDLE_32:
                return xfs_compat_fssetdm_by_handle(filp, arg);
        default:
-                return -XFS_ERROR(ENOIOCTLCMD);
+                return -ENOIOCTLCMD;
        }
 }
diff --git a/fs/xfs/xfs_iomap.c b/fs/xfs/xfs_iomap.c
index 6d3ec2b6ee29..e9c47b6f5e5a 100644
--- a/fs/xfs/xfs_iomap.c
+++ b/fs/xfs/xfs_iomap.c
@@ -110,7 +110,7 @@ xfs_alert_fsblock_zero(
                (unsigned long long)imap->br_startoff,
                (unsigned long long)imap->br_blockcount,
                imap->br_state);
-        return EFSCORRUPTED;
+        return -EFSCORRUPTED;
 }
 int
@@ -138,7 +138,7 @@ xfs_iomap_write_direct(
        error = xfs_qm_dqattach(ip, 0);
        if (error)
-                return XFS_ERROR(error);
+                return error;
        rt = XFS_IS_REALTIME_INODE(ip);
        extsz = xfs_get_extsz_hint(ip);
@@ -148,7 +148,7 @@ xfs_iomap_write_direct(
        if ((offset + count) > XFS_ISIZE(ip)) {
                error = xfs_iomap_eof_align_last_fsb(mp, ip, extsz, &last_fsb);
                if (error)
-                        return XFS_ERROR(error);
+                        return error;
        } else {
                if (nmaps && (imap->br_startblock == HOLESTARTBLOCK))
                        last_fsb = MIN(last_fsb, (xfs_fileoff_t)
@@ -188,7 +188,7 @@ xfs_iomap_write_direct(
         */
        if (error) {
                xfs_trans_cancel(tp, 0);
-                return XFS_ERROR(error);
+                return error;
        }
        xfs_ilock(ip, XFS_ILOCK_EXCL);
@@ -225,7 +225,7 @@ xfs_iomap_write_direct(
         * Copy any maps to caller's array and return any error.
         */
        if (nimaps == 0) {
-                error = XFS_ERROR(ENOSPC);
+                error = -ENOSPC;
                goto out_unlock;
        }
@@ -397,7 +397,8 @@ xfs_quota_calc_throttle(
        struct xfs_inode *ip,
        int type,
        xfs_fsblock_t *qblocks,
-        int *qshift)
+        int *qshift,
+        int64_t *qfreesp)
 {
        int64_t freesp;
        int shift = 0;
@@ -406,6 +407,7 @@ xfs_quota_calc_throttle(
        /* over hi wmark, squash the prealloc completely */
        if (dq->q_res_bcount >= dq->q_prealloc_hi_wmark) {
                *qblocks = 0;
+                *qfreesp = 0;
                return;
        }
@@ -418,6 +420,9 @@ xfs_quota_calc_throttle(
                        shift += 2;
        }
+        if (freesp < *qfreesp)
+                *qfreesp = freesp;
        /* only overwrite the throttle values if we are more aggressive */
        if ((freesp >> shift) < (*qblocks >> *qshift)) {
                *qblocks = freesp;
@@ -476,15 +481,18 @@ xfs_iomap_prealloc_size(
        }
        /*
-         * Check each quota to cap the prealloc size and provide a shift
+         * Check each quota to cap the prealloc size, provide a shift value to
-         * value to throttle with.
+         * throttle with and adjust amount of available space.
         */
        if (xfs_quota_need_throttle(ip, XFS_DQ_USER, alloc_blocks))
-                xfs_quota_calc_throttle(ip, XFS_DQ_USER, &qblocks, &qshift);
+                xfs_quota_calc_throttle(ip, XFS_DQ_USER, &qblocks, &qshift,
+                                        &freesp);
        if (xfs_quota_need_throttle(ip, XFS_DQ_GROUP, alloc_blocks))
-                xfs_quota_calc_throttle(ip, XFS_DQ_GROUP, &qblocks, &qshift);
+                xfs_quota_calc_throttle(ip, XFS_DQ_GROUP, &qblocks, &qshift,
+                                        &freesp);
        if (xfs_quota_need_throttle(ip, XFS_DQ_PROJ, alloc_blocks))
-                xfs_quota_calc_throttle(ip, XFS_DQ_PROJ, &qblocks, &qshift);
+                xfs_quota_calc_throttle(ip, XFS_DQ_PROJ, &qblocks, &qshift,
+                                        &freesp);
        /*
         * The final prealloc size is set to the minimum of free space available
@@ -552,7 +560,7 @@ xfs_iomap_write_delay(
         */
        error = xfs_qm_dqattach_locked(ip, 0);
        if (error)
-                return XFS_ERROR(error);
+                return error;
        extsz = xfs_get_extsz_hint(ip);
        offset_fsb = XFS_B_TO_FSBT(mp, offset);
@@ -596,11 +604,11 @@ retry:
                                imap, &nimaps, XFS_BMAPI_ENTIRE);
        switch (error) {
        case 0:
-        case ENOSPC:
+        case -ENOSPC:
-        case EDQUOT:
+        case -EDQUOT:
                break;
        default:
-                return XFS_ERROR(error);
+                return error;
        }
        /*
@@ -614,7 +622,7 @@ retry:
                        error = 0;
                        goto retry;
                }
-                return XFS_ERROR(error ? error : ENOSPC);
+                return error ? error : -ENOSPC;
        }
        if (!(imap[0].br_startblock || XFS_IS_REALTIME_INODE(ip)))
@@ -663,7 +671,7 @@ xfs_iomap_write_allocate(
         */
        error = xfs_qm_dqattach(ip, 0);
        if (error)
-                return XFS_ERROR(error);
+                return error;
        offset_fsb = XFS_B_TO_FSBT(mp, offset);
        count_fsb = imap->br_blockcount;
@@ -690,7 +698,7 @@ xfs_iomap_write_allocate(
                                                  nres, 0);
                        if (error) {
                                xfs_trans_cancel(tp, 0);
-                                return XFS_ERROR(error);
+                                return error;
                        }
                        xfs_ilock(ip, XFS_ILOCK_EXCL);
                        xfs_trans_ijoin(tp, ip, 0);
@@ -739,7 +747,7 @@ xfs_iomap_write_allocate(
                        if ((map_start_fsb + count_fsb) > last_block) {
                                count_fsb = last_block - map_start_fsb;
                                if (count_fsb == 0) {
-                                        error = EAGAIN;
+                                        error = -EAGAIN;
                                        goto trans_cancel;
                                }
                        }
@@ -793,7 +801,7 @@ trans_cancel:
        xfs_trans_cancel(tp, XFS_TRANS_RELEASE_LOG_RES | XFS_TRANS_ABORT);
 error0:
        xfs_iunlock(ip, XFS_ILOCK_EXCL);
-        return XFS_ERROR(error);
+        return error;
 }
 int
@@ -853,7 +861,7 @@ xfs_iomap_write_unwritten(
                                          resblks, 0);
                if (error) {
                        xfs_trans_cancel(tp, 0);
-                        return XFS_ERROR(error);
+                        return error;
                }
                xfs_ilock(ip, XFS_ILOCK_EXCL);
@@ -892,7 +900,7 @@ xfs_iomap_write_unwritten(
                error = xfs_trans_commit(tp, XFS_TRANS_RELEASE_LOG_RES);
                xfs_iunlock(ip, XFS_ILOCK_EXCL);
                if (error)
-                        return XFS_ERROR(error);
+                        return error;
                if (!(imap.br_startblock || XFS_IS_REALTIME_INODE(ip)))
                        return xfs_alert_fsblock_zero(ip, &imap);
@@ -915,5 +923,5 @@ error_on_bmapi_transaction:
        xfs_bmap_cancel(&free_list);
        xfs_trans_cancel(tp, (XFS_TRANS_RELEASE_LOG_RES | XFS_TRANS_ABORT));
        xfs_iunlock(ip, XFS_ILOCK_EXCL);
-        return XFS_ERROR(error);
+        return error;
 }
diff --git a/fs/xfs/xfs_iops.c b/fs/xfs/xfs_iops.c
index 205613a06068..72129493e9d3 100644
--- a/fs/xfs/xfs_iops.c
+++ b/fs/xfs/xfs_iops.c
@@ -72,7 +72,7 @@ xfs_initxattrs(
        int                     error = 0;
        for (xattr = xattr_array; xattr->name != NULL; xattr++) {
-                error = -xfs_attr_set(ip, xattr->name, xattr->value,
+                error = xfs_attr_set(ip, xattr->name, xattr->value,
                                      xattr->value_len, ATTR_SECURE);
                if (error < 0)
                        break;
@@ -93,7 +93,7 @@ xfs_init_security(
        struct inode    *dir,
        const struct qstr *qstr)
 {
-        return -security_inode_init_security(inode, dir, qstr,
+        return security_inode_init_security(inode, dir, qstr,
                                             &xfs_initxattrs, NULL);
 }
@@ -173,12 +173,12 @@ xfs_generic_create(
 #ifdef CONFIG_XFS_POSIX_ACL
        if (default_acl) {
-                error = -xfs_set_acl(inode, default_acl, ACL_TYPE_DEFAULT);
+                error = xfs_set_acl(inode, default_acl, ACL_TYPE_DEFAULT);
                if (error)
                        goto out_cleanup_inode;
        }
        if (acl) {
-                error = -xfs_set_acl(inode, acl, ACL_TYPE_ACCESS);
+                error = xfs_set_acl(inode, acl, ACL_TYPE_ACCESS);
                if (error)
                        goto out_cleanup_inode;
        }
@@ -194,7 +194,7 @@ xfs_generic_create(
                posix_acl_release(default_acl);
        if (acl)
                posix_acl_release(acl);
-        return -error;
+        return error;
 out_cleanup_inode:
        if (!tmpfile)
@@ -248,8 +248,8 @@ xfs_vn_lookup(
        xfs_dentry_to_name(&name, dentry, 0);
        error = xfs_lookup(XFS_I(dir), &name, &cip, NULL);
        if (unlikely(error)) {
-                if (unlikely(error != ENOENT))
+                if (unlikely(error != -ENOENT))
-                        return ERR_PTR(-error);
+                        return ERR_PTR(error);
                d_add(dentry, NULL);
                return NULL;
        }
@@ -275,8 +275,8 @@ xfs_vn_ci_lookup(
        xfs_dentry_to_name(&xname, dentry, 0);
        error = xfs_lookup(XFS_I(dir), &xname, &ip, &ci_name);
        if (unlikely(error)) {
-                if (unlikely(error != ENOENT))
+                if (unlikely(error != -ENOENT))
-                        return ERR_PTR(-error);
+                        return ERR_PTR(error);
                /*
                 * call d_add(dentry, NULL) here when d_drop_negative_children
                 * is called in xfs_vn_mknod (ie. allow negative dentries
@@ -311,7 +311,7 @@ xfs_vn_link(
        error = xfs_link(XFS_I(dir), XFS_I(inode), &name);
        if (unlikely(error))
-                return -error;
+                return error;
        ihold(inode);
        d_instantiate(dentry, inode);
@@ -328,7 +328,7 @@ xfs_vn_unlink(
        xfs_dentry_to_name(&name, dentry, 0);
-        error = -xfs_remove(XFS_I(dir), &name, XFS_I(dentry->d_inode));
+        error = xfs_remove(XFS_I(dir), &name, XFS_I(dentry->d_inode));
        if (error)
                return error;
@@ -375,7 +375,7 @@ xfs_vn_symlink(
        xfs_cleanup_inode(dir, inode, dentry);
        iput(inode);
 out:
-        return -error;
+        return error;
 }
 STATIC int
@@ -392,8 +392,8 @@ xfs_vn_rename(
        xfs_dentry_to_name(&oname, odentry, 0);
        xfs_dentry_to_name(&nname, ndentry, odentry->d_inode->i_mode);
-        return -xfs_rename(XFS_I(odir), &oname, XFS_I(odentry->d_inode),
+        return xfs_rename(XFS_I(odir), &oname, XFS_I(odentry->d_inode),
-                           XFS_I(ndir), &nname, new_inode ?
+                          XFS_I(ndir), &nname, new_inode ?
                                                XFS_I(new_inode) : NULL);
 }
@@ -414,7 +414,7 @@ xfs_vn_follow_link(
        if (!link)
                goto out_err;
-        error = -xfs_readlink(XFS_I(dentry->d_inode), link);
+        error = xfs_readlink(XFS_I(dentry->d_inode), link);
        if (unlikely(error))
                goto out_kfree;
@@ -441,7 +441,7 @@ xfs_vn_getattr(
        trace_xfs_getattr(ip);
        if (XFS_FORCED_SHUTDOWN(mp))
-                return -XFS_ERROR(EIO);
+                return -EIO;
        stat->size = XFS_ISIZE(ip);
        stat->dev = inode->i_sb->s_dev;
@@ -546,14 +546,14 @@ xfs_setattr_nonsize(
        /* If acls are being inherited, we already have this checked */
        if (!(flags & XFS_ATTR_NOACL)) {
                if (mp->m_flags & XFS_MOUNT_RDONLY)
-                        return XFS_ERROR(EROFS);
+                        return -EROFS;
                if (XFS_FORCED_SHUTDOWN(mp))
-                        return XFS_ERROR(EIO);
+                        return -EIO;
-                error = -inode_change_ok(inode, iattr);
+                error = inode_change_ok(inode, iattr);
                if (error)
-                        return XFS_ERROR(error);
+                        return error;
        }
        ASSERT((mask & ATTR_SIZE) == 0);
@@ -703,7 +703,7 @@ xfs_setattr_nonsize(
        xfs_qm_dqrele(gdqp);
        if (error)
-                return XFS_ERROR(error);
+                return error;
        /*
         * XXX(hch): Updating the ACL entries is not atomic vs the i_mode
@@ -713,9 +713,9 @@ xfs_setattr_nonsize(
         *           Posix ACL code seems to care about this issue either.
         */
        if ((mask & ATTR_MODE) && !(flags & XFS_ATTR_NOACL)) {
-                error = -posix_acl_chmod(inode, inode->i_mode);
+                error = posix_acl_chmod(inode, inode->i_mode);
                if (error)
-                        return XFS_ERROR(error);
+                        return error;
        }
        return 0;
@@ -748,14 +748,14 @@ xfs_setattr_size(
        trace_xfs_setattr(ip);
        if (mp->m_flags & XFS_MOUNT_RDONLY)
-                return XFS_ERROR(EROFS);
+                return -EROFS;
        if (XFS_FORCED_SHUTDOWN(mp))
-                return XFS_ERROR(EIO);
+                return -EIO;
-        error = -inode_change_ok(inode, iattr);
+        error = inode_change_ok(inode, iattr);
        if (error)
-                return XFS_ERROR(error);
+                return error;
        ASSERT(xfs_isilocked(ip, XFS_IOLOCK_EXCL));
        ASSERT(S_ISREG(ip->i_d.di_mode));
@@ -818,7 +818,7 @@ xfs_setattr_size(
         * care about here.
         */
        if (oldsize != ip->i_d.di_size && newsize > ip->i_d.di_size) {
-                error = -filemap_write_and_wait_range(VFS_I(ip)->i_mapping,
+                error = filemap_write_and_wait_range(VFS_I(ip)->i_mapping,
                                                      ip->i_d.di_size, newsize);
                if (error)
                        return error;
@@ -844,7 +844,7 @@ xfs_setattr_size(
         * much we can do about this, except to hope that the caller sees ENOMEM
         * and retries the truncate operation.
         */
-        error = -block_truncate_page(inode->i_mapping, newsize, xfs_get_blocks);
+        error = block_truncate_page(inode->i_mapping, newsize, xfs_get_blocks);
        if (error)
                return error;
        truncate_setsize(inode, newsize);
@@ -950,7 +950,7 @@ xfs_vn_setattr(
                error = xfs_setattr_nonsize(ip, iattr, 0);
        }
-        return -error;
+        return error;
 }
 STATIC int
@@ -970,7 +970,7 @@ xfs_vn_update_time(
        error = xfs_trans_reserve(tp, &M_RES(mp)->tr_fsyncts, 0, 0);
        if (error) {
                xfs_trans_cancel(tp, 0);
-                return -error;
+                return error;
        }
        xfs_ilock(ip, XFS_ILOCK_EXCL);
@@ -991,7 +991,7 @@ xfs_vn_update_time(
        }
        xfs_trans_ijoin(tp, ip, XFS_ILOCK_EXCL);
        xfs_trans_log_inode(tp, ip, XFS_ILOG_TIMESTAMP);
-        return -xfs_trans_commit(tp, 0);
+        return xfs_trans_commit(tp, 0);
 }
 #define XFS_FIEMAP_FLAGS        (FIEMAP_FLAG_SYNC|FIEMAP_FLAG_XATTR)
@@ -1036,7 +1036,7 @@ xfs_fiemap_format(
                *full = 1;      /* user array now full */
        }
-        return -error;
+        return error;
 }
 STATIC int
@@ -1055,12 +1055,12 @@ xfs_vn_fiemap(
                return error;
        /* Set up bmap header for xfs internal routine */
-        bm.bmv_offset = BTOBB(start);
+        bm.bmv_offset = BTOBBT(start);
        /* Special case for whole file */
        if (length == FIEMAP_MAX_OFFSET)
                bm.bmv_length = -1LL;
        else
-                bm.bmv_length = BTOBB(length);
+                bm.bmv_length = BTOBB(start + length) - bm.bmv_offset;
        /* We add one because in getbmap world count includes the header */
        bm.bmv_count = !fieinfo->fi_extents_max ? MAXEXTNUM :
@@ -1075,7 +1075,7 @@ xfs_vn_fiemap(
        error = xfs_getbmap(ip, &bm, xfs_fiemap_format, fieinfo);
        if (error)
-                return -error;
+                return error;
        return 0;
 }
diff --git a/fs/xfs/xfs_itable.c b/fs/xfs/xfs_itable.c
index cb64f222d607..f71be9c68017 100644
--- a/fs/xfs/xfs_itable.c
+++ b/fs/xfs/xfs_itable.c
@@ -67,19 +67,17 @@ xfs_bulkstat_one_int(
        *stat = BULKSTAT_RV_NOTHING;
        if (!buffer || xfs_internal_inum(mp, ino))
-                return XFS_ERROR(EINVAL);
+                return -EINVAL;
        buf = kmem_alloc(sizeof(*buf), KM_SLEEP | KM_MAYFAIL);
        if (!buf)
-                return XFS_ERROR(ENOMEM);
+                return -ENOMEM;
        error = xfs_iget(mp, NULL, ino,
                         (XFS_IGET_DONTCACHE | XFS_IGET_UNTRUSTED),
                         XFS_ILOCK_SHARED, &ip);
-        if (error) {
+        if (error)
-                *stat = BULKSTAT_RV_NOTHING;
                goto out_free;
-        }
        ASSERT(ip != NULL);
        ASSERT(ip->i_imap.im_blkno != 0);
@@ -136,7 +134,6 @@ xfs_bulkstat_one_int(
        IRELE(ip);
        error = formatter(buffer, ubsize, ubused, buf);
        if (!error)
                *stat = BULKSTAT_RV_DIDONE;
@@ -154,9 +151,9 @@ xfs_bulkstat_one_fmt(
        const xfs_bstat_t       *buffer)
 {
        if (ubsize < sizeof(*buffer))
-                return XFS_ERROR(ENOMEM);
+                return -ENOMEM;
        if (copy_to_user(ubuffer, buffer, sizeof(*buffer)))
-                return XFS_ERROR(EFAULT);
+                return -EFAULT;
        if (ubused)
                *ubused = sizeof(*buffer);
        return 0;
@@ -175,9 +172,170 @@ xfs_bulkstat_one(
                                    xfs_bulkstat_one_fmt, ubused, stat);
 }
+/*
+ * Loop over all clusters in a chunk for a given incore inode allocation btree
+ * record.  Do a readahead if there are any allocated inodes in that cluster.
+ */
+STATIC void
+xfs_bulkstat_ichunk_ra(
+        struct xfs_mount                *mp,
+        xfs_agnumber_t                  agno,
+        struct xfs_inobt_rec_incore     *irec)
+{
+        xfs_agblock_t                   agbno;
+        struct blk_plug                 plug;
+        int                             blks_per_cluster;
+        int                             inodes_per_cluster;
+        int                             i;      /* inode chunk index */
+        agbno = XFS_AGINO_TO_AGBNO(mp, irec->ir_startino);
+        blks_per_cluster = xfs_icluster_size_fsb(mp);
+        inodes_per_cluster = blks_per_cluster << mp->m_sb.sb_inopblog;
+        blk_start_plug(&plug);
+        for (i = 0; i < XFS_INODES_PER_CHUNK;
+             i += inodes_per_cluster, agbno += blks_per_cluster) {
+                if (xfs_inobt_maskn(i, inodes_per_cluster) & ~irec->ir_free) {
+                        xfs_btree_reada_bufs(mp, agno, agbno, blks_per_cluster,
+                                             &xfs_inode_buf_ops);
+                }
+        }
+        blk_finish_plug(&plug);
+}
+/*
+ * Lookup the inode chunk that the given inode lives in and then get the record
+ * if we found the chunk.  If the inode was not the last in the chunk and there
+ * are some left allocated, update the data for the pointed-to record as well as
+ * return the count of grabbed inodes.
+ */
+STATIC int
+xfs_bulkstat_grab_ichunk(
+        struct xfs_btree_cur            *cur,   /* btree cursor */
+        xfs_agino_t                     agino,  /* starting inode of chunk */
+        int                             *icount,/* return # of inodes grabbed */
+        struct xfs_inobt_rec_incore     *irec)  /* btree record */
+{
+        int                             idx;    /* index into inode chunk */
+        int                             stat;
+        int                             error = 0;
+        /* Lookup the inode chunk that this inode lives in */
+        error = xfs_inobt_lookup(cur, agino, XFS_LOOKUP_LE, &stat);
+        if (error)
+                return error;
+        if (!stat) {
+                *icount = 0;
+                return error;
+        }
+        /* Get the record, should always work */
+        error = xfs_inobt_get_rec(cur, irec, &stat);
+        if (error)
+                return error;
+        XFS_WANT_CORRUPTED_RETURN(stat == 1);
+        /* Check if the record contains the inode in request */
+        if (irec->ir_startino + XFS_INODES_PER_CHUNK <= agino)
+                return -EINVAL;
+        idx = agino - irec->ir_startino + 1;
+        if (idx < XFS_INODES_PER_CHUNK &&
+            (xfs_inobt_maskn(idx, XFS_INODES_PER_CHUNK - idx) & ~irec->ir_free)) {
+                int     i;
+                /* We got a right chunk with some left inodes allocated at it.
+                 * Grab the chunk record.  Mark all the uninteresting inodes
+                 * free -- because they're before our start point.
+                 */
+                for (i = 0; i < idx; i++) {
+                        if (XFS_INOBT_MASK(i) & ~irec->ir_free)
+                                irec->ir_freecount++;
+                }
+                irec->ir_free |= xfs_inobt_maskn(0, idx);
+                *icount = XFS_INODES_PER_CHUNK - irec->ir_freecount;
+        }
+        return 0;
+}
 #define XFS_BULKSTAT_UBLEFT(ubleft)     ((ubleft) >= statstruct_size)
 /*
+ * Process inodes in chunk with a pointer to a formatter function
+ * that will iget the inode and fill in the appropriate structure.
+ */
+int
+xfs_bulkstat_ag_ichunk(
+        struct xfs_mount                *mp,
+        xfs_agnumber_t                  agno,
+        struct xfs_inobt_rec_incore     *irbp,
+        bulkstat_one_pf                 formatter,
+        size_t                          statstruct_size,
+        struct xfs_bulkstat_agichunk    *acp)
+{
+        xfs_ino_t                       lastino = acp->ac_lastino;
+        char                            __user **ubufp = acp->ac_ubuffer;
+        int                             ubleft = acp->ac_ubleft;
+        int                             ubelem = acp->ac_ubelem;
+        int                             chunkidx, clustidx;
+        int                             error = 0;
+        xfs_agino_t                     agino;
+        for (agino = irbp->ir_startino, chunkidx = clustidx = 0;
+             XFS_BULKSTAT_UBLEFT(ubleft) &&
+             irbp->ir_freecount < XFS_INODES_PER_CHUNK;
+             chunkidx++, clustidx++, agino++) {
+                int             fmterror;       /* bulkstat formatter result */
+                int             ubused;
+                xfs_ino_t       ino = XFS_AGINO_TO_INO(mp, agno, agino);
+                ASSERT(chunkidx < XFS_INODES_PER_CHUNK);
+                /* Skip if this inode is free */
+                if (XFS_INOBT_MASK(chunkidx) & irbp->ir_free) {
+                        lastino = ino;
+                        continue;
+                }
+                /*
+                 * Count used inodes as free so we can tell when the
+                 * chunk is used up.
+                 */
+                irbp->ir_freecount++;
+                /* Get the inode and fill in a single buffer */
+                ubused = statstruct_size;
+                error = formatter(mp, ino, *ubufp, ubleft, &ubused, &fmterror);
+                if (fmterror == BULKSTAT_RV_NOTHING) {
+                        if (error && error != -ENOENT && error != -EINVAL) {
+                                ubleft = 0;
+                                break;
+                        }
+                        lastino = ino;
+                        continue;
+                }
+                if (fmterror == BULKSTAT_RV_GIVEUP) {
+                        ubleft = 0;
+                        ASSERT(error);
+                        break;
+                }
+                if (*ubufp)
+                        *ubufp += ubused;
+                ubleft -= ubused;
+                ubelem++;
+                lastino = ino;
+        }
+        acp->ac_lastino = lastino;
+        acp->ac_ubleft = ubleft;
+        acp->ac_ubelem = ubelem;
+        return error;
+}
+/*
 * Return stat information in bulk (by-inode) for the filesystem.
 */
 int                                     /* error status */
@@ -190,13 +348,10 @@ xfs_bulkstat(
        char                    __user *ubuffer, /* buffer with inode stats */
        int                     *done)  /* 1 if there are more stats to get */
 {
-        xfs_agblock_t           agbno=0;/* allocation group block number */
        xfs_buf_t               *agbp;  /* agi header buffer */
        xfs_agi_t               *agi;   /* agi header data */
        xfs_agino_t             agino;  /* inode # in allocation group */
        xfs_agnumber_t          agno;   /* allocation group number */
-        int                     chunkidx; /* current index into inode chunk */
-        int                     clustidx; /* current index into inode cluster */
        xfs_btree_cur_t         *cur;   /* btree cursor for ialloc btree */
        int                     end_of_ag; /* set if we've seen the ag end */
        int                     error;  /* error code */
@@ -209,8 +364,6 @@ xfs_bulkstat(
        xfs_inobt_rec_incore_t  *irbuf; /* start of irec buffer */
        xfs_inobt_rec_incore_t  *irbufend; /* end of good irec buffer entries */
        xfs_ino_t               lastino; /* last inode number returned */
-        int                     blks_per_cluster; /* # of blocks per cluster */
-        int                     inodes_per_cluster;/* # of inodes per cluster */
        int                     nirbuf; /* size of irbuf */
        int                     rval;   /* return value error code */
        int                     tmp;    /* result value from btree calls */
@@ -218,7 +371,6 @@ xfs_bulkstat(
        int                     ubleft; /* bytes left in user's buffer */
        char                    __user *ubufp;  /* pointer into user's buffer */
        int                     ubelem; /* spaces used in user's buffer */
-        int                     ubused; /* bytes used by formatter */
        /*
         * Get the last inode value, see if there's nothing to do.
@@ -233,20 +385,16 @@ xfs_bulkstat(
                *ubcountp = 0;
                return 0;
        }
-        if (!ubcountp || *ubcountp <= 0) {
-                return EINVAL;
-        }
        ubcount = *ubcountp; /* statstruct's */
        ubleft = ubcount * statstruct_size; /* bytes */
        *ubcountp = ubelem = 0;
        *done = 0;
        fmterror = 0;
        ubufp = ubuffer;
-        blks_per_cluster = xfs_icluster_size_fsb(mp);
-        inodes_per_cluster = blks_per_cluster << mp->m_sb.sb_inopblog;
        irbuf = kmem_zalloc_greedy(&irbsize, PAGE_SIZE, PAGE_SIZE * 4);
        if (!irbuf)
-                return ENOMEM;
+                return -ENOMEM;
        nirbuf = irbsize / sizeof(*irbuf);
@@ -258,14 +406,8 @@ xfs_bulkstat(
        while (XFS_BULKSTAT_UBLEFT(ubleft) && agno < mp->m_sb.sb_agcount) {
                cond_resched();
                error = xfs_ialloc_read_agi(mp, NULL, agno, &agbp);
-                if (error) {
+                if (error)
-                        /*
+                        break;
-                         * Skip this allocation group and go to the next one.
-                         */
-                        agno++;
-                        agino = 0;
-                        continue;
-                }
                agi = XFS_BUF_TO_AGI(agbp);
                /*
                 * Allocate and initialize a btree cursor for ialloc btree.
@@ -275,96 +417,39 @@ xfs_bulkstat(
                irbp = irbuf;
                irbufend = irbuf + nirbuf;
                end_of_ag = 0;
-                /*
+                icount = 0;
-                 * If we're returning in the middle of an allocation group,
-                 * we need to get the remainder of the chunk we're in.
-                 */
                if (agino > 0) {
-                        xfs_inobt_rec_incore_t r;
                        /*
-                         * Lookup the inode chunk that this inode lives in.
+                         * In the middle of an allocation group, we need to get
+                         * the remainder of the chunk we're in.
                         */
-                        error = xfs_inobt_lookup(cur, agino, XFS_LOOKUP_LE,
+                        struct xfs_inobt_rec_incore     r;
-                                                 &tmp);
-                        if (!error &&   /* no I/O error */
+                        error = xfs_bulkstat_grab_ichunk(cur, agino, &icount, &r);
-                            tmp &&      /* lookup succeeded */
+                        if (error)
-                                        /* got the record, should always work */
+                                break;
-                            !(error = xfs_inobt_get_rec(cur, &r, &i)) &&
+                        if (icount) {
-                            i == 1 &&
-                                        /* this is the right chunk */
-                            agino < r.ir_startino + XFS_INODES_PER_CHUNK &&
-                                        /* lastino was not last in chunk */
-                            (chunkidx = agino - r.ir_startino + 1) <
-                                    XFS_INODES_PER_CHUNK &&
-                                        /* there are some left allocated */
-                            xfs_inobt_maskn(chunkidx,
-                                    XFS_INODES_PER_CHUNK - chunkidx) &
-                                    ~r.ir_free) {
-                                /*
-                                 * Grab the chunk record.  Mark all the
-                                 * uninteresting inodes (because they're
-                                 * before our start point) free.
-                                 */
-                                for (i = 0; i < chunkidx; i++) {
-                                        if (XFS_INOBT_MASK(i) & ~r.ir_free)
-                                                r.ir_freecount++;
-                                }
-                                r.ir_free |= xfs_inobt_maskn(0, chunkidx);
                                irbp->ir_startino = r.ir_startino;
                                irbp->ir_freecount = r.ir_freecount;
                                irbp->ir_free = r.ir_free;
                                irbp++;
                                agino = r.ir_startino + XFS_INODES_PER_CHUNK;
-                                icount = XFS_INODES_PER_CHUNK - r.ir_freecount;
-                        } else {
-                                /*
-                                 * If any of those tests failed, bump the
-                                 * inode number (just in case).
-                                 */
-                                agino++;
-                                icount = 0;
                        }
-                        /*
+                        /* Increment to the next record */
-                         * In any case, increment to the next record.
+                        error = xfs_btree_increment(cur, 0, &tmp);
-                         */
-                        if (!error)
-                                error = xfs_btree_increment(cur, 0, &tmp);
                } else {
-                        /*
+                        /* Start of ag.  Lookup the first inode chunk */
-                         * Start of ag.  Lookup the first inode chunk.
-                         */
                        error = xfs_inobt_lookup(cur, 0, XFS_LOOKUP_GE, &tmp);
-                        icount = 0;
                }
+                if (error)
+                        break;
                /*
                 * Loop through inode btree records in this ag,
                 * until we run out of inodes or space in the buffer.
                 */
                while (irbp < irbufend && icount < ubcount) {
-                        xfs_inobt_rec_incore_t r;
+                        struct xfs_inobt_rec_incore     r;
-                        /*
-                         * Loop as long as we're unable to read the
-                         * inode btree.
-                         */
-                        while (error) {
-                                agino += XFS_INODES_PER_CHUNK;
-                                if (XFS_AGINO_TO_AGBNO(mp, agino) >=
-                                                be32_to_cpu(agi->agi_length))
-                                        break;
-                                error = xfs_inobt_lookup(cur, agino,
-                                                         XFS_LOOKUP_GE, &tmp);
-                                cond_resched();
-                        }
-                        /*
-                         * If ran off the end of the ag either with an error,
-                         * or the normal way, set end and stop collecting.
-                         */
-                        if (error) {
-                                end_of_ag = 1;
-                                break;
-                        }
                        error = xfs_inobt_get_rec(cur, &r, &i);
                        if (error || i == 0) {
@@ -377,25 +462,7 @@ xfs_bulkstat(
                         * Also start read-ahead now for this chunk.
                         */
                        if (r.ir_freecount < XFS_INODES_PER_CHUNK) {
-                                struct blk_plug plug;
+                                xfs_bulkstat_ichunk_ra(mp, agno, &r);
-                                /*
-                                 * Loop over all clusters in the next chunk.
-                                 * Do a readahead if there are any allocated
-                                 * inodes in that cluster.
-                                 */
-                                blk_start_plug(&plug);
-                                agbno = XFS_AGINO_TO_AGBNO(mp, r.ir_startino);
-                                for (chunkidx = 0;
-                                     chunkidx < XFS_INODES_PER_CHUNK;
-                                     chunkidx += inodes_per_cluster,
-                                     agbno += blks_per_cluster) {
-                                        if (xfs_inobt_maskn(chunkidx,
-                                            inodes_per_cluster) & ~r.ir_free)
-                                                xfs_btree_reada_bufs(mp, agno,
-                                                        agbno, blks_per_cluster,
-                                                        &xfs_inode_buf_ops);
-                                }
-                                blk_finish_plug(&plug);
                                irbp->ir_startino = r.ir_startino;
                                irbp->ir_freecount = r.ir_freecount;
                                irbp->ir_free = r.ir_free;
@@ -422,57 +489,20 @@ xfs_bulkstat(
                irbufend = irbp;
                for (irbp = irbuf;
                     irbp < irbufend && XFS_BULKSTAT_UBLEFT(ubleft); irbp++) {
-                        /*
+                        struct xfs_bulkstat_agichunk ac;
-                         * Now process this chunk of inodes.
-                         */
+                        ac.ac_lastino = lastino;
-                        for (agino = irbp->ir_startino, chunkidx = clustidx = 0;
+                        ac.ac_ubuffer = &ubuffer;
-                             XFS_BULKSTAT_UBLEFT(ubleft) &&
+                        ac.ac_ubleft = ubleft;
-                                irbp->ir_freecount < XFS_INODES_PER_CHUNK;
+                        ac.ac_ubelem = ubelem;
-                             chunkidx++, clustidx++, agino++) {
+                        error = xfs_bulkstat_ag_ichunk(mp, agno, irbp,
-                                ASSERT(chunkidx < XFS_INODES_PER_CHUNK);
+                                        formatter, statstruct_size, &ac);
+                        if (error)
-                                ino = XFS_AGINO_TO_INO(mp, agno, agino);
+                                rval = error;
-                                /*
-                                 * Skip if this inode is free.
+                        lastino = ac.ac_lastino;
-                                 */
+                        ubleft = ac.ac_ubleft;
-                                if (XFS_INOBT_MASK(chunkidx) & irbp->ir_free) {
+                        ubelem = ac.ac_ubelem;
-                                        lastino = ino;
-                                        continue;
-                                }
-                                /*
-                                 * Count used inodes as free so we can tell
-                                 * when the chunk is used up.
-                                 */
-                                irbp->ir_freecount++;
-                                /*
-                                 * Get the inode and fill in a single buffer.
-                                 */
-                                ubused = statstruct_size;
-                                error = formatter(mp, ino, ubufp, ubleft,
-                                                  &ubused, &fmterror);
-                                if (fmterror == BULKSTAT_RV_NOTHING) {
-                                        if (error && error != ENOENT &&
-                                                error != EINVAL) {
-                                                ubleft = 0;
-                                                rval = error;
-                                                break;
-                                        }
-                                        lastino = ino;
-                                        continue;
-                                }
-                                if (fmterror == BULKSTAT_RV_GIVEUP) {
-                                        ubleft = 0;
-                                        ASSERT(error);
-                                        rval = error;
-                                        break;
-                                }
-                                if (ubufp)
-                                        ubufp += ubused;
-                                ubleft -= ubused;
-                                ubelem++;
-                                lastino = ino;
-                        }
                        cond_resched();
                }
@@ -512,58 +542,10 @@ xfs_bulkstat(
        return rval;
 }
-/*
- * Return stat information in bulk (by-inode) for the filesystem.
- * Special case for non-sequential one inode bulkstat.
- */
-int                                     /* error status */
-xfs_bulkstat_single(
-        xfs_mount_t             *mp,    /* mount point for filesystem */
-        xfs_ino_t               *lastinop, /* inode to return */
-        char                    __user *buffer, /* buffer with inode stats */
-        int                     *done)  /* 1 if there are more stats to get */
-{
-        int                     count;  /* count value for bulkstat call */
-        int                     error;  /* return value */
-        xfs_ino_t               ino;    /* filesystem inode number */
-        int                     res;    /* result from bs1 */
-        /*
-         * note that requesting valid inode numbers which are not allocated
-         * to inodes will most likely cause xfs_imap_to_bp to generate warning
-         * messages about bad magic numbers. This is ok. The fact that
-         * the inode isn't actually an inode is handled by the
-         * error check below. Done this way to make the usual case faster
-         * at the expense of the error case.
-         */
-        ino = *lastinop;
-        error = xfs_bulkstat_one(mp, ino, buffer, sizeof(xfs_bstat_t),
-                                 NULL, &res);
-        if (error) {
-                /*
-                 * Special case way failed, do it the "long" way
-                 * to see if that works.
-                 */
-                (*lastinop)--;
-                count = 1;
-                if (xfs_bulkstat(mp, lastinop, &count, xfs_bulkstat_one,
-                                sizeof(xfs_bstat_t), buffer, done))
-                        return error;
-                if (count == 0 || (xfs_ino_t)*lastinop != ino)
-                        return error == EFSCORRUPTED ?
-                                XFS_ERROR(EINVAL) : error;
-                else
-                        return 0;
-        }
-        *done = 0;
-        return 0;
-}
 int
 xfs_inumbers_fmt(
        void                    __user *ubuffer, /* buffer to write to */
-        const xfs_inogrp_t      *buffer,        /* buffer to read from */
+        const struct xfs_inogrp *buffer,        /* buffer to read from */
        long                    count,          /* # of elements to read */
        long                    *written)       /* # of bytes written */
 {
@@ -578,127 +560,104 @@ xfs_inumbers_fmt(
 */
 int                                     /* error status */
 xfs_inumbers(
-        xfs_mount_t     *mp,            /* mount point for filesystem */
+        struct xfs_mount        *mp,/* mount point for filesystem */
-        xfs_ino_t       *lastino,       /* last inode returned */
+        xfs_ino_t               *lastino,/* last inode returned */
-        int             *count,         /* size of buffer/count returned */
+        int                     *count,/* size of buffer/count returned */
-        void            __user *ubuffer,/* buffer with inode descriptions */
+        void                    __user *ubuffer,/* buffer with inode descriptions */
-        inumbers_fmt_pf formatter)
+        inumbers_fmt_pf         formatter)
 {
-        xfs_buf_t       *agbp;
+        xfs_agnumber_t          agno = XFS_INO_TO_AGNO(mp, *lastino);
-        xfs_agino_t     agino;
+        xfs_agino_t             agino = XFS_INO_TO_AGINO(mp, *lastino);
-        xfs_agnumber_t  agno;
+        struct xfs_btree_cur    *cur = NULL;
-        int             bcount;
+        struct xfs_buf          *agbp = NULL;
-        xfs_inogrp_t    *buffer;
+        struct xfs_inogrp       *buffer;
-        int             bufidx;
+        int                     bcount;
-        xfs_btree_cur_t *cur;
+        int                     left = *count;
-        int             error;
+        int                     bufidx = 0;
-        xfs_inobt_rec_incore_t r;
+        int                     error = 0;
-        int             i;
-        xfs_ino_t       ino;
-        int             left;
-        int             tmp;
-        ino = (xfs_ino_t)*lastino;
-        agno = XFS_INO_TO_AGNO(mp, ino);
-        agino = XFS_INO_TO_AGINO(mp, ino);
-        left = *count;
        *count = 0;
+        if (agno >= mp->m_sb.sb_agcount ||
+            *lastino != XFS_AGINO_TO_INO(mp, agno, agino))
+                return error;
        bcount = MIN(left, (int)(PAGE_SIZE / sizeof(*buffer)));
        buffer = kmem_alloc(bcount * sizeof(*buffer), KM_SLEEP);
-        error = bufidx = 0;
+        do {
-        cur = NULL;
+                struct xfs_inobt_rec_incore     r;
-        agbp = NULL;
+                int                             stat;
-        while (left > 0 && agno < mp->m_sb.sb_agcount) {
-                if (agbp == NULL) {
+                if (!agbp) {
                        error = xfs_ialloc_read_agi(mp, NULL, agno, &agbp);
-                        if (error) {
+                        if (error)
-                                /*
+                                break;
-                                 * If we can't read the AGI of this ag,
-                                 * then just skip to the next one.
-                                 */
-                                ASSERT(cur == NULL);
-                                agbp = NULL;
-                                agno++;
-                                agino = 0;
-                                continue;
-                        }
                        cur = xfs_inobt_init_cursor(mp, NULL, agbp, agno,
                                                    XFS_BTNUM_INO);
                        error = xfs_inobt_lookup(cur, agino, XFS_LOOKUP_GE,
-                                                 &tmp);
+                                                 &stat);
-                        if (error) {
+                        if (error)
-                                xfs_btree_del_cursor(cur, XFS_BTREE_ERROR);
+                                break;
-                                cur = NULL;
+                        if (!stat)
-                                xfs_buf_relse(agbp);
+                                goto next_ag;
-                                agbp = NULL;
-                                /*
-                                 * Move up the last inode in the current
-                                 * chunk.  The lookup_ge will always get
-                                 * us the first inode in the next chunk.
-                                 */
-                                agino += XFS_INODES_PER_CHUNK - 1;
-                                continue;
-                        }
-                }
-                error = xfs_inobt_get_rec(cur, &r, &i);
-                if (error || i == 0) {
-                        xfs_buf_relse(agbp);
-                        agbp = NULL;
-                        xfs_btree_del_cursor(cur, XFS_BTREE_NOERROR);
-                        cur = NULL;
-                        agno++;
-                        agino = 0;
-                        continue;
                }
+                error = xfs_inobt_get_rec(cur, &r, &stat);
+                if (error)
+                        break;
+                if (!stat)
+                        goto next_ag;
                agino = r.ir_startino + XFS_INODES_PER_CHUNK - 1;
                buffer[bufidx].xi_startino =
                        XFS_AGINO_TO_INO(mp, agno, r.ir_startino);
                buffer[bufidx].xi_alloccount =
                        XFS_INODES_PER_CHUNK - r.ir_freecount;
                buffer[bufidx].xi_allocmask = ~r.ir_free;
-                bufidx++;
+                if (++bufidx == bcount) {
-                left--;
+                        long    written;
-                if (bufidx == bcount) {
-                        long written;
+                        error = formatter(ubuffer, buffer, bufidx, &written);
-                        if (formatter(ubuffer, buffer, bufidx, &written)) {
+                        if (error)
-                                error = XFS_ERROR(EFAULT);
                                break;
-                        }
                        ubuffer += written;
                        *count += bufidx;
                        bufidx = 0;
                }
-                if (left) {
+                if (!--left)
-                        error = xfs_btree_increment(cur, 0, &tmp);
+                        break;
-                        if (error) {
-                                xfs_btree_del_cursor(cur, XFS_BTREE_ERROR);
+                error = xfs_btree_increment(cur, 0, &stat);
-                                cur = NULL;
+                if (error)
-                                xfs_buf_relse(agbp);
+                        break;
-                                agbp = NULL;
+                if (stat)
-                                /*
+                        continue;
-                                 * The agino value has already been bumped.
-                                 * Just try to skip up to it.
+next_ag:
-                                 */
+                xfs_btree_del_cursor(cur, XFS_BTREE_ERROR);
-                                agino += XFS_INODES_PER_CHUNK;
+                cur = NULL;
-                                continue;
+                xfs_buf_relse(agbp);
-                        }
+                agbp = NULL;
-                }
+                agino = 0;
-        }
+        } while (++agno < mp->m_sb.sb_agcount);
        if (!error) {
                if (bufidx) {
-                        long written;
+                        long    written;
-                        if (formatter(ubuffer, buffer, bufidx, &written))
-                                error = XFS_ERROR(EFAULT);
+                        error = formatter(ubuffer, buffer, bufidx, &written);
-                        else
+                        if (!error)
                                *count += bufidx;
                }
                *lastino = XFS_AGINO_TO_INO(mp, agno, agino);
        }
        kmem_free(buffer);
        if (cur)
                xfs_btree_del_cursor(cur, (error ? XFS_BTREE_ERROR :
                                           XFS_BTREE_NOERROR));
        if (agbp)
                xfs_buf_relse(agbp);
        return error;
 }
diff --git a/fs/xfs/xfs_itable.h b/fs/xfs/xfs_itable.h
index 97295d91d170..aaed08022eb9 100644
--- a/fs/xfs/xfs_itable.h
+++ b/fs/xfs/xfs_itable.h
@@ -30,6 +30,22 @@ typedef int (*bulkstat_one_pf)(struct xfs_mount	*mp,
                               int              *ubused,
                               int              *stat);
+struct xfs_bulkstat_agichunk {
+        xfs_ino_t       ac_lastino;     /* last inode returned */
+        char            __user **ac_ubuffer;/* pointer into user's buffer */
+        int             ac_ubleft;      /* bytes left in user's buffer */
+        int             ac_ubelem;      /* spaces used in user's buffer */
+};
+int
+xfs_bulkstat_ag_ichunk(
+        struct xfs_mount                *mp,
+        xfs_agnumber_t                  agno,
+        struct xfs_inobt_rec_incore     *irbp,
+        bulkstat_one_pf                 formatter,
+        size_t                          statstruct_size,
+        struct xfs_bulkstat_agichunk    *acp);
 /*
 * Values for stat return value.
 */
@@ -50,13 +66,6 @@ xfs_bulkstat(
        char            __user *ubuffer,/* buffer with inode stats */
        int             *done);         /* 1 if there are more stats to get */
-int
-xfs_bulkstat_single(
-        xfs_mount_t             *mp,
-        xfs_ino_t               *lastinop,
-        char                    __user *buffer,
-        int                     *done);
 typedef int (*bulkstat_one_fmt_pf)(  /* used size in bytes or negative error */
        void                    __user *ubuffer, /* buffer to write to */
        int                     ubsize,          /* remaining user buffer sz */
diff --git a/fs/xfs/xfs_linux.h b/fs/xfs/xfs_linux.h
index 825249d2dfc1..d10dc8f397c9 100644
--- a/fs/xfs/xfs_linux.h
+++ b/fs/xfs/xfs_linux.h
@@ -21,18 +21,6 @@
 #include <linux/types.h>
 /*
- * XFS_BIG_BLKNOS needs block layer disk addresses to be 64 bits.
- * XFS_BIG_INUMS requires XFS_BIG_BLKNOS to be set.
- */
-#if defined(CONFIG_LBDAF) || (BITS_PER_LONG == 64)
-# define XFS_BIG_BLKNOS 1
-# define XFS_BIG_INUMS  1
-#else
-# define XFS_BIG_BLKNOS 0
-# define XFS_BIG_INUMS  0
-#endif
-/*
 * Kernel specific type declarations for XFS
 */
 typedef signed char             __int8_t;
@@ -113,7 +101,7 @@ typedef __uint64_t __psunsigned_t;
 #include <asm/byteorder.h>
 #include <asm/unaligned.h>
-#include "xfs_vnode.h"
+#include "xfs_fs.h"
 #include "xfs_stats.h"
 #include "xfs_sysctl.h"
 #include "xfs_iops.h"
@@ -191,6 +179,17 @@ typedef __uint64_t __psunsigned_t;
 #define MAX(a,b)        (max(a,b))
 #define howmany(x, y)   (((x)+((y)-1))/(y))
+/*
+ * XFS wrapper structure for sysfs support. It depends on external data
+ * structures and is embedded in various internal data structures to implement
+ * the XFS sysfs object heirarchy. Define it here for broad access throughout
+ * the codebase.
+ */
+struct xfs_kobj {
+        struct kobject          kobject;
+        struct completion       complete;
+};
 /* Kernel uid/gid conversion. These are used to convert to/from the on disk
 * uid_t/gid_t types to the kuid_t/kgid_t types that the kernel uses internally.
 * The conversion here is type only, the value will remain the same since we
@@ -331,7 +330,7 @@ static inline __uint64_t roundup_64(__uint64_t x, __uint32_t y)
 {
        x += y - 1;
        do_div(x, y);
-        return(x * y);
+        return x * y;
 }
 static inline __uint64_t howmany_64(__uint64_t x, __uint32_t y)
diff --git a/fs/xfs/xfs_log.c b/fs/xfs/xfs_log.c
index 292308dede6d..ca4fd5bd8522 100644
--- a/fs/xfs/xfs_log.c
+++ b/fs/xfs/xfs_log.c
@@ -34,6 +34,7 @@
 #include "xfs_trace.h"
 #include "xfs_fsops.h"
 #include "xfs_cksum.h"
+#include "xfs_sysfs.h"
 kmem_zone_t     *xfs_log_ticket_zone;
@@ -283,7 +284,7 @@ xlog_grant_head_wait(
        return 0;
 shutdown:
        list_del_init(&tic->t_queue);
-        return XFS_ERROR(EIO);
+        return -EIO;
 }
 /*
@@ -377,7 +378,7 @@ xfs_log_regrant(
        int                     error = 0;
        if (XLOG_FORCED_SHUTDOWN(log))
-                return XFS_ERROR(EIO);
+                return -EIO;
        XFS_STATS_INC(xs_try_logspace);
@@ -446,7 +447,7 @@ xfs_log_reserve(
        ASSERT(client == XFS_TRANSACTION || client == XFS_LOG);
        if (XLOG_FORCED_SHUTDOWN(log))
-                return XFS_ERROR(EIO);
+                return -EIO;
        XFS_STATS_INC(xs_try_logspace);
@@ -454,7 +455,7 @@ xfs_log_reserve(
        tic = xlog_ticket_alloc(log, unit_bytes, cnt, client, permanent,
                                KM_SLEEP | KM_MAYFAIL);
        if (!tic)
-                return XFS_ERROR(ENOMEM);
+                return -ENOMEM;
        tic->t_trans_type = t_type;
        *ticp = tic;
@@ -590,7 +591,7 @@ xfs_log_release_iclog(
 {
        if (xlog_state_release_iclog(mp->m_log, iclog)) {
                xfs_force_shutdown(mp, SHUTDOWN_LOG_IO_ERROR);
-                return EIO;
+                return -EIO;
        }
        return 0;
@@ -628,7 +629,7 @@ xfs_log_mount(
        mp->m_log = xlog_alloc_log(mp, log_target, blk_offset, num_bblks);
        if (IS_ERR(mp->m_log)) {
-                error = -PTR_ERR(mp->m_log);
+                error = PTR_ERR(mp->m_log);
                goto out;
        }
@@ -652,18 +653,18 @@ xfs_log_mount(
                xfs_warn(mp,
                "Log size %d blocks too small, minimum size is %d blocks",
                         mp->m_sb.sb_logblocks, min_logfsbs);
-                error = EINVAL;
+                error = -EINVAL;
        } else if (mp->m_sb.sb_logblocks > XFS_MAX_LOG_BLOCKS) {
                xfs_warn(mp,
                "Log size %d blocks too large, maximum size is %lld blocks",
                         mp->m_sb.sb_logblocks, XFS_MAX_LOG_BLOCKS);
-                error = EINVAL;
+                error = -EINVAL;
        } else if (XFS_FSB_TO_B(mp, mp->m_sb.sb_logblocks) > XFS_MAX_LOG_BYTES) {
                xfs_warn(mp,
                "log size %lld bytes too large, maximum size is %lld bytes",
                         XFS_FSB_TO_B(mp, mp->m_sb.sb_logblocks),
                         XFS_MAX_LOG_BYTES);
-                error = EINVAL;
+                error = -EINVAL;
        }
        if (error) {
                if (xfs_sb_version_hascrc(&mp->m_sb)) {
@@ -707,6 +708,11 @@ xfs_log_mount(
                }
        }
+        error = xfs_sysfs_init(&mp->m_log->l_kobj, &xfs_log_ktype, &mp->m_kobj,
+                               "log");
+        if (error)
+                goto out_destroy_ail;
        /* Normal transactions can now occur */
        mp->m_log->l_flags &= ~XLOG_ACTIVE_RECOVERY;
@@ -947,6 +953,9 @@ xfs_log_unmount(
        xfs_log_quiesce(mp);
        xfs_trans_ail_destroy(mp);
+        xfs_sysfs_del(&mp->m_log->l_kobj);
        xlog_dealloc_log(mp->m_log);
 }
@@ -1313,7 +1322,7 @@ xlog_alloc_log(
        xlog_in_core_t          *iclog, *prev_iclog=NULL;
        xfs_buf_t               *bp;
        int                     i;
-        int                     error = ENOMEM;
+        int                     error = -ENOMEM;
        uint                    log2_size = 0;
        log = kmem_zalloc(sizeof(struct xlog), KM_MAYFAIL);
@@ -1340,7 +1349,7 @@ xlog_alloc_log(
        xlog_grant_head_init(&log->l_reserve_head);
        xlog_grant_head_init(&log->l_write_head);
-        error = EFSCORRUPTED;
+        error = -EFSCORRUPTED;
        if (xfs_sb_version_hassector(&mp->m_sb)) {
                log2_size = mp->m_sb.sb_logsectlog;
                if (log2_size < BBSHIFT) {
@@ -1369,8 +1378,14 @@ xlog_alloc_log(
        xlog_get_iclog_buffer_size(mp, log);
-        error = ENOMEM;
+        /*
-        bp = xfs_buf_alloc(mp->m_logdev_targp, 0, BTOBB(log->l_iclog_size), 0);
+         * Use a NULL block for the extra log buffer used during splits so that
+         * it will trigger errors if we ever try to do IO on it without first
+         * having set it up properly.
+         */
+        error = -ENOMEM;
+        bp = xfs_buf_alloc(mp->m_logdev_targp, XFS_BUF_DADDR_NULL,
+                           BTOBB(log->l_iclog_size), 0);
        if (!bp)
                goto out_free_log;
@@ -1463,7 +1478,7 @@ out_free_iclog:
 out_free_log:
        kmem_free(log);
 out:
-        return ERR_PTR(-error);
+        return ERR_PTR(error);
 }       /* xlog_alloc_log */
@@ -1661,7 +1676,7 @@ xlog_bdstrat(
        xfs_buf_lock(bp);
        if (iclog->ic_state & XLOG_STATE_IOERROR) {
-                xfs_buf_ioerror(bp, EIO);
+                xfs_buf_ioerror(bp, -EIO);
                xfs_buf_stale(bp);
                xfs_buf_ioend(bp, 0);
                /*
@@ -2360,7 +2375,7 @@ xlog_write(
                        ophdr = xlog_write_setup_ophdr(log, ptr, ticket, flags);
                        if (!ophdr)
-                                return XFS_ERROR(EIO);
+                                return -EIO;
                        xlog_write_adv_cnt(&ptr, &len, &log_offset,
                                           sizeof(struct xlog_op_header));
@@ -2859,7 +2874,7 @@ restart:
        spin_lock(&log->l_icloglock);
        if (XLOG_FORCED_SHUTDOWN(log)) {
                spin_unlock(&log->l_icloglock);
-                return XFS_ERROR(EIO);
+                return -EIO;
        }
        iclog = log->l_iclog;
@@ -3047,7 +3062,7 @@ xlog_state_release_iclog(
        int             sync = 0;       /* do we sync? */
        if (iclog->ic_state & XLOG_STATE_IOERROR)
-                return XFS_ERROR(EIO);
+                return -EIO;
        ASSERT(atomic_read(&iclog->ic_refcnt) > 0);
        if (!atomic_dec_and_lock(&iclog->ic_refcnt, &log->l_icloglock))
@@ -3055,7 +3070,7 @@ xlog_state_release_iclog(
        if (iclog->ic_state & XLOG_STATE_IOERROR) {
                spin_unlock(&log->l_icloglock);
-                return XFS_ERROR(EIO);
+                return -EIO;
        }
        ASSERT(iclog->ic_state == XLOG_STATE_ACTIVE ||
               iclog->ic_state == XLOG_STATE_WANT_SYNC);
@@ -3172,7 +3187,7 @@ _xfs_log_force(
        iclog = log->l_iclog;
        if (iclog->ic_state & XLOG_STATE_IOERROR) {
                spin_unlock(&log->l_icloglock);
-                return XFS_ERROR(EIO);
+                return -EIO;
        }
        /* If the head iclog is not active nor dirty, we just attach
@@ -3210,7 +3225,7 @@ _xfs_log_force(
                                spin_unlock(&log->l_icloglock);
                                if (xlog_state_release_iclog(log, iclog))
-                                        return XFS_ERROR(EIO);
+                                        return -EIO;
                                if (log_flushed)
                                        *log_flushed = 1;
@@ -3246,7 +3261,7 @@ maybe_sleep:
                 */
                if (iclog->ic_state & XLOG_STATE_IOERROR) {
                        spin_unlock(&log->l_icloglock);
-                        return XFS_ERROR(EIO);
+                        return -EIO;
                }
                XFS_STATS_INC(xs_log_force_sleep);
                xlog_wait(&iclog->ic_force_wait, &log->l_icloglock);
@@ -3256,7 +3271,7 @@ maybe_sleep:
                 * and the memory read should be atomic.
                 */
                if (iclog->ic_state & XLOG_STATE_IOERROR)
-                        return XFS_ERROR(EIO);
+                        return -EIO;
                if (log_flushed)
                        *log_flushed = 1;
        } else {
@@ -3324,7 +3339,7 @@ try_again:
        iclog = log->l_iclog;
        if (iclog->ic_state & XLOG_STATE_IOERROR) {
                spin_unlock(&log->l_icloglock);
-                return XFS_ERROR(EIO);
+                return -EIO;
        }
        do {
@@ -3375,7 +3390,7 @@ try_again:
                        xlog_state_switch_iclogs(log, iclog, 0);
                        spin_unlock(&log->l_icloglock);
                        if (xlog_state_release_iclog(log, iclog))
-                                return XFS_ERROR(EIO);
+                                return -EIO;
                        if (log_flushed)
                                *log_flushed = 1;
                        spin_lock(&log->l_icloglock);
@@ -3390,7 +3405,7 @@ try_again:
                         */
                        if (iclog->ic_state & XLOG_STATE_IOERROR) {
                                spin_unlock(&log->l_icloglock);
-                                return XFS_ERROR(EIO);
+                                return -EIO;
                        }
                        XFS_STATS_INC(xs_log_force_sleep);
                        xlog_wait(&iclog->ic_force_wait, &log->l_icloglock);
@@ -3400,7 +3415,7 @@ try_again:
                         * and the memory read should be atomic.
                         */
                        if (iclog->ic_state & XLOG_STATE_IOERROR)
-                                return XFS_ERROR(EIO);
+                                return -EIO;
                        if (log_flushed)
                                *log_flushed = 1;
diff --git a/fs/xfs/xfs_log_cil.c b/fs/xfs/xfs_log_cil.c
index b3425b34e3d5..f6b79e5325dd 100644
--- a/fs/xfs/xfs_log_cil.c
+++ b/fs/xfs/xfs_log_cil.c
@@ -78,8 +78,6 @@ xlog_cil_init_post_recovery(
 {
        log->l_cilp->xc_ctx->ticket = xlog_cil_ticket_alloc(log);
        log->l_cilp->xc_ctx->sequence = 1;
-        log->l_cilp->xc_ctx->commit_lsn = xlog_assign_lsn(log->l_curr_cycle,
-                                                                log->l_curr_block);
 }
 /*
@@ -634,7 +632,7 @@ out_abort_free_ticket:
        xfs_log_ticket_put(tic);
 out_abort:
        xlog_cil_committed(ctx, XFS_LI_ABORTED);
-        return XFS_ERROR(EIO);
+        return -EIO;
 }
 static void
@@ -928,12 +926,12 @@ xlog_cil_init(
        cil = kmem_zalloc(sizeof(*cil), KM_SLEEP|KM_MAYFAIL);
        if (!cil)
-                return ENOMEM;
+                return -ENOMEM;
        ctx = kmem_zalloc(sizeof(*ctx), KM_SLEEP|KM_MAYFAIL);
        if (!ctx) {
                kmem_free(cil);
-                return ENOMEM;
+                return -ENOMEM;
        }
        INIT_WORK(&cil->xc_push_work, xlog_cil_push_work);
diff --git a/fs/xfs/xfs_log_priv.h b/fs/xfs/xfs_log_priv.h
index 9bc403a9e54f..db7cbdeb2b42 100644
--- a/fs/xfs/xfs_log_priv.h
+++ b/fs/xfs/xfs_log_priv.h
@@ -405,6 +405,8 @@ struct xlog {
        struct xlog_grant_head  l_reserve_head;
        struct xlog_grant_head  l_write_head;
+        struct xfs_kobj         l_kobj;
        /* The following field are used for debugging; need to hold icloglock */
 #ifdef DEBUG
        char                    *l_iclog_bak[XLOG_MAX_ICLOGS];
diff --git a/fs/xfs/xfs_log_recover.c b/fs/xfs/xfs_log_recover.c
index 981af0f6504b..1fd5787add99 100644
--- a/fs/xfs/xfs_log_recover.c
+++ b/fs/xfs/xfs_log_recover.c
@@ -179,7 +179,7 @@ xlog_bread_noalign(
                xfs_warn(log->l_mp, "Invalid block length (0x%x) for buffer",
                        nbblks);
                XFS_ERROR_REPORT(__func__, XFS_ERRLEVEL_HIGH, log->l_mp);
-                return EFSCORRUPTED;
+                return -EFSCORRUPTED;
        }
        blk_no = round_down(blk_no, log->l_sectBBsize);
@@ -194,7 +194,7 @@ xlog_bread_noalign(
        bp->b_error = 0;
        if (XFS_FORCED_SHUTDOWN(log->l_mp))
-                return XFS_ERROR(EIO);
+                return -EIO;
        xfs_buf_iorequest(bp);
        error = xfs_buf_iowait(bp);
@@ -268,7 +268,7 @@ xlog_bwrite(
                xfs_warn(log->l_mp, "Invalid block length (0x%x) for buffer",
                        nbblks);
                XFS_ERROR_REPORT(__func__, XFS_ERRLEVEL_HIGH, log->l_mp);
-                return EFSCORRUPTED;
+                return -EFSCORRUPTED;
        }
        blk_no = round_down(blk_no, log->l_sectBBsize);
@@ -330,14 +330,14 @@ xlog_header_check_recover(
                xlog_header_check_dump(mp, head);
                XFS_ERROR_REPORT("xlog_header_check_recover(1)",
                                 XFS_ERRLEVEL_HIGH, mp);
-                return XFS_ERROR(EFSCORRUPTED);
+                return -EFSCORRUPTED;
        } else if (unlikely(!uuid_equal(&mp->m_sb.sb_uuid, &head->h_fs_uuid))) {
                xfs_warn(mp,
        "dirty log entry has mismatched uuid - can't recover");
                xlog_header_check_dump(mp, head);
                XFS_ERROR_REPORT("xlog_header_check_recover(2)",
                                 XFS_ERRLEVEL_HIGH, mp);
-                return XFS_ERROR(EFSCORRUPTED);
+                return -EFSCORRUPTED;
        }
        return 0;
 }
@@ -364,7 +364,7 @@ xlog_header_check_mount(
                xlog_header_check_dump(mp, head);
                XFS_ERROR_REPORT("xlog_header_check_mount",
                                 XFS_ERRLEVEL_HIGH, mp);
-                return XFS_ERROR(EFSCORRUPTED);
+                return -EFSCORRUPTED;
        }
        return 0;
 }
@@ -462,7 +462,7 @@ xlog_find_verify_cycle(
        while (!(bp = xlog_get_bp(log, bufblks))) {
                bufblks >>= 1;
                if (bufblks < log->l_sectBBsize)
-                        return ENOMEM;
+                        return -ENOMEM;
        }
        for (i = start_blk; i < start_blk + nbblks; i += bufblks) {
@@ -524,7 +524,7 @@ xlog_find_verify_log_record(
        if (!(bp = xlog_get_bp(log, num_blks))) {
                if (!(bp = xlog_get_bp(log, 1)))
-                        return ENOMEM;
+                        return -ENOMEM;
                smallmem = 1;
        } else {
                error = xlog_bread(log, start_blk, num_blks, bp, &offset);
@@ -539,7 +539,7 @@ xlog_find_verify_log_record(
                        xfs_warn(log->l_mp,
                "Log inconsistent (didn't find previous header)");
                        ASSERT(0);
-                        error = XFS_ERROR(EIO);
+                        error = -EIO;
                        goto out;
                }
@@ -564,7 +564,7 @@ xlog_find_verify_log_record(
         * will be called again for the end of the physical log.
         */
        if (i == -1) {
-                error = -1;
+                error = 1;
                goto out;
        }
@@ -628,7 +628,12 @@ xlog_find_head(
        int             error, log_bbnum = log->l_logBBsize;
        /* Is the end of the log device zeroed? */
-        if ((error = xlog_find_zeroed(log, &first_blk)) == -1) {
+        error = xlog_find_zeroed(log, &first_blk);
+        if (error < 0) {
+                xfs_warn(log->l_mp, "empty log check failed");
+                return error;
+        }
+        if (error == 1) {
                *return_head_blk = first_blk;
                /* Is the whole lot zeroed? */
@@ -641,15 +646,12 @@ xlog_find_head(
                }
                return 0;
-        } else if (error) {
-                xfs_warn(log->l_mp, "empty log check failed");
-                return error;
        }
        first_blk = 0;                  /* get cycle # of 1st block */
        bp = xlog_get_bp(log, 1);
        if (!bp)
-                return ENOMEM;
+                return -ENOMEM;
        error = xlog_bread(log, 0, 1, bp, &offset);
        if (error)
@@ -818,29 +820,29 @@ validate_head:
                start_blk = head_blk - num_scan_bblks; /* don't read head_blk */
                /* start ptr at last block ptr before head_blk */
-                if ((error = xlog_find_verify_log_record(log, start_blk,
+                error = xlog_find_verify_log_record(log, start_blk, &head_blk, 0);
-                                                        &head_blk, 0)) == -1) {
+                if (error == 1)
-                        error = XFS_ERROR(EIO);
+                        error = -EIO;
-                        goto bp_err;
+                if (error)
-                } else if (error)
                        goto bp_err;
        } else {
                start_blk = 0;
                ASSERT(head_blk <= INT_MAX);
-                if ((error = xlog_find_verify_log_record(log, start_blk,
+                error = xlog_find_verify_log_record(log, start_blk, &head_blk, 0);
-                                                        &head_blk, 0)) == -1) {
+                if (error < 0)
+                        goto bp_err;
+                if (error == 1) {
                        /* We hit the beginning of the log during our search */
                        start_blk = log_bbnum - (num_scan_bblks - head_blk);
                        new_blk = log_bbnum;
                        ASSERT(start_blk <= INT_MAX &&
                                (xfs_daddr_t) log_bbnum-start_blk >= 0);
                        ASSERT(head_blk <= INT_MAX);
-                        if ((error = xlog_find_verify_log_record(log,
+                        error = xlog_find_verify_log_record(log, start_blk,
-                                                        start_blk, &new_blk,
+                                                        &new_blk, (int)head_blk);
-                                                        (int)head_blk)) == -1) {
+                        if (error == 1)
-                                error = XFS_ERROR(EIO);
+                                error = -EIO;
-                                goto bp_err;
+                        if (error)
-                        } else if (error)
                                goto bp_err;
                        if (new_blk != log_bbnum)
                                head_blk = new_blk;
@@ -911,7 +913,7 @@ xlog_find_tail(
        bp = xlog_get_bp(log, 1);
        if (!bp)
-                return ENOMEM;
+                return -ENOMEM;
        if (*head_blk == 0) {                           /* special case */
                error = xlog_bread(log, 0, 1, bp, &offset);
                if (error)
@@ -961,7 +963,7 @@ xlog_find_tail(
                xfs_warn(log->l_mp, "%s: couldn't find sync record", __func__);
                xlog_put_bp(bp);
                ASSERT(0);
-                return XFS_ERROR(EIO);
+                return -EIO;
        }
        /* find blk_no of tail of log */
@@ -1092,8 +1094,8 @@ done:
 *
 * Return:
 *      0  => the log is completely written to
- *      -1 => use *blk_no as the first block of the log
+ *      1 => use *blk_no as the first block of the log
- *      >0 => error has occurred
+ *      <0 => error has occurred
 */
 STATIC int
 xlog_find_zeroed(
@@ -1112,7 +1114,7 @@ xlog_find_zeroed(
        /* check totally zeroed log */
        bp = xlog_get_bp(log, 1);
        if (!bp)
-                return ENOMEM;
+                return -ENOMEM;
        error = xlog_bread(log, 0, 1, bp, &offset);
        if (error)
                goto bp_err;
@@ -1121,7 +1123,7 @@ xlog_find_zeroed(
        if (first_cycle == 0) {         /* completely zeroed log */
                *blk_no = 0;
                xlog_put_bp(bp);
-                return -1;
+                return 1;
        }
        /* check partially zeroed log */
@@ -1141,7 +1143,7 @@ xlog_find_zeroed(
                 */
                xfs_warn(log->l_mp,
                        "Log inconsistent or not a log (last==0, first!=1)");
-                error = XFS_ERROR(EINVAL);
+                error = -EINVAL;
                goto bp_err;
        }
@@ -1179,19 +1181,18 @@ xlog_find_zeroed(
         * Potentially backup over partial log record write.  We don't need
         * to search the end of the log because we know it is zero.
         */
-        if ((error = xlog_find_verify_log_record(log, start_blk,
+        error = xlog_find_verify_log_record(log, start_blk, &last_blk, 0);
-                                &last_blk, 0)) == -1) {
+        if (error == 1)
-            error = XFS_ERROR(EIO);
+                error = -EIO;
-            goto bp_err;
+        if (error)
-        } else if (error)
+                goto bp_err;
-            goto bp_err;
        *blk_no = last_blk;
 bp_err:
        xlog_put_bp(bp);
        if (error)
                return error;
-        return -1;
+        return 1;
 }
 /*
@@ -1251,7 +1252,7 @@ xlog_write_log_records(
        while (!(bp = xlog_get_bp(log, bufblks))) {
                bufblks >>= 1;
                if (bufblks < sectbb)
-                        return ENOMEM;
+                        return -ENOMEM;
        }
        /* We may need to do a read at the start to fill in part of
@@ -1354,7 +1355,7 @@ xlog_clear_stale_blocks(
                if (unlikely(head_block < tail_block || head_block >= log->l_logBBsize)) {
                        XFS_ERROR_REPORT("xlog_clear_stale_blocks(1)",
                                         XFS_ERRLEVEL_LOW, log->l_mp);
-                        return XFS_ERROR(EFSCORRUPTED);
+                        return -EFSCORRUPTED;
                }
                tail_distance = tail_block + (log->l_logBBsize - head_block);
        } else {
@@ -1366,7 +1367,7 @@ xlog_clear_stale_blocks(
                if (unlikely(head_block >= tail_block || head_cycle != (tail_cycle + 1))){
                        XFS_ERROR_REPORT("xlog_clear_stale_blocks(2)",
                                         XFS_ERRLEVEL_LOW, log->l_mp);
-                        return XFS_ERROR(EFSCORRUPTED);
+                        return -EFSCORRUPTED;
                }
                tail_distance = tail_block - head_block;
        }
@@ -1551,7 +1552,7 @@ xlog_recover_add_to_trans(
                        xfs_warn(log->l_mp, "%s: bad header magic number",
                                __func__);
                        ASSERT(0);
-                        return XFS_ERROR(EIO);
+                        return -EIO;
                }
                if (len == sizeof(xfs_trans_header_t))
                        xlog_recover_add_item(&trans->r_itemq);
@@ -1581,7 +1582,7 @@ xlog_recover_add_to_trans(
                                  in_f->ilf_size);
                        ASSERT(0);
                        kmem_free(ptr);
-                        return XFS_ERROR(EIO);
+                        return -EIO;
                }
                item->ri_total = in_f->ilf_size;
@@ -1702,7 +1703,7 @@ xlog_recover_reorder_trans(
                         */
                        if (!list_empty(&sort_list))
                                list_splice_init(&sort_list, &trans->r_itemq);
-                        error = XFS_ERROR(EIO);
+                        error = -EIO;
                        goto out;
                }
        }
@@ -1943,7 +1944,7 @@ xlog_recover_do_inode_buffer(
                                item, bp);
                        XFS_ERROR_REPORT("xlog_recover_do_inode_buf",
                                         XFS_ERRLEVEL_LOW, mp);
-                        return XFS_ERROR(EFSCORRUPTED);
+                        return -EFSCORRUPTED;
                }
                buffer_nextp = (xfs_agino_t *)xfs_buf_offset(bp,
@@ -2125,6 +2126,17 @@ xlog_recover_validate_buf_type(
        __uint16_t              magic16;
        __uint16_t              magicda;
+        /*
+         * We can only do post recovery validation on items on CRC enabled
+         * fielsystems as we need to know when the buffer was written to be able
+         * to determine if we should have replayed the item. If we replay old
+         * metadata over a newer buffer, then it will enter a temporarily
+         * inconsistent state resulting in verification failures. Hence for now
+         * just avoid the verification stage for non-crc filesystems
+         */
+        if (!xfs_sb_version_hascrc(&mp->m_sb))
+                return;
        magic32 = be32_to_cpu(*(__be32 *)bp->b_addr);
        magic16 = be16_to_cpu(*(__be16*)bp->b_addr);
        magicda = be16_to_cpu(info->magic);
@@ -2162,8 +2174,6 @@ xlog_recover_validate_buf_type(
                bp->b_ops = &xfs_agf_buf_ops;
                break;
        case XFS_BLFT_AGFL_BUF:
-                if (!xfs_sb_version_hascrc(&mp->m_sb))
-                        break;
                if (magic32 != XFS_AGFL_MAGIC) {
                        xfs_warn(mp, "Bad AGFL block magic!");
                        ASSERT(0);
@@ -2196,10 +2206,6 @@ xlog_recover_validate_buf_type(
 #endif
                break;
        case XFS_BLFT_DINO_BUF:
-                /*
-                 * we get here with inode allocation buffers, not buffers that
-                 * track unlinked list changes.
-                 */
                if (magic16 != XFS_DINODE_MAGIC) {
                        xfs_warn(mp, "Bad INODE block magic!");
                        ASSERT(0);
@@ -2279,8 +2285,6 @@ xlog_recover_validate_buf_type(
                bp->b_ops = &xfs_attr3_leaf_buf_ops;
                break;
        case XFS_BLFT_ATTR_RMT_BUF:
-                if (!xfs_sb_version_hascrc(&mp->m_sb))
-                        break;
                if (magic32 != XFS_ATTR3_RMT_MAGIC) {
                        xfs_warn(mp, "Bad attr remote magic!");
                        ASSERT(0);
@@ -2387,16 +2391,7 @@ xlog_recover_do_reg_buffer(
        /* Shouldn't be any more regions */
        ASSERT(i == item->ri_total);
-        /*
+        xlog_recover_validate_buf_type(mp, bp, buf_f);
-         * We can only do post recovery validation on items on CRC enabled
-         * fielsystems as we need to know when the buffer was written to be able
-         * to determine if we should have replayed the item. If we replay old
-         * metadata over a newer buffer, then it will enter a temporarily
-         * inconsistent state resulting in verification failures. Hence for now
-         * just avoid the verification stage for non-crc filesystems
-         */
-        if (xfs_sb_version_hascrc(&mp->m_sb))
-                xlog_recover_validate_buf_type(mp, bp, buf_f);
 }
 /*
@@ -2404,8 +2399,11 @@ xlog_recover_do_reg_buffer(
 * Simple algorithm: if we have found a QUOTAOFF log item of the same type
 * (ie. USR or GRP), then just toss this buffer away; don't recover it.
 * Else, treat it as a regular buffer and do recovery.
+ *
+ * Return false if the buffer was tossed and true if we recovered the buffer to
+ * indicate to the caller if the buffer needs writing.
 */
-STATIC void
+STATIC bool
 xlog_recover_do_dquot_buffer(
        struct xfs_mount                *mp,
        struct xlog                     *log,
@@ -2420,9 +2418,8 @@ xlog_recover_do_dquot_buffer(
        /*
         * Filesystems are required to send in quota flags at mount time.
         */
-        if (mp->m_qflags == 0) {
+        if (!mp->m_qflags)
-                return;
+                return false;
-        }
        type = 0;
        if (buf_f->blf_flags & XFS_BLF_UDQUOT_BUF)
@@ -2435,9 +2432,10 @@ xlog_recover_do_dquot_buffer(
         * This type of quotas was turned off, so ignore this buffer
         */
        if (log->l_quotaoffs_flag & type)
-                return;
+                return false;
        xlog_recover_do_reg_buffer(mp, item, bp, buf_f);
+        return true;
 }
 /*
@@ -2496,7 +2494,7 @@ xlog_recover_buffer_pass2(
        bp = xfs_buf_read(mp->m_ddev_targp, buf_f->blf_blkno, buf_f->blf_len,
                          buf_flags, NULL);
        if (!bp)
-                return XFS_ERROR(ENOMEM);
+                return -ENOMEM;
        error = bp->b_error;
        if (error) {
                xfs_buf_ioerror_alert(bp, "xlog_recover_do..(read#1)");
@@ -2504,23 +2502,44 @@ xlog_recover_buffer_pass2(
        }
        /*
-         * recover the buffer only if we get an LSN from it and it's less than
+         * Recover the buffer only if we get an LSN from it and it's less than
         * the lsn of the transaction we are replaying.
+         *
+         * Note that we have to be extremely careful of readahead here.
+         * Readahead does not attach verfiers to the buffers so if we don't
+         * actually do any replay after readahead because of the LSN we found
+         * in the buffer if more recent than that current transaction then we
+         * need to attach the verifier directly. Failure to do so can lead to
+         * future recovery actions (e.g. EFI and unlinked list recovery) can
+         * operate on the buffers and they won't get the verifier attached. This
+         * can lead to blocks on disk having the correct content but a stale
+         * CRC.
+         *
+         * It is safe to assume these clean buffers are currently up to date.
+         * If the buffer is dirtied by a later transaction being replayed, then
+         * the verifier will be reset to match whatever recover turns that
+         * buffer into.
         */
        lsn = xlog_recover_get_buf_lsn(mp, bp);
-        if (lsn && lsn != -1 && XFS_LSN_CMP(lsn, current_lsn) >= 0)
+        if (lsn && lsn != -1 && XFS_LSN_CMP(lsn, current_lsn) >= 0) {
+                xlog_recover_validate_buf_type(mp, bp, buf_f);
                goto out_release;
+        }
        if (buf_f->blf_flags & XFS_BLF_INODE_BUF) {
                error = xlog_recover_do_inode_buffer(mp, item, bp, buf_f);
+                if (error)
+                        goto out_release;
        } else if (buf_f->blf_flags &
                  (XFS_BLF_UDQUOT_BUF|XFS_BLF_PDQUOT_BUF|XFS_BLF_GDQUOT_BUF)) {
-                xlog_recover_do_dquot_buffer(mp, log, item, bp, buf_f);
+                bool    dirty;
+                dirty = xlog_recover_do_dquot_buffer(mp, log, item, bp, buf_f);
+                if (!dirty)
+                        goto out_release;
        } else {
                xlog_recover_do_reg_buffer(mp, item, bp, buf_f);
        }
-        if (error)
-                goto out_release;
        /*
         * Perform delayed write on the buffer.  Asynchronous writes will be
@@ -2598,7 +2617,7 @@ xfs_recover_inode_owner_change(
        ip = xfs_inode_alloc(mp, in_f->ilf_ino);
        if (!ip)
-                return ENOMEM;
+                return -ENOMEM;
        /* instantiate the inode */
        xfs_dinode_from_disk(&ip->i_d, dip);
@@ -2676,7 +2695,7 @@ xlog_recover_inode_pass2(
        bp = xfs_buf_read(mp->m_ddev_targp, in_f->ilf_blkno, in_f->ilf_len, 0,
                          &xfs_inode_buf_ops);
        if (!bp) {
-                error = ENOMEM;
+                error = -ENOMEM;
                goto error;
        }
        error = bp->b_error;
@@ -2697,7 +2716,7 @@ xlog_recover_inode_pass2(
                        __func__, dip, bp, in_f->ilf_ino);
                XFS_ERROR_REPORT("xlog_recover_inode_pass2(1)",
                                 XFS_ERRLEVEL_LOW, mp);
-                error = EFSCORRUPTED;
+                error = -EFSCORRUPTED;
                goto out_release;
        }
        dicp = item->ri_buf[1].i_addr;
@@ -2707,7 +2726,7 @@ xlog_recover_inode_pass2(
                        __func__, item, in_f->ilf_ino);
                XFS_ERROR_REPORT("xlog_recover_inode_pass2(2)",
                                 XFS_ERRLEVEL_LOW, mp);
-                error = EFSCORRUPTED;
+                error = -EFSCORRUPTED;
                goto out_release;
        }
@@ -2764,7 +2783,7 @@ xlog_recover_inode_pass2(
                "%s: Bad regular inode log record, rec ptr 0x%p, "
                "ino ptr = 0x%p, ino bp = 0x%p, ino %Ld",
                                __func__, item, dip, bp, in_f->ilf_ino);
-                        error = EFSCORRUPTED;
+                        error = -EFSCORRUPTED;
                        goto out_release;
                }
        } else if (unlikely(S_ISDIR(dicp->di_mode))) {
@@ -2777,7 +2796,7 @@ xlog_recover_inode_pass2(
                "%s: Bad dir inode log record, rec ptr 0x%p, "
                "ino ptr = 0x%p, ino bp = 0x%p, ino %Ld",
                                __func__, item, dip, bp, in_f->ilf_ino);
-                        error = EFSCORRUPTED;
+                        error = -EFSCORRUPTED;
                        goto out_release;
                }
        }
@@ -2790,7 +2809,7 @@ xlog_recover_inode_pass2(
                        __func__, item, dip, bp, in_f->ilf_ino,
                        dicp->di_nextents + dicp->di_anextents,
                        dicp->di_nblocks);
-                error = EFSCORRUPTED;
+                error = -EFSCORRUPTED;
                goto out_release;
        }
        if (unlikely(dicp->di_forkoff > mp->m_sb.sb_inodesize)) {
@@ -2800,7 +2819,7 @@ xlog_recover_inode_pass2(
        "%s: Bad inode log record, rec ptr 0x%p, dino ptr 0x%p, "
        "dino bp 0x%p, ino %Ld, forkoff 0x%x", __func__,
                        item, dip, bp, in_f->ilf_ino, dicp->di_forkoff);
-                error = EFSCORRUPTED;
+                error = -EFSCORRUPTED;
                goto out_release;
        }
        isize = xfs_icdinode_size(dicp->di_version);
@@ -2810,7 +2829,7 @@ xlog_recover_inode_pass2(
                xfs_alert(mp,
                        "%s: Bad inode log record length %d, rec ptr 0x%p",
                        __func__, item->ri_buf[1].i_len, item);
-                error = EFSCORRUPTED;
+                error = -EFSCORRUPTED;
                goto out_release;
        }
@@ -2898,7 +2917,7 @@ xlog_recover_inode_pass2(
                default:
                        xfs_warn(log->l_mp, "%s: Invalid flag", __func__);
                        ASSERT(0);
-                        error = EIO;
+                        error = -EIO;
                        goto out_release;
                }
        }
@@ -2919,7 +2938,7 @@ out_release:
 error:
        if (need_free)
                kmem_free(in_f);
-        return XFS_ERROR(error);
+        return error;
 }
 /*
@@ -2946,7 +2965,7 @@ xlog_recover_quotaoff_pass1(
        if (qoff_f->qf_flags & XFS_GQUOTA_ACCT)
                log->l_quotaoffs_flag |= XFS_DQ_GROUP;
-        return (0);
+        return 0;
 }
 /*
@@ -2971,17 +2990,17 @@ xlog_recover_dquot_pass2(
         * Filesystems are required to send in quota flags at mount time.
         */
        if (mp->m_qflags == 0)
-                return (0);
+                return 0;
        recddq = item->ri_buf[1].i_addr;
        if (recddq == NULL) {
                xfs_alert(log->l_mp, "NULL dquot in %s.", __func__);
-                return XFS_ERROR(EIO);
+                return -EIO;
        }
        if (item->ri_buf[1].i_len < sizeof(xfs_disk_dquot_t)) {
                xfs_alert(log->l_mp, "dquot too small (%d) in %s.",
                        item->ri_buf[1].i_len, __func__);
-                return XFS_ERROR(EIO);
+                return -EIO;
        }
        /*
@@ -2990,7 +3009,7 @@ xlog_recover_dquot_pass2(
        type = recddq->d_flags & (XFS_DQ_USER | XFS_DQ_PROJ | XFS_DQ_GROUP);
        ASSERT(type);
        if (log->l_quotaoffs_flag & type)
-                return (0);
+                return 0;
        /*
         * At this point we know that quota was _not_ turned off.
@@ -3007,12 +3026,19 @@ xlog_recover_dquot_pass2(
        error = xfs_dqcheck(mp, recddq, dq_f->qlf_id, 0, XFS_QMOPT_DOWARN,
                           "xlog_recover_dquot_pass2 (log copy)");
        if (error)
-                return XFS_ERROR(EIO);
+                return -EIO;
        ASSERT(dq_f->qlf_len == 1);
+        /*
+         * At this point we are assuming that the dquots have been allocated
+         * and hence the buffer has valid dquots stamped in it. It should,
+         * therefore, pass verifier validation. If the dquot is bad, then the
+         * we'll return an error here, so we don't need to specifically check
+         * the dquot in the buffer after the verifier has run.
+         */
        error = xfs_trans_read_buf(mp, NULL, mp->m_ddev_targp, dq_f->qlf_blkno,
                                   XFS_FSB_TO_BB(mp, dq_f->qlf_len), 0, &bp,
-                                   NULL);
+                                   &xfs_dquot_buf_ops);
        if (error)
                return error;
@@ -3020,18 +3046,6 @@ xlog_recover_dquot_pass2(
        ddq = (xfs_disk_dquot_t *)xfs_buf_offset(bp, dq_f->qlf_boffset);
        /*
-         * At least the magic num portion should be on disk because this
-         * was among a chunk of dquots created earlier, and we did some
-         * minimal initialization then.
-         */
-        error = xfs_dqcheck(mp, ddq, dq_f->qlf_id, 0, XFS_QMOPT_DOWARN,
-                           "xlog_recover_dquot_pass2");
-        if (error) {
-                xfs_buf_relse(bp);
-                return XFS_ERROR(EIO);
-        }
-        /*
         * If the dquot has an LSN in it, recover the dquot only if it's less
         * than the lsn of the transaction we are replaying.
         */
@@ -3178,38 +3192,38 @@ xlog_recover_do_icreate_pass2(
        icl = (struct xfs_icreate_log *)item->ri_buf[0].i_addr;
        if (icl->icl_type != XFS_LI_ICREATE) {
                xfs_warn(log->l_mp, "xlog_recover_do_icreate_trans: bad type");
-                return EINVAL;
+                return -EINVAL;
        }
        if (icl->icl_size != 1) {
                xfs_warn(log->l_mp, "xlog_recover_do_icreate_trans: bad icl size");
-                return EINVAL;
+                return -EINVAL;
        }
        agno = be32_to_cpu(icl->icl_ag);
        if (agno >= mp->m_sb.sb_agcount) {
                xfs_warn(log->l_mp, "xlog_recover_do_icreate_trans: bad agno");
-                return EINVAL;
+                return -EINVAL;
        }
        agbno = be32_to_cpu(icl->icl_agbno);
        if (!agbno || agbno == NULLAGBLOCK || agbno >= mp->m_sb.sb_agblocks) {
                xfs_warn(log->l_mp, "xlog_recover_do_icreate_trans: bad agbno");
-                return EINVAL;
+                return -EINVAL;
        }
        isize = be32_to_cpu(icl->icl_isize);
        if (isize != mp->m_sb.sb_inodesize) {
                xfs_warn(log->l_mp, "xlog_recover_do_icreate_trans: bad isize");
-                return EINVAL;
+                return -EINVAL;
        }
        count = be32_to_cpu(icl->icl_count);
        if (!count) {
                xfs_warn(log->l_mp, "xlog_recover_do_icreate_trans: bad count");
-                return EINVAL;
+                return -EINVAL;
        }
        length = be32_to_cpu(icl->icl_length);
        if (!length || length >= mp->m_sb.sb_agblocks) {
                xfs_warn(log->l_mp, "xlog_recover_do_icreate_trans: bad length");
-                return EINVAL;
+                return -EINVAL;
        }
        /* existing allocation is fixed value */
@@ -3218,7 +3232,7 @@ xlog_recover_do_icreate_pass2(
        if (count != mp->m_ialloc_inos ||
             length != mp->m_ialloc_blks) {
                xfs_warn(log->l_mp, "xlog_recover_do_icreate_trans: bad count 2");
-                return EINVAL;
+                return -EINVAL;
        }
        /*
@@ -3389,7 +3403,7 @@ xlog_recover_commit_pass1(
                xfs_warn(log->l_mp, "%s: invalid item type (%d)",
                        __func__, ITEM_TYPE(item));
                ASSERT(0);
-                return XFS_ERROR(EIO);
+                return -EIO;
        }
 }
@@ -3425,7 +3439,7 @@ xlog_recover_commit_pass2(
                xfs_warn(log->l_mp, "%s: invalid item type (%d)",
                        __func__, ITEM_TYPE(item));
                ASSERT(0);
-                return XFS_ERROR(EIO);
+                return -EIO;
        }
 }
@@ -3560,7 +3574,7 @@ xlog_recover_process_data(
        /* check the log format matches our own - else we can't recover */
        if (xlog_header_check_recover(log->l_mp, rhead))
-                return (XFS_ERROR(EIO));
+                return -EIO;
        while ((dp < lp) && num_logops) {
                ASSERT(dp + sizeof(xlog_op_header_t) <= lp);
@@ -3571,7 +3585,7 @@ xlog_recover_process_data(
                        xfs_warn(log->l_mp, "%s: bad clientid 0x%x",
                                        __func__, ohead->oh_clientid);
                        ASSERT(0);
-                        return (XFS_ERROR(EIO));
+                        return -EIO;
                }
                tid = be32_to_cpu(ohead->oh_tid);
                hash = XLOG_RHASH(tid);
@@ -3585,7 +3599,7 @@ xlog_recover_process_data(
                                xfs_warn(log->l_mp, "%s: bad length 0x%x",
                                        __func__, be32_to_cpu(ohead->oh_len));
                                WARN_ON(1);
-                                return (XFS_ERROR(EIO));
+                                return -EIO;
                        }
                        flags = ohead->oh_flags & ~XLOG_END_TRANS;
                        if (flags & XLOG_WAS_CONT_TRANS)
@@ -3607,7 +3621,7 @@ xlog_recover_process_data(
                                xfs_warn(log->l_mp, "%s: bad transaction",
                                        __func__);
                                ASSERT(0);
-                                error = XFS_ERROR(EIO);
+                                error = -EIO;
                                break;
                        case 0:
                        case XLOG_CONTINUE_TRANS:
@@ -3618,7 +3632,7 @@ xlog_recover_process_data(
                                xfs_warn(log->l_mp, "%s: bad flag 0x%x",
                                        __func__, flags);
                                ASSERT(0);
-                                error = XFS_ERROR(EIO);
+                                error = -EIO;
                                break;
                        }
                        if (error) {
@@ -3669,7 +3683,7 @@ xlog_recover_process_efi(
                         */
                        set_bit(XFS_EFI_RECOVERED, &efip->efi_flags);
                        xfs_efi_release(efip, efip->efi_format.efi_nextents);
-                        return XFS_ERROR(EIO);
+                        return -EIO;
                }
        }
@@ -3969,7 +3983,7 @@ xlog_unpack_data_crc(
                 * CRC protection by punting an error back up the stack.
                 */
                if (xfs_sb_version_hascrc(&log->l_mp->m_sb))
-                        return EFSCORRUPTED;
+                        return -EFSCORRUPTED;
        }
        return 0;
@@ -4018,14 +4032,14 @@ xlog_valid_rec_header(
        if (unlikely(rhead->h_magicno != cpu_to_be32(XLOG_HEADER_MAGIC_NUM))) {
                XFS_ERROR_REPORT("xlog_valid_rec_header(1)",
                                XFS_ERRLEVEL_LOW, log->l_mp);
-                return XFS_ERROR(EFSCORRUPTED);
+                return -EFSCORRUPTED;
        }
        if (unlikely(
            (!rhead->h_version ||
            (be32_to_cpu(rhead->h_version) & (~XLOG_VERSION_OKBITS))))) {
                xfs_warn(log->l_mp, "%s: unrecognised log version (%d).",
                        __func__, be32_to_cpu(rhead->h_version));
-                return XFS_ERROR(EIO);
+                return -EIO;
        }
        /* LR body must have data or it wouldn't have been written */
@@ -4033,12 +4047,12 @@ xlog_valid_rec_header(
        if (unlikely( hlen <= 0 || hlen > INT_MAX )) {
                XFS_ERROR_REPORT("xlog_valid_rec_header(2)",
                                XFS_ERRLEVEL_LOW, log->l_mp);
-                return XFS_ERROR(EFSCORRUPTED);
+                return -EFSCORRUPTED;
        }
        if (unlikely( blkno > log->l_logBBsize || blkno > INT_MAX )) {
                XFS_ERROR_REPORT("xlog_valid_rec_header(3)",
                                XFS_ERRLEVEL_LOW, log->l_mp);
-                return XFS_ERROR(EFSCORRUPTED);
+                return -EFSCORRUPTED;
        }
        return 0;
 }
@@ -4081,7 +4095,7 @@ xlog_do_recovery_pass(
                 */
                hbp = xlog_get_bp(log, 1);
                if (!hbp)
-                        return ENOMEM;
+                        return -ENOMEM;
                error = xlog_bread(log, tail_blk, 1, hbp, &offset);
                if (error)
@@ -4110,11 +4124,11 @@ xlog_do_recovery_pass(
        }
        if (!hbp)
-                return ENOMEM;
+                return -ENOMEM;
        dbp = xlog_get_bp(log, BTOBB(h_size));
        if (!dbp) {
                xlog_put_bp(hbp);
-                return ENOMEM;
+                return -ENOMEM;
        }
        memset(rhash, 0, sizeof(rhash));
@@ -4388,7 +4402,7 @@ xlog_do_recover(
         * If IO errors happened during recovery, bail out.
         */
        if (XFS_FORCED_SHUTDOWN(log->l_mp)) {
-                return (EIO);
+                return -EIO;
        }
        /*
@@ -4415,7 +4429,7 @@ xlog_do_recover(
        if (XFS_FORCED_SHUTDOWN(log->l_mp)) {
                xfs_buf_relse(bp);
-                return XFS_ERROR(EIO);
+                return -EIO;
        }
        xfs_buf_iorequest(bp);
@@ -4492,7 +4506,7 @@ xlog_recover(
 "Please recover the log on a kernel that supports the unknown features.",
                                (log->l_mp->m_sb.sb_features_log_incompat &
                                        XFS_SB_FEAT_INCOMPAT_LOG_UNKNOWN));
-                        return EINVAL;
+                        return -EINVAL;
                }
                xfs_notice(log->l_mp, "Starting recovery (logdev: %s)",
diff --git a/fs/xfs/xfs_mount.c b/fs/xfs/xfs_mount.c
index 3507cd0ec400..fbf0384a466f 100644
--- a/fs/xfs/xfs_mount.c
+++ b/fs/xfs/xfs_mount.c
@@ -42,6 +42,7 @@
 #include "xfs_trace.h"
 #include "xfs_icache.h"
 #include "xfs_dinode.h"
+#include "xfs_sysfs.h"
 #ifdef HAVE_PERCPU_SB
@@ -60,6 +61,8 @@ static DEFINE_MUTEX(xfs_uuid_table_mutex);
 static int xfs_uuid_table_size;
 static uuid_t *xfs_uuid_table;
+extern struct kset *xfs_kset;
 /*
 * See if the UUID is unique among mounted XFS filesystems.
 * Mount fails if UUID is nil or a FS with the same UUID is already mounted.
@@ -76,7 +79,7 @@ xfs_uuid_mount(
        if (uuid_is_nil(uuid)) {
                xfs_warn(mp, "Filesystem has nil UUID - can't mount");
-                return XFS_ERROR(EINVAL);
+                return -EINVAL;
        }
        mutex_lock(&xfs_uuid_table_mutex);
@@ -104,7 +107,7 @@ xfs_uuid_mount(
 out_duplicate:
        mutex_unlock(&xfs_uuid_table_mutex);
        xfs_warn(mp, "Filesystem has duplicate UUID %pU - can't mount", uuid);
-        return XFS_ERROR(EINVAL);
+        return -EINVAL;
 }
 STATIC void
@@ -173,13 +176,9 @@ xfs_sb_validate_fsb_count(
        ASSERT(PAGE_SHIFT >= sbp->sb_blocklog);
        ASSERT(sbp->sb_blocklog >= BBSHIFT);
-#if XFS_BIG_BLKNOS     /* Limited by ULONG_MAX of page cache index */
+        /* Limited by ULONG_MAX of page cache index */
        if (nblocks >> (PAGE_CACHE_SHIFT - sbp->sb_blocklog) > ULONG_MAX)
-                return EFBIG;
+                return -EFBIG;
-#else                  /* Limited by UINT_MAX of sectors */
-        if (nblocks << (sbp->sb_blocklog - BBSHIFT) > UINT_MAX)
-                return EFBIG;
-#endif
        return 0;
 }
@@ -250,9 +249,9 @@ xfs_initialize_perag(
                mp->m_flags &= ~XFS_MOUNT_32BITINODES;
        if (mp->m_flags & XFS_MOUNT_32BITINODES)
-                index = xfs_set_inode32(mp);
+                index = xfs_set_inode32(mp, agcount);
        else
-                index = xfs_set_inode64(mp);
+                index = xfs_set_inode64(mp, agcount);
        if (maxagi)
                *maxagi = index;
@@ -308,15 +307,15 @@ reread:
        if (!bp) {
                if (loud)
                        xfs_warn(mp, "SB buffer read failed");
-                return EIO;
+                return -EIO;
        }
        if (bp->b_error) {
                error = bp->b_error;
                if (loud)
                        xfs_warn(mp, "SB validate failed with error %d.", error);
                /* bad CRC means corrupted metadata */
-                if (error == EFSBADCRC)
+                if (error == -EFSBADCRC)
-                        error = EFSCORRUPTED;
+                        error = -EFSCORRUPTED;
                goto release_buf;
        }
@@ -324,7 +323,6 @@ reread:
         * Initialize the mount structure from the superblock.
         */
        xfs_sb_from_disk(sbp, XFS_BUF_TO_SBP(bp));
-        xfs_sb_quota_from_disk(sbp);
        /*
         * If we haven't validated the superblock, do so now before we try
@@ -333,7 +331,7 @@ reread:
        if (sbp->sb_magicnum != XFS_SB_MAGIC) {
                if (loud)
                        xfs_warn(mp, "Invalid superblock magic number");
-                error = EINVAL;
+                error = -EINVAL;
                goto release_buf;
        }
@@ -344,7 +342,7 @@ reread:
                if (loud)
                        xfs_warn(mp, "device supports %u byte sectors (not %u)",
                                sector_size, sbp->sb_sectsize);
-                error = ENOSYS;
+                error = -ENOSYS;
                goto release_buf;
        }
@@ -392,7 +390,7 @@ xfs_update_alignment(xfs_mount_t *mp)
                        xfs_warn(mp,
                "alignment check failed: sunit/swidth vs. blocksize(%d)",
                                sbp->sb_blocksize);
-                        return XFS_ERROR(EINVAL);
+                        return -EINVAL;
                } else {
                        /*
                         * Convert the stripe unit and width to FSBs.
@@ -402,14 +400,14 @@ xfs_update_alignment(xfs_mount_t *mp)
                                xfs_warn(mp,
                        "alignment check failed: sunit/swidth vs. agsize(%d)",
                                         sbp->sb_agblocks);
-                                return XFS_ERROR(EINVAL);
+                                return -EINVAL;
                        } else if (mp->m_dalign) {
                                mp->m_swidth = XFS_BB_TO_FSBT(mp, mp->m_swidth);
                        } else {
                                xfs_warn(mp,
                        "alignment check failed: sunit(%d) less than bsize(%d)",
                                         mp->m_dalign, sbp->sb_blocksize);
-                                return XFS_ERROR(EINVAL);
+                                return -EINVAL;
                        }
                }
@@ -429,7 +427,7 @@ xfs_update_alignment(xfs_mount_t *mp)
                } else {
                        xfs_warn(mp,
        "cannot change alignment: superblock does not support data alignment");
-                        return XFS_ERROR(EINVAL);
+                        return -EINVAL;
                }
        } else if ((mp->m_flags & XFS_MOUNT_NOALIGN) != XFS_MOUNT_NOALIGN &&
                    xfs_sb_version_hasdalign(&mp->m_sb)) {
@@ -556,14 +554,14 @@ xfs_check_sizes(xfs_mount_t *mp)
        d = (xfs_daddr_t)XFS_FSB_TO_BB(mp, mp->m_sb.sb_dblocks);
        if (XFS_BB_TO_FSB(mp, d) != mp->m_sb.sb_dblocks) {
                xfs_warn(mp, "filesystem size mismatch detected");
-                return XFS_ERROR(EFBIG);
+                return -EFBIG;
        }
        bp = xfs_buf_read_uncached(mp->m_ddev_targp,
                                        d - XFS_FSS_TO_BB(mp, 1),
                                        XFS_FSS_TO_BB(mp, 1), 0, NULL);
        if (!bp) {
                xfs_warn(mp, "last sector read failed");
-                return EIO;
+                return -EIO;
        }
        xfs_buf_relse(bp);
@@ -571,14 +569,14 @@ xfs_check_sizes(xfs_mount_t *mp)
                d = (xfs_daddr_t)XFS_FSB_TO_BB(mp, mp->m_sb.sb_logblocks);
                if (XFS_BB_TO_FSB(mp, d) != mp->m_sb.sb_logblocks) {
                        xfs_warn(mp, "log size mismatch detected");
-                        return XFS_ERROR(EFBIG);
+                        return -EFBIG;
                }
                bp = xfs_buf_read_uncached(mp->m_logdev_targp,
                                        d - XFS_FSB_TO_BB(mp, 1),
                                        XFS_FSB_TO_BB(mp, 1), 0, NULL);
                if (!bp) {
                        xfs_warn(mp, "log device read failed");
-                        return EIO;
+                        return -EIO;
                }
                xfs_buf_relse(bp);
        }
@@ -731,10 +729,15 @@ xfs_mountfs(
        xfs_set_maxicount(mp);
-        error = xfs_uuid_mount(mp);
+        mp->m_kobj.kobject.kset = xfs_kset;
+        error = xfs_sysfs_init(&mp->m_kobj, &xfs_mp_ktype, NULL, mp->m_fsname);
        if (error)
                goto out;
+        error = xfs_uuid_mount(mp);
+        if (error)
+                goto out_remove_sysfs;
        /*
         * Set the minimum read and write sizes
         */
@@ -816,7 +819,7 @@ xfs_mountfs(
        if (!sbp->sb_logblocks) {
                xfs_warn(mp, "no log defined");
                XFS_ERROR_REPORT("xfs_mountfs", XFS_ERRLEVEL_LOW, mp);
-                error = XFS_ERROR(EFSCORRUPTED);
+                error = -EFSCORRUPTED;
                goto out_free_perag;
        }
@@ -855,7 +858,7 @@ xfs_mountfs(
             !mp->m_sb.sb_inprogress) {
                error = xfs_initialize_perag_data(mp, sbp->sb_agcount);
                if (error)
-                        goto out_fail_wait;
+                        goto out_log_dealloc;
        }
        /*
@@ -876,7 +879,7 @@ xfs_mountfs(
                xfs_iunlock(rip, XFS_ILOCK_EXCL);
                XFS_ERROR_REPORT("xfs_mountfs_int(2)", XFS_ERRLEVEL_LOW,
                                 mp);
-                error = XFS_ERROR(EFSCORRUPTED);
+                error = -EFSCORRUPTED;
                goto out_rele_rip;
        }
        mp->m_rootip = rip;     /* save it */
@@ -927,7 +930,7 @@ xfs_mountfs(
                        xfs_notice(mp, "resetting quota flags");
                        error = xfs_mount_reset_sbqflags(mp);
                        if (error)
-                                return error;
+                                goto out_rtunmount;
                }
        }
@@ -989,6 +992,8 @@ xfs_mountfs(
        xfs_da_unmount(mp);
 out_remove_uuid:
        xfs_uuid_unmount(mp);
+ out_remove_sysfs:
+        xfs_sysfs_del(&mp->m_kobj);
 out:
        return error;
 }
@@ -1071,6 +1076,8 @@ xfs_unmountfs(
        xfs_errortag_clearall(mp, 0);
 #endif
        xfs_free_perag(mp);
+        xfs_sysfs_del(&mp->m_kobj);
 }
 int
@@ -1152,7 +1159,7 @@ xfs_mod_incore_sb_unlocked(
                lcounter += delta;
                if (lcounter < 0) {
                        ASSERT(0);
-                        return XFS_ERROR(EINVAL);
+                        return -EINVAL;
                }
                mp->m_sb.sb_icount = lcounter;
                return 0;
@@ -1161,7 +1168,7 @@ xfs_mod_incore_sb_unlocked(
                lcounter += delta;
                if (lcounter < 0) {
                        ASSERT(0);
-                        return XFS_ERROR(EINVAL);
+                        return -EINVAL;
                }
                mp->m_sb.sb_ifree = lcounter;
                return 0;
@@ -1191,7 +1198,7 @@ xfs_mod_incore_sb_unlocked(
                         * blocks if were allowed to.
                         */
                        if (!rsvd)
-                                return XFS_ERROR(ENOSPC);
+                                return -ENOSPC;
                        lcounter = (long long)mp->m_resblks_avail + delta;
                        if (lcounter >= 0) {
@@ -1202,7 +1209,7 @@ xfs_mod_incore_sb_unlocked(
                                "Filesystem \"%s\": reserve blocks depleted! "
                                "Consider increasing reserve pool size.",
                                mp->m_fsname);
-                        return XFS_ERROR(ENOSPC);
+                        return -ENOSPC;
                }
                mp->m_sb.sb_fdblocks = lcounter + XFS_ALLOC_SET_ASIDE(mp);
@@ -1211,7 +1218,7 @@ xfs_mod_incore_sb_unlocked(
                lcounter = (long long)mp->m_sb.sb_frextents;
                lcounter += delta;
                if (lcounter < 0) {
-                        return XFS_ERROR(ENOSPC);
+                        return -ENOSPC;
                }
                mp->m_sb.sb_frextents = lcounter;
                return 0;
@@ -1220,7 +1227,7 @@ xfs_mod_incore_sb_unlocked(
                lcounter += delta;
                if (lcounter < 0) {
                        ASSERT(0);
-                        return XFS_ERROR(EINVAL);
+                        return -EINVAL;
                }
                mp->m_sb.sb_dblocks = lcounter;
                return 0;
@@ -1229,7 +1236,7 @@ xfs_mod_incore_sb_unlocked(
                scounter += delta;
                if (scounter < 0) {
                        ASSERT(0);
-                        return XFS_ERROR(EINVAL);
+                        return -EINVAL;
                }
                mp->m_sb.sb_agcount = scounter;
                return 0;
@@ -1238,7 +1245,7 @@ xfs_mod_incore_sb_unlocked(
                scounter += delta;
                if (scounter < 0) {
                        ASSERT(0);
-                        return XFS_ERROR(EINVAL);
+                        return -EINVAL;
                }
                mp->m_sb.sb_imax_pct = scounter;
                return 0;
@@ -1247,7 +1254,7 @@ xfs_mod_incore_sb_unlocked(
                scounter += delta;
                if (scounter < 0) {
                        ASSERT(0);
-                        return XFS_ERROR(EINVAL);
+                        return -EINVAL;
                }
                mp->m_sb.sb_rextsize = scounter;
                return 0;
@@ -1256,7 +1263,7 @@ xfs_mod_incore_sb_unlocked(
                scounter += delta;
                if (scounter < 0) {
                        ASSERT(0);
-                        return XFS_ERROR(EINVAL);
+                        return -EINVAL;
                }
                mp->m_sb.sb_rbmblocks = scounter;
                return 0;
@@ -1265,7 +1272,7 @@ xfs_mod_incore_sb_unlocked(
                lcounter += delta;
                if (lcounter < 0) {
                        ASSERT(0);
-                        return XFS_ERROR(EINVAL);
+                        return -EINVAL;
                }
                mp->m_sb.sb_rblocks = lcounter;
                return 0;
@@ -1274,7 +1281,7 @@ xfs_mod_incore_sb_unlocked(
                lcounter += delta;
                if (lcounter < 0) {
                        ASSERT(0);
-                        return XFS_ERROR(EINVAL);
+                        return -EINVAL;
                }
                mp->m_sb.sb_rextents = lcounter;
                return 0;
@@ -1283,13 +1290,13 @@ xfs_mod_incore_sb_unlocked(
                scounter += delta;
                if (scounter < 0) {
                        ASSERT(0);
-                        return XFS_ERROR(EINVAL);
+                        return -EINVAL;
                }
                mp->m_sb.sb_rextslog = scounter;
                return 0;
        default:
                ASSERT(0);
-                return XFS_ERROR(EINVAL);
+                return -EINVAL;
        }
 }
@@ -1452,7 +1459,7 @@ xfs_dev_is_read_only(
            (mp->m_rtdev_targp && xfs_readonly_buftarg(mp->m_rtdev_targp))) {
                xfs_notice(mp, "%s required on read-only device.", message);
                xfs_notice(mp, "write access unavailable, cannot proceed.");
-                return EROFS;
+                return -EROFS;
        }
        return 0;
 }
@@ -1995,7 +2002,7 @@ slow_path:
         * (e.g. lots of space just got freed). After that
         * we are done.
         */
-        if (ret != ENOSPC)
+        if (ret != -ENOSPC)
                xfs_icsb_balance_counter(mp, field, 0);
        xfs_icsb_unlock(mp);
        return ret;
diff --git a/fs/xfs/xfs_mount.h b/fs/xfs/xfs_mount.h
index 7295a0b7c343..b0447c86e7e2 100644
--- a/fs/xfs/xfs_mount.h
+++ b/fs/xfs/xfs_mount.h
@@ -166,6 +166,7 @@ typedef struct xfs_mount {
                                                   on the next remount,rw */
        int64_t                 m_low_space[XFS_LOWSP_MAX];
                                                /* low free space thresholds */
+        struct xfs_kobj         m_kobj;
        struct workqueue_struct *m_data_workqueue;
        struct workqueue_struct *m_unwritten_workqueue;
diff --git a/fs/xfs/xfs_mru_cache.c b/fs/xfs/xfs_mru_cache.c
index f99b4933dc22..1eb6f3df698c 100644
--- a/fs/xfs/xfs_mru_cache.c
+++ b/fs/xfs/xfs_mru_cache.c
@@ -337,20 +337,20 @@ xfs_mru_cache_create(
                *mrup = NULL;
        if (!mrup || !grp_count || !lifetime_ms || !free_func)
-                return EINVAL;
+                return -EINVAL;
        if (!(grp_time = msecs_to_jiffies(lifetime_ms) / grp_count))
-                return EINVAL;
+                return -EINVAL;
        if (!(mru = kmem_zalloc(sizeof(*mru), KM_SLEEP)))
-                return ENOMEM;
+                return -ENOMEM;
        /* An extra list is needed to avoid reaping up to a grp_time early. */
        mru->grp_count = grp_count + 1;
        mru->lists = kmem_zalloc(mru->grp_count * sizeof(*mru->lists), KM_SLEEP);
        if (!mru->lists) {
-                err = ENOMEM;
+                err = -ENOMEM;
                goto exit;
        }
@@ -434,16 +434,16 @@ xfs_mru_cache_insert(
        ASSERT(mru && mru->lists);
        if (!mru || !mru->lists)
-                return EINVAL;
+                return -EINVAL;
        if (radix_tree_preload(GFP_KERNEL))
-                return ENOMEM;
+                return -ENOMEM;
        INIT_LIST_HEAD(&elem->list_node);
        elem->key = key;
        spin_lock(&mru->lock);
-        error = -radix_tree_insert(&mru->store, key, elem);
+        error = radix_tree_insert(&mru->store, key, elem);
        radix_tree_preload_end();
        if (!error)
                _xfs_mru_cache_list_insert(mru, elem);
diff --git a/fs/xfs/xfs_qm.c b/fs/xfs/xfs_qm.c
index 6d26759c779a..10232102b4a6 100644
--- a/fs/xfs/xfs_qm.c
+++ b/fs/xfs/xfs_qm.c
@@ -98,18 +98,18 @@ restart:
                        next_index = be32_to_cpu(dqp->q_core.d_id) + 1;
                        error = execute(batch[i], data);
-                        if (error == EAGAIN) {
+                        if (error == -EAGAIN) {
                                skipped++;
                                continue;
                        }
-                        if (error && last_error != EFSCORRUPTED)
+                        if (error && last_error != -EFSCORRUPTED)
                                last_error = error;
                }
                mutex_unlock(&qi->qi_tree_lock);
                /* bail out if the filesystem is corrupted.  */
-                if (last_error == EFSCORRUPTED) {
+                if (last_error == -EFSCORRUPTED) {
                        skipped = 0;
                        break;
                }
@@ -138,7 +138,7 @@ xfs_qm_dqpurge(
        xfs_dqlock(dqp);
        if ((dqp->dq_flags & XFS_DQ_FREEING) || dqp->q_nrefs != 0) {
                xfs_dqunlock(dqp);
-                return EAGAIN;
+                return -EAGAIN;
        }
        dqp->dq_flags |= XFS_DQ_FREEING;
@@ -221,100 +221,6 @@ xfs_qm_unmount(
        }
 }
-/*
- * This is called from xfs_mountfs to start quotas and initialize all
- * necessary data structures like quotainfo.  This is also responsible for
- * running a quotacheck as necessary.  We are guaranteed that the superblock
- * is consistently read in at this point.
- *
- * If we fail here, the mount will continue with quota turned off. We don't
- * need to inidicate success or failure at all.
- */
-void
-xfs_qm_mount_quotas(
-        xfs_mount_t     *mp)
-{
-        int             error = 0;
-        uint            sbf;
-        /*
-         * If quotas on realtime volumes is not supported, we disable
-         * quotas immediately.
-         */
-        if (mp->m_sb.sb_rextents) {
-                xfs_notice(mp, "Cannot turn on quotas for realtime filesystem");
-                mp->m_qflags = 0;
-                goto write_changes;
-        }
-        ASSERT(XFS_IS_QUOTA_RUNNING(mp));
-        /*
-         * Allocate the quotainfo structure inside the mount struct, and
-         * create quotainode(s), and change/rev superblock if necessary.
-         */
-        error = xfs_qm_init_quotainfo(mp);
-        if (error) {
-                /*
-                 * We must turn off quotas.
-                 */
-                ASSERT(mp->m_quotainfo == NULL);
-                mp->m_qflags = 0;
-                goto write_changes;
-        }
-        /*
-         * If any of the quotas are not consistent, do a quotacheck.
-         */
-        if (XFS_QM_NEED_QUOTACHECK(mp)) {
-                error = xfs_qm_quotacheck(mp);
-                if (error) {
-                        /* Quotacheck failed and disabled quotas. */
-                        return;
-                }
-        }
-        /* 
-         * If one type of quotas is off, then it will lose its
-         * quotachecked status, since we won't be doing accounting for
-         * that type anymore.
-         */
-        if (!XFS_IS_UQUOTA_ON(mp))
-                mp->m_qflags &= ~XFS_UQUOTA_CHKD;
-        if (!XFS_IS_GQUOTA_ON(mp))
-                mp->m_qflags &= ~XFS_GQUOTA_CHKD;
-        if (!XFS_IS_PQUOTA_ON(mp))
-                mp->m_qflags &= ~XFS_PQUOTA_CHKD;
- write_changes:
-        /*
-         * We actually don't have to acquire the m_sb_lock at all.
-         * This can only be called from mount, and that's single threaded. XXX
-         */
-        spin_lock(&mp->m_sb_lock);
-        sbf = mp->m_sb.sb_qflags;
-        mp->m_sb.sb_qflags = mp->m_qflags & XFS_MOUNT_QUOTA_ALL;
-        spin_unlock(&mp->m_sb_lock);
-        if (sbf != (mp->m_qflags & XFS_MOUNT_QUOTA_ALL)) {
-                if (xfs_qm_write_sb_changes(mp, XFS_SB_QFLAGS)) {
-                        /*
-                         * We could only have been turning quotas off.
-                         * We aren't in very good shape actually because
-                         * the incore structures are convinced that quotas are
-                         * off, but the on disk superblock doesn't know that !
-                         */
-                        ASSERT(!(XFS_IS_QUOTA_RUNNING(mp)));
-                        xfs_alert(mp, "%s: Superblock update failed!",
-                                __func__);
-                }
-        }
-        if (error) {
-                xfs_warn(mp, "Failed to initialize disk quotas.");
-                return;
-        }
-}
 /*
 * Called from the vfsops layer.
 */
@@ -671,7 +577,7 @@ xfs_qm_init_quotainfo(
        qinf = mp->m_quotainfo = kmem_zalloc(sizeof(xfs_quotainfo_t), KM_SLEEP);
-        error = -list_lru_init(&qinf->qi_lru);
+        error = list_lru_init(&qinf->qi_lru);
        if (error)
                goto out_free_qinf;
@@ -995,7 +901,7 @@ xfs_qm_dqiter_bufs(
                 * will leave a trace in the log indicating corruption has
                 * been detected.
                 */
-                if (error == EFSCORRUPTED) {
+                if (error == -EFSCORRUPTED) {
                        error = xfs_trans_read_buf(mp, NULL, mp->m_ddev_targp,
                                      XFS_FSB_TO_DADDR(mp, bno),
                                      mp->m_quotainfo->qi_dqchunklen, 0, &bp,
@@ -1005,6 +911,12 @@ xfs_qm_dqiter_bufs(
                if (error)
                        break;
+                /*
+                 * A corrupt buffer might not have a verifier attached, so
+                 * make sure we have the correct one attached before writeback
+                 * occurs.
+                 */
+                bp->b_ops = &xfs_dquot_buf_ops;
                xfs_qm_reset_dqcounts(mp, bp, firstid, type);
                xfs_buf_delwri_queue(bp, buffer_list);
                xfs_buf_relse(bp);
@@ -1090,7 +1002,7 @@ xfs_qm_dqiterate(
                                        xfs_buf_readahead(mp->m_ddev_targp,
                                               XFS_FSB_TO_DADDR(mp, rablkno),
                                               mp->m_quotainfo->qi_dqchunklen,
-                                               NULL);
+                                               &xfs_dquot_buf_ops);
                                        rablkno++;
                                }
                        }
@@ -1138,8 +1050,8 @@ xfs_qm_quotacheck_dqadjust(
                /*
                 * Shouldn't be able to turn off quotas here.
                 */
-                ASSERT(error != ESRCH);
+                ASSERT(error != -ESRCH);
-                ASSERT(error != ENOENT);
+                ASSERT(error != -ENOENT);
                return error;
        }
@@ -1226,7 +1138,7 @@ xfs_qm_dqusage_adjust(
         */
        if (xfs_is_quota_inode(&mp->m_sb, ino)) {
                *res = BULKSTAT_RV_NOTHING;
-                return XFS_ERROR(EINVAL);
+                return -EINVAL;
        }
        /*
@@ -1330,7 +1242,7 @@ out_unlock:
 * Walk thru all the filesystem inodes and construct a consistent view
 * of the disk quota world. If the quotacheck fails, disable quotas.
 */
-int
+STATIC int
 xfs_qm_quotacheck(
        xfs_mount_t     *mp)
 {
@@ -1463,7 +1375,100 @@ xfs_qm_quotacheck(
                }
        } else
                xfs_notice(mp, "Quotacheck: Done.");
-        return (error);
+        return error;
+}
+/*
+ * This is called from xfs_mountfs to start quotas and initialize all
+ * necessary data structures like quotainfo.  This is also responsible for
+ * running a quotacheck as necessary.  We are guaranteed that the superblock
+ * is consistently read in at this point.
+ *
+ * If we fail here, the mount will continue with quota turned off. We don't
+ * need to inidicate success or failure at all.
+ */
+void
+xfs_qm_mount_quotas(
+        struct xfs_mount        *mp)
+{
+        int                     error = 0;
+        uint                    sbf;
+        /*
+         * If quotas on realtime volumes is not supported, we disable
+         * quotas immediately.
+         */
+        if (mp->m_sb.sb_rextents) {
+                xfs_notice(mp, "Cannot turn on quotas for realtime filesystem");
+                mp->m_qflags = 0;
+                goto write_changes;
+        }
+        ASSERT(XFS_IS_QUOTA_RUNNING(mp));
+        /*
+         * Allocate the quotainfo structure inside the mount struct, and
+         * create quotainode(s), and change/rev superblock if necessary.
+         */
+        error = xfs_qm_init_quotainfo(mp);
+        if (error) {
+                /*
+                 * We must turn off quotas.
+                 */
+                ASSERT(mp->m_quotainfo == NULL);
+                mp->m_qflags = 0;
+                goto write_changes;
+        }
+        /*
+         * If any of the quotas are not consistent, do a quotacheck.
+         */
+        if (XFS_QM_NEED_QUOTACHECK(mp)) {
+                error = xfs_qm_quotacheck(mp);
+                if (error) {
+                        /* Quotacheck failed and disabled quotas. */
+                        return;
+                }
+        }
+        /*
+         * If one type of quotas is off, then it will lose its
+         * quotachecked status, since we won't be doing accounting for
+         * that type anymore.
+         */
+        if (!XFS_IS_UQUOTA_ON(mp))
+                mp->m_qflags &= ~XFS_UQUOTA_CHKD;
+        if (!XFS_IS_GQUOTA_ON(mp))
+                mp->m_qflags &= ~XFS_GQUOTA_CHKD;
+        if (!XFS_IS_PQUOTA_ON(mp))
+                mp->m_qflags &= ~XFS_PQUOTA_CHKD;
+ write_changes:
+        /*
+         * We actually don't have to acquire the m_sb_lock at all.
+         * This can only be called from mount, and that's single threaded. XXX
+         */
+        spin_lock(&mp->m_sb_lock);
+        sbf = mp->m_sb.sb_qflags;
+        mp->m_sb.sb_qflags = mp->m_qflags & XFS_MOUNT_QUOTA_ALL;
+        spin_unlock(&mp->m_sb_lock);
+        if (sbf != (mp->m_qflags & XFS_MOUNT_QUOTA_ALL)) {
+                if (xfs_qm_write_sb_changes(mp, XFS_SB_QFLAGS)) {
+                        /*
+                         * We could only have been turning quotas off.
+                         * We aren't in very good shape actually because
+                         * the incore structures are convinced that quotas are
+                         * off, but the on disk superblock doesn't know that !
+                         */
+                        ASSERT(!(XFS_IS_QUOTA_RUNNING(mp)));
+                        xfs_alert(mp, "%s: Superblock update failed!",
+                                __func__);
+                }
+        }
+        if (error) {
+                xfs_warn(mp, "Failed to initialize disk quotas.");
+                return;
+        }
 }
 /*
@@ -1493,7 +1498,7 @@ xfs_qm_init_quotainos(
                        error = xfs_iget(mp, NULL, mp->m_sb.sb_uquotino,
                                             0, 0, &uip);
                        if (error)
-                                return XFS_ERROR(error);
+                                return error;
                }
                if (XFS_IS_GQUOTA_ON(mp) &&
                    mp->m_sb.sb_gquotino != NULLFSINO) {
@@ -1563,7 +1568,7 @@ error_rele:
                IRELE(gip);
        if (pip)
                IRELE(pip);
-        return XFS_ERROR(error);
+        return error;
 }
 STATIC void
@@ -1679,7 +1684,7 @@ xfs_qm_vop_dqalloc(
                                                 XFS_QMOPT_DOWARN,
                                                 &uq);
                        if (error) {
-                                ASSERT(error != ENOENT);
+                                ASSERT(error != -ENOENT);
                                return error;
                        }
                        /*
@@ -1706,7 +1711,7 @@ xfs_qm_vop_dqalloc(
                                                 XFS_QMOPT_DOWARN,
                                                 &gq);
                        if (error) {
-                                ASSERT(error != ENOENT);
+                                ASSERT(error != -ENOENT);
                                goto error_rele;
                        }
                        xfs_dqunlock(gq);
@@ -1726,7 +1731,7 @@ xfs_qm_vop_dqalloc(
                                                 XFS_QMOPT_DOWARN,
                                                 &pq);
                        if (error) {
-                                ASSERT(error != ENOENT);
+                                ASSERT(error != -ENOENT);
                                goto error_rele;
                        }
                        xfs_dqunlock(pq);
@@ -1895,7 +1900,7 @@ xfs_qm_vop_chown_reserve(
                                -((xfs_qcnt_t)delblks), 0, blkflags);
        }
-        return (0);
+        return 0;
 }
 int
diff --git a/fs/xfs/xfs_qm.h b/fs/xfs/xfs_qm.h
index 797fd4636273..3a07a937e232 100644
--- a/fs/xfs/xfs_qm.h
+++ b/fs/xfs/xfs_qm.h
@@ -157,7 +157,6 @@ struct xfs_dquot_acct {
 #define XFS_QM_RTBWARNLIMIT     5
 extern void             xfs_qm_destroy_quotainfo(struct xfs_mount *);
-extern int              xfs_qm_quotacheck(struct xfs_mount *);
 extern int              xfs_qm_write_sb_changes(struct xfs_mount *, __int64_t);
 /* dquot stuff */
diff --git a/fs/xfs/xfs_qm_bhv.c b/fs/xfs/xfs_qm_bhv.c
index e9be63abd8d2..2c61e61b0205 100644
--- a/fs/xfs/xfs_qm_bhv.c
+++ b/fs/xfs/xfs_qm_bhv.c
@@ -117,7 +117,7 @@ xfs_qm_newmount(
                        (uquotaondisk ? " usrquota" : ""),
                        (gquotaondisk ? " grpquota" : ""),
                        (pquotaondisk ? " prjquota" : ""));
-                return XFS_ERROR(EPERM);
+                return -EPERM;
        }
        if (XFS_IS_QUOTA_ON(mp) || quotaondisk) {
diff --git a/fs/xfs/xfs_qm_syscalls.c b/fs/xfs/xfs_qm_syscalls.c
index bbc813caba4c..80f2d77d929a 100644
--- a/fs/xfs/xfs_qm_syscalls.c
+++ b/fs/xfs/xfs_qm_syscalls.c
@@ -64,10 +64,10 @@ xfs_qm_scall_quotaoff(
        /*
         * No file system can have quotas enabled on disk but not in core.
         * Note that quota utilities (like quotaoff) _expect_
-         * errno == EEXIST here.
+         * errno == -EEXIST here.
         */
        if ((mp->m_qflags & flags) == 0)
-                return XFS_ERROR(EEXIST);
+                return -EEXIST;
        error = 0;
        flags &= (XFS_ALL_QUOTA_ACCT | XFS_ALL_QUOTA_ENFD);
@@ -94,7 +94,7 @@ xfs_qm_scall_quotaoff(
                /* XXX what to do if error ? Revert back to old vals incore ? */
                error = xfs_qm_write_sb_changes(mp, XFS_SB_QFLAGS);
-                return (error);
+                return error;
        }
        dqtype = 0;
@@ -198,7 +198,7 @@ xfs_qm_scall_quotaoff(
        if (mp->m_qflags == 0) {
                mutex_unlock(&q->qi_quotaofflock);
                xfs_qm_destroy_quotainfo(mp);
-                return (0);
+                return 0;
        }
        /*
@@ -278,13 +278,13 @@ xfs_qm_scall_trunc_qfiles(
        xfs_mount_t     *mp,
        uint            flags)
 {
-        int             error = EINVAL;
+        int             error = -EINVAL;
        if (!xfs_sb_version_hasquota(&mp->m_sb) || flags == 0 ||
            (flags & ~XFS_DQ_ALLTYPES)) {
                xfs_debug(mp, "%s: flags=%x m_qflags=%x",
                        __func__, flags, mp->m_qflags);
-                return XFS_ERROR(EINVAL);
+                return -EINVAL;
        }
        if (flags & XFS_DQ_USER) {
@@ -328,7 +328,7 @@ xfs_qm_scall_quotaon(
        if (flags == 0) {
                xfs_debug(mp, "%s: zero flags, m_qflags=%x",
                        __func__, mp->m_qflags);
-                return XFS_ERROR(EINVAL);
+                return -EINVAL;
        }
        /* No fs can turn on quotas with a delayed effect */
@@ -351,13 +351,13 @@ xfs_qm_scall_quotaon(
                xfs_debug(mp,
                        "%s: Can't enforce without acct, flags=%x sbflags=%x",
                        __func__, flags, mp->m_sb.sb_qflags);
-                return XFS_ERROR(EINVAL);
+                return -EINVAL;
        }
        /*
         * If everything's up to-date incore, then don't waste time.
         */
        if ((mp->m_qflags & flags) == flags)
-                return XFS_ERROR(EEXIST);
+                return -EEXIST;
        /*
         * Change sb_qflags on disk but not incore mp->qflags
@@ -372,11 +372,11 @@ xfs_qm_scall_quotaon(
         * There's nothing to change if it's the same.
         */
        if ((qf & flags) == flags && sbflags == 0)
-                return XFS_ERROR(EEXIST);
+                return -EEXIST;
        sbflags |= XFS_SB_QFLAGS;
        if ((error = xfs_qm_write_sb_changes(mp, sbflags)))
-                return (error);
+                return error;
        /*
         * If we aren't trying to switch on quota enforcement, we are done.
         */
@@ -387,10 +387,10 @@ xfs_qm_scall_quotaon(
             ((mp->m_sb.sb_qflags & XFS_GQUOTA_ACCT) !=
             (mp->m_qflags & XFS_GQUOTA_ACCT)) ||
            (flags & XFS_ALL_QUOTA_ENFD) == 0)
-                return (0);
+                return 0;
        if (! XFS_IS_QUOTA_RUNNING(mp))
-                return XFS_ERROR(ESRCH);
+                return -ESRCH;
        /*
         * Switch on quota enforcement in core.
@@ -399,7 +399,7 @@ xfs_qm_scall_quotaon(
        mp->m_qflags |= (flags & XFS_ALL_QUOTA_ENFD);
        mutex_unlock(&mp->m_quotainfo->qi_quotaofflock);
-        return (0);
+        return 0;
 }
@@ -426,7 +426,7 @@ xfs_qm_scall_getqstat(
        if (!xfs_sb_version_hasquota(&mp->m_sb)) {
                out->qs_uquota.qfs_ino = NULLFSINO;
                out->qs_gquota.qfs_ino = NULLFSINO;
-                return (0);
+                return 0;
        }
        out->qs_flags = (__uint16_t) xfs_qm_export_flags(mp->m_qflags &
@@ -514,7 +514,7 @@ xfs_qm_scall_getqstatv(
                out->qs_uquota.qfs_ino = NULLFSINO;
                out->qs_gquota.qfs_ino = NULLFSINO;
                out->qs_pquota.qfs_ino = NULLFSINO;
-                return (0);
+                return 0;
        }
        out->qs_flags = (__uint16_t) xfs_qm_export_flags(mp->m_qflags &
@@ -595,7 +595,7 @@ xfs_qm_scall_setqlim(
        xfs_qcnt_t              hard, soft;
        if (newlim->d_fieldmask & ~XFS_DQ_MASK)
-                return EINVAL;
+                return -EINVAL;
        if ((newlim->d_fieldmask & XFS_DQ_MASK) == 0)
                return 0;
@@ -615,7 +615,7 @@ xfs_qm_scall_setqlim(
         */
        error = xfs_qm_dqget(mp, NULL, id, type, XFS_QMOPT_DQALLOC, &dqp);
        if (error) {
-                ASSERT(error != ENOENT);
+                ASSERT(error != -ENOENT);
                goto out_unlock;
        }
        xfs_dqunlock(dqp);
@@ -758,7 +758,7 @@ xfs_qm_log_quotaoff_end(
        error = xfs_trans_reserve(tp, &M_RES(mp)->tr_qm_equotaoff, 0, 0);
        if (error) {
                xfs_trans_cancel(tp, 0);
-                return (error);
+                return error;
        }
        qoffi = xfs_trans_get_qoff_item(tp, startqoff,
@@ -772,7 +772,7 @@ xfs_qm_log_quotaoff_end(
         */
        xfs_trans_set_sync(tp);
        error = xfs_trans_commit(tp, 0);
-        return (error);
+        return error;
 }
@@ -822,7 +822,7 @@ error0:
                spin_unlock(&mp->m_sb_lock);
        }
        *qoffstartp = qoffi;
-        return (error);
+        return error;
 }
@@ -850,7 +850,7 @@ xfs_qm_scall_getquota(
         * our utility programs are concerned.
         */
        if (XFS_IS_DQUOT_UNINITIALIZED(dqp)) {
-                error = XFS_ERROR(ENOENT);
+                error = -ENOENT;
                goto out_put;
        }
@@ -953,7 +953,7 @@ xfs_qm_export_flags(
                uflags |= FS_QUOTA_GDQ_ENFD;
        if (flags & XFS_PQUOTA_ENFD)
                uflags |= FS_QUOTA_PDQ_ENFD;
-        return (uflags);
+        return uflags;
 }
diff --git a/fs/xfs/xfs_quotaops.c b/fs/xfs/xfs_quotaops.c
index 2ad1b9822e92..b238027df987 100644
--- a/fs/xfs/xfs_quotaops.c
+++ b/fs/xfs/xfs_quotaops.c
@@ -51,7 +51,7 @@ xfs_fs_get_xstate(
        if (!XFS_IS_QUOTA_RUNNING(mp))
                return -ENOSYS;
-        return -xfs_qm_scall_getqstat(mp, fqs);
+        return xfs_qm_scall_getqstat(mp, fqs);
 }
 STATIC int
@@ -63,7 +63,7 @@ xfs_fs_get_xstatev(
        if (!XFS_IS_QUOTA_RUNNING(mp))
                return -ENOSYS;
-        return -xfs_qm_scall_getqstatv(mp, fqs);
+        return xfs_qm_scall_getqstatv(mp, fqs);
 }
 STATIC int
@@ -95,11 +95,11 @@ xfs_fs_set_xstate(
        switch (op) {
        case Q_XQUOTAON:
-                return -xfs_qm_scall_quotaon(mp, flags);
+                return xfs_qm_scall_quotaon(mp, flags);
        case Q_XQUOTAOFF:
                if (!XFS_IS_QUOTA_ON(mp))
                        return -EINVAL;
-                return -xfs_qm_scall_quotaoff(mp, flags);
+                return xfs_qm_scall_quotaoff(mp, flags);
        }
        return -EINVAL;
@@ -112,7 +112,7 @@ xfs_fs_rm_xquota(
 {
        struct xfs_mount        *mp = XFS_M(sb);
        unsigned int            flags = 0;
-        
        if (sb->s_flags & MS_RDONLY)
                return -EROFS;
@@ -123,11 +123,11 @@ xfs_fs_rm_xquota(
                flags |= XFS_DQ_USER;
        if (uflags & FS_GROUP_QUOTA)
                flags |= XFS_DQ_GROUP;
-        if (uflags & FS_USER_QUOTA)
+        if (uflags & FS_PROJ_QUOTA)
                flags |= XFS_DQ_PROJ;
-        return -xfs_qm_scall_trunc_qfiles(mp, flags);
+        return xfs_qm_scall_trunc_qfiles(mp, flags);
-}       
+}
 STATIC int
 xfs_fs_get_dqblk(
@@ -142,7 +142,7 @@ xfs_fs_get_dqblk(
        if (!XFS_IS_QUOTA_ON(mp))
                return -ESRCH;
-        return -xfs_qm_scall_getquota(mp, from_kqid(&init_user_ns, qid),
+        return xfs_qm_scall_getquota(mp, from_kqid(&init_user_ns, qid),
                                      xfs_quota_type(qid.type), fdq);
 }
@@ -161,7 +161,7 @@ xfs_fs_set_dqblk(
        if (!XFS_IS_QUOTA_ON(mp))
                return -ESRCH;
-        return -xfs_qm_scall_setqlim(mp, from_kqid(&init_user_ns, qid),
+        return xfs_qm_scall_setqlim(mp, from_kqid(&init_user_ns, qid),
                                     xfs_quota_type(qid.type), fdq);
 }
diff --git a/fs/xfs/xfs_rtalloc.c b/fs/xfs/xfs_rtalloc.c
index ec5ca65c6211..909e143b87ae 100644
--- a/fs/xfs/xfs_rtalloc.c
+++ b/fs/xfs/xfs_rtalloc.c
@@ -863,7 +863,7 @@ xfs_growfs_rt_alloc(
                                        XFS_BMAPI_METADATA, &firstblock,
                                        resblks, &map, &nmap, &flist);
                if (!error && nmap < 1)
-                        error = XFS_ERROR(ENOSPC);
+                        error = -ENOSPC;
                if (error)
                        goto error_cancel;
                /*
@@ -903,7 +903,7 @@ xfs_growfs_rt_alloc(
                        bp = xfs_trans_get_buf(tp, mp->m_ddev_targp, d,
                                mp->m_bsize, 0);
                        if (bp == NULL) {
-                                error = XFS_ERROR(EIO);
+                                error = -EIO;
 error_cancel:
                                xfs_trans_cancel(tp, cancelflags);
                                goto error;
@@ -944,9 +944,9 @@ xfs_growfs_rt(
        xfs_buf_t       *bp;            /* temporary buffer */
        int             error;          /* error return value */
        xfs_mount_t     *nmp;           /* new (fake) mount structure */
-        xfs_drfsbno_t   nrblocks;       /* new number of realtime blocks */
+        xfs_rfsblock_t  nrblocks;       /* new number of realtime blocks */
        xfs_extlen_t    nrbmblocks;     /* new number of rt bitmap blocks */
-        xfs_drtbno_t    nrextents;      /* new number of realtime extents */
+        xfs_rtblock_t   nrextents;      /* new number of realtime extents */
        uint8_t         nrextslog;      /* new log2 of sb_rextents */
        xfs_extlen_t    nrsumblocks;    /* new number of summary blocks */
        uint            nrsumlevels;    /* new rt summary levels */
@@ -962,11 +962,11 @@ xfs_growfs_rt(
         * Initial error checking.
         */
        if (!capable(CAP_SYS_ADMIN))
-                return XFS_ERROR(EPERM);
+                return -EPERM;
        if (mp->m_rtdev_targp == NULL || mp->m_rbmip == NULL ||
            (nrblocks = in->newblocks) <= sbp->sb_rblocks ||
            (sbp->sb_rblocks && (in->extsize != sbp->sb_rextsize)))
-                return XFS_ERROR(EINVAL);
+                return -EINVAL;
        if ((error = xfs_sb_validate_fsb_count(sbp, nrblocks)))
                return error;
        /*
@@ -976,7 +976,7 @@ xfs_growfs_rt(
                                XFS_FSB_TO_BB(mp, nrblocks - 1),
                                XFS_FSB_TO_BB(mp, 1), 0, NULL);
        if (!bp)
-                return EIO;
+                return -EIO;
        if (bp->b_error) {
                error = bp->b_error;
                xfs_buf_relse(bp);
@@ -1001,7 +1001,7 @@ xfs_growfs_rt(
         * since we'll log basically the whole summary file at once.
         */
        if (nrsumblocks > (mp->m_sb.sb_logblocks >> 1))
-                return XFS_ERROR(EINVAL);
+                return -EINVAL;
        /*
         * Get the old block counts for bitmap and summary inodes.
         * These can't change since other growfs callers are locked out.
@@ -1208,7 +1208,7 @@ xfs_rtallocate_extent(
                                len, &sumbp, &sb, prod, &r);
                break;
        default:
-                error = EIO;
+                error = -EIO;
                ASSERT(0);
        }
        if (error)
@@ -1247,7 +1247,7 @@ xfs_rtmount_init(
        if (mp->m_rtdev_targp == NULL) {
                xfs_warn(mp,
        "Filesystem has a realtime volume, use rtdev=device option");
-                return XFS_ERROR(ENODEV);
+                return -ENODEV;
        }
        mp->m_rsumlevels = sbp->sb_rextslog + 1;
        mp->m_rsumsize =
@@ -1263,7 +1263,7 @@ xfs_rtmount_init(
                xfs_warn(mp, "realtime mount -- %llu != %llu",
                        (unsigned long long) XFS_BB_TO_FSB(mp, d),
                        (unsigned long long) mp->m_sb.sb_rblocks);
-                return XFS_ERROR(EFBIG);
+                return -EFBIG;
        }
        bp = xfs_buf_read_uncached(mp->m_rtdev_targp,
                                        d - XFS_FSB_TO_BB(mp, 1),
@@ -1272,7 +1272,7 @@ xfs_rtmount_init(
                xfs_warn(mp, "realtime device size check failed");
                if (bp)
                        xfs_buf_relse(bp);
-                return EIO;
+                return -EIO;
        }
        xfs_buf_relse(bp);
        return 0;
diff --git a/fs/xfs/xfs_rtalloc.h b/fs/xfs/xfs_rtalloc.h
index 752b63d10300..c642795324af 100644
--- a/fs/xfs/xfs_rtalloc.h
+++ b/fs/xfs/xfs_rtalloc.h
@@ -132,7 +132,7 @@ xfs_rtmount_init(
                return 0;
        xfs_warn(mp, "Not built with CONFIG_XFS_RT");
-        return ENOSYS;
+        return -ENOSYS;
 }
 # define xfs_rtmount_inodes(m)  (((mp)->m_sb.sb_rblocks == 0)? 0 : (ENOSYS))
 # define xfs_rtunmount_inodes(m)
diff --git a/fs/xfs/xfs_super.c b/fs/xfs/xfs_super.c
index 8f0333b3f7a0..b194652033cd 100644
--- a/fs/xfs/xfs_super.c
+++ b/fs/xfs/xfs_super.c
@@ -61,6 +61,7 @@
 static const struct super_operations xfs_super_operations;
 static kmem_zone_t *xfs_ioend_zone;
 mempool_t *xfs_ioend_pool;
+struct kset *xfs_kset;
 #define MNTOPT_LOGBUFS  "logbufs"       /* number of XFS log buffers */
 #define MNTOPT_LOGBSIZE "logbsize"      /* size of XFS log buffers */
@@ -185,7 +186,7 @@ xfs_parseargs(
         */
        mp->m_fsname = kstrndup(sb->s_id, MAXNAMELEN, GFP_KERNEL);
        if (!mp->m_fsname)
-                return ENOMEM;
+                return -ENOMEM;
        mp->m_fsname_len = strlen(mp->m_fsname) + 1;
        /*
@@ -204,9 +205,6 @@ xfs_parseargs(
         */
        mp->m_flags |= XFS_MOUNT_BARRIER;
        mp->m_flags |= XFS_MOUNT_COMPAT_IOSIZE;
-#if !XFS_BIG_INUMS
-        mp->m_flags |= XFS_MOUNT_SMALL_INUMS;
-#endif
        /*
         * These can be overridden by the mount option parsing.
@@ -227,57 +225,57 @@ xfs_parseargs(
                        if (!value || !*value) {
                                xfs_warn(mp, "%s option requires an argument",
                                        this_char);
-                                return EINVAL;
+                                return -EINVAL;
                        }
                        if (kstrtoint(value, 10, &mp->m_logbufs))
-                                return EINVAL;
+                                return -EINVAL;
                } else if (!strcmp(this_char, MNTOPT_LOGBSIZE)) {
                        if (!value || !*value) {
                                xfs_warn(mp, "%s option requires an argument",
                                        this_char);
-                                return EINVAL;
+                                return -EINVAL;
                        }
                        if (suffix_kstrtoint(value, 10, &mp->m_logbsize))
-                                return EINVAL;
+                                return -EINVAL;
                } else if (!strcmp(this_char, MNTOPT_LOGDEV)) {
                        if (!value || !*value) {
                                xfs_warn(mp, "%s option requires an argument",
                                        this_char);
-                                return EINVAL;
+                                return -EINVAL;
                        }
                        mp->m_logname = kstrndup(value, MAXNAMELEN, GFP_KERNEL);
                        if (!mp->m_logname)
-                                return ENOMEM;
+                                return -ENOMEM;
                } else if (!strcmp(this_char, MNTOPT_MTPT)) {
                        xfs_warn(mp, "%s option not allowed on this system",
                                this_char);
-                        return EINVAL;
+                        return -EINVAL;
                } else if (!strcmp(this_char, MNTOPT_RTDEV)) {
                        if (!value || !*value) {
                                xfs_warn(mp, "%s option requires an argument",
                                        this_char);
-                                return EINVAL;
+                                return -EINVAL;
                        }
                        mp->m_rtname = kstrndup(value, MAXNAMELEN, GFP_KERNEL);
                        if (!mp->m_rtname)
-                                return ENOMEM;
+                                return -ENOMEM;
                } else if (!strcmp(this_char, MNTOPT_BIOSIZE)) {
                        if (!value || !*value) {
                                xfs_warn(mp, "%s option requires an argument",
                                        this_char);
-                                return EINVAL;
+                                return -EINVAL;
                        }
                        if (kstrtoint(value, 10, &iosize))
-                                return EINVAL;
+                                return -EINVAL;
                        iosizelog = ffs(iosize) - 1;
                } else if (!strcmp(this_char, MNTOPT_ALLOCSIZE)) {
                        if (!value || !*value) {
                                xfs_warn(mp, "%s option requires an argument",
                                        this_char);
-                                return EINVAL;
+                                return -EINVAL;
                        }
                        if (suffix_kstrtoint(value, 10, &iosize))
-                                return EINVAL;
+                                return -EINVAL;
                        iosizelog = ffs(iosize) - 1;
                } else if (!strcmp(this_char, MNTOPT_GRPID) ||
                           !strcmp(this_char, MNTOPT_BSDGROUPS)) {
@@ -297,27 +295,22 @@ xfs_parseargs(
                        if (!value || !*value) {
                                xfs_warn(mp, "%s option requires an argument",
                                        this_char);
-                                return EINVAL;
+                                return -EINVAL;
                        }
                        if (kstrtoint(value, 10, &dsunit))
-                                return EINVAL;
+                                return -EINVAL;
                } else if (!strcmp(this_char, MNTOPT_SWIDTH)) {
                        if (!value || !*value) {
                                xfs_warn(mp, "%s option requires an argument",
                                        this_char);
-                                return EINVAL;
+                                return -EINVAL;
                        }
                        if (kstrtoint(value, 10, &dswidth))
-                                return EINVAL;
+                                return -EINVAL;
                } else if (!strcmp(this_char, MNTOPT_32BITINODE)) {
                        mp->m_flags |= XFS_MOUNT_SMALL_INUMS;
                } else if (!strcmp(this_char, MNTOPT_64BITINODE)) {
                        mp->m_flags &= ~XFS_MOUNT_SMALL_INUMS;
-#if !XFS_BIG_INUMS
-                        xfs_warn(mp, "%s option not allowed on this system",
-                                this_char);
-                        return EINVAL;
-#endif
                } else if (!strcmp(this_char, MNTOPT_NOUUID)) {
                        mp->m_flags |= XFS_MOUNT_NOUUID;
                } else if (!strcmp(this_char, MNTOPT_BARRIER)) {
@@ -390,7 +383,7 @@ xfs_parseargs(
        "irixsgid is now a sysctl(2) variable, option is deprecated.");
                } else {
                        xfs_warn(mp, "unknown mount option [%s].", this_char);
-                        return EINVAL;
+                        return -EINVAL;
                }
        }
@@ -400,32 +393,32 @@ xfs_parseargs(
        if ((mp->m_flags & XFS_MOUNT_NORECOVERY) &&
            !(mp->m_flags & XFS_MOUNT_RDONLY)) {
                xfs_warn(mp, "no-recovery mounts must be read-only.");
-                return EINVAL;
+                return -EINVAL;
        }
        if ((mp->m_flags & XFS_MOUNT_NOALIGN) && (dsunit || dswidth)) {
                xfs_warn(mp,
        "sunit and swidth options incompatible with the noalign option");
-                return EINVAL;
+                return -EINVAL;
        }
 #ifndef CONFIG_XFS_QUOTA
        if (XFS_IS_QUOTA_RUNNING(mp)) {
                xfs_warn(mp, "quota support not available in this kernel.");
-                return EINVAL;
+                return -EINVAL;
        }
 #endif
        if ((dsunit && !dswidth) || (!dsunit && dswidth)) {
                xfs_warn(mp, "sunit and swidth must be specified together");
-                return EINVAL;
+                return -EINVAL;
        }
        if (dsunit && (dswidth % dsunit != 0)) {
                xfs_warn(mp,
        "stripe width (%d) must be a multiple of the stripe unit (%d)",
                        dswidth, dsunit);
-                return EINVAL;
+                return -EINVAL;
        }
 done:
@@ -446,7 +439,7 @@ done:
             mp->m_logbufs > XLOG_MAX_ICLOGS)) {
                xfs_warn(mp, "invalid logbufs value: %d [not %d-%d]",
                        mp->m_logbufs, XLOG_MIN_ICLOGS, XLOG_MAX_ICLOGS);
-                return XFS_ERROR(EINVAL);
+                return -EINVAL;
        }
        if (mp->m_logbsize != -1 &&
            mp->m_logbsize !=  0 &&
@@ -456,7 +449,7 @@ done:
                xfs_warn(mp,
                        "invalid logbufsize: %d [not 16k,32k,64k,128k or 256k]",
                        mp->m_logbsize);
-                return XFS_ERROR(EINVAL);
+                return -EINVAL;
        }
        if (iosizelog) {
@@ -465,7 +458,7 @@ done:
                        xfs_warn(mp, "invalid log iosize: %d [not %d-%d]",
                                iosizelog, XFS_MIN_IO_LOG,
                                XFS_MAX_IO_LOG);
-                        return XFS_ERROR(EINVAL);
+                        return -EINVAL;
                }
                mp->m_flags |= XFS_MOUNT_DFLT_IOSIZE;
@@ -597,15 +590,20 @@ xfs_max_file_offset(
        return (((__uint64_t)pagefactor) << bitshift) - 1;
 }
+/*
+ * xfs_set_inode32() and xfs_set_inode64() are passed an agcount
+ * because in the growfs case, mp->m_sb.sb_agcount is not updated
+ * yet to the potentially higher ag count.
+ */
 xfs_agnumber_t
-xfs_set_inode32(struct xfs_mount *mp)
+xfs_set_inode32(struct xfs_mount *mp, xfs_agnumber_t agcount)
 {
        xfs_agnumber_t  index = 0;
        xfs_agnumber_t  maxagi = 0;
        xfs_sb_t        *sbp = &mp->m_sb;
        xfs_agnumber_t  max_metadata;
-        xfs_agino_t     agino = XFS_OFFBNO_TO_AGINO(mp, sbp->sb_agblocks -1, 0);
+        xfs_agino_t     agino;
-        xfs_ino_t       ino = XFS_AGINO_TO_INO(mp, sbp->sb_agcount -1, agino);
+        xfs_ino_t       ino;
        xfs_perag_t     *pag;
        /* Calculate how much should be reserved for inodes to meet
@@ -620,10 +618,12 @@ xfs_set_inode32(struct xfs_mount *mp)
                do_div(icount, sbp->sb_agblocks);
                max_metadata = icount;
        } else {
-                max_metadata = sbp->sb_agcount;
+                max_metadata = agcount;
        }
-        for (index = 0; index < sbp->sb_agcount; index++) {
+        agino = XFS_OFFBNO_TO_AGINO(mp, sbp->sb_agblocks - 1, 0);
+        for (index = 0; index < agcount; index++) {
                ino = XFS_AGINO_TO_INO(mp, index, agino);
                if (ino > XFS_MAXINUMBER_32) {
@@ -648,11 +648,11 @@ xfs_set_inode32(struct xfs_mount *mp)
 }
 xfs_agnumber_t
-xfs_set_inode64(struct xfs_mount *mp)
+xfs_set_inode64(struct xfs_mount *mp, xfs_agnumber_t agcount)
 {
        xfs_agnumber_t index = 0;
-        for (index = 0; index < mp->m_sb.sb_agcount; index++) {
+        for (index = 0; index < agcount; index++) {
                struct xfs_perag        *pag;
                pag = xfs_perag_get(mp, index);
@@ -686,7 +686,7 @@ xfs_blkdev_get(
                xfs_warn(mp, "Invalid device [%s], error=%d\n", name, error);
        }
-        return -error;
+        return error;
 }
 STATIC void
@@ -756,7 +756,7 @@ xfs_open_devices(
                if (rtdev == ddev || rtdev == logdev) {
                        xfs_warn(mp,
        "Cannot mount filesystem with identical rtdev and ddev/logdev.");
-                        error = EINVAL;
+                        error = -EINVAL;
                        goto out_close_rtdev;
                }
        }
@@ -764,7 +764,7 @@ xfs_open_devices(
        /*
         * Setup xfs_mount buffer target pointers
         */
-        error = ENOMEM;
+        error = -ENOMEM;
        mp->m_ddev_targp = xfs_alloc_buftarg(mp, ddev);
        if (!mp->m_ddev_targp)
                goto out_close_rtdev;
@@ -1188,6 +1188,7 @@ xfs_fs_remount(
        char                    *options)
 {
        struct xfs_mount        *mp = XFS_M(sb);
+        xfs_sb_t                *sbp = &mp->m_sb;
        substring_t             args[MAX_OPT_ARGS];
        char                    *p;
        int                     error;
@@ -1208,10 +1209,10 @@ xfs_fs_remount(
                        mp->m_flags &= ~XFS_MOUNT_BARRIER;
                        break;
                case Opt_inode64:
-                        mp->m_maxagi = xfs_set_inode64(mp);
+                        mp->m_maxagi = xfs_set_inode64(mp, sbp->sb_agcount);
                        break;
                case Opt_inode32:
-                        mp->m_maxagi = xfs_set_inode32(mp);
+                        mp->m_maxagi = xfs_set_inode32(mp, sbp->sb_agcount);
                        break;
                default:
                        /*
@@ -1295,7 +1296,7 @@ xfs_fs_freeze(
        xfs_save_resvblks(mp);
        xfs_quiesce_attr(mp);
-        return -xfs_fs_log_dummy(mp);
+        return xfs_fs_log_dummy(mp);
 }
 STATIC int
@@ -1314,7 +1315,7 @@ xfs_fs_show_options(
        struct seq_file         *m,
        struct dentry           *root)
 {
-        return -xfs_showargs(XFS_M(root->d_sb), m);
+        return xfs_showargs(XFS_M(root->d_sb), m);
 }
 /*
@@ -1336,14 +1337,14 @@ xfs_finish_flags(
                           mp->m_logbsize < mp->m_sb.sb_logsunit) {
                        xfs_warn(mp,
                "logbuf size must be greater than or equal to log stripe size");
-                        return XFS_ERROR(EINVAL);
+                        return -EINVAL;
                }
        } else {
                /* Fail a mount if the logbuf is larger than 32K */
                if (mp->m_logbsize > XLOG_BIG_RECORD_BSIZE) {
                        xfs_warn(mp,
                "logbuf size for version 1 logs must be 16K or 32K");
-                        return XFS_ERROR(EINVAL);
+                        return -EINVAL;
                }
        }
@@ -1355,7 +1356,7 @@ xfs_finish_flags(
                xfs_warn(mp,
 "Cannot mount a V5 filesystem as %s. %s is always enabled for V5 filesystems.",
                        MNTOPT_NOATTR2, MNTOPT_ATTR2);
-                return XFS_ERROR(EINVAL);
+                return -EINVAL;
        }
        /*
@@ -1372,7 +1373,7 @@ xfs_finish_flags(
        if ((mp->m_sb.sb_flags & XFS_SBF_READONLY) && !ronly) {
                xfs_warn(mp,
                        "cannot mount a read-only filesystem as read-write");
-                return XFS_ERROR(EROFS);
+                return -EROFS;
        }
        if ((mp->m_qflags & (XFS_GQUOTA_ACCT | XFS_GQUOTA_ACTIVE)) &&
@@ -1380,7 +1381,7 @@ xfs_finish_flags(
            !xfs_sb_version_has_pquotino(&mp->m_sb)) {
                xfs_warn(mp,
                  "Super block does not support project and group quota together");
-                return XFS_ERROR(EINVAL);
+                return -EINVAL;
        }
        return 0;
@@ -1394,7 +1395,7 @@ xfs_fs_fill_super(
 {
        struct inode            *root;
        struct xfs_mount        *mp = NULL;
-        int                     flags = 0, error = ENOMEM;
+        int                     flags = 0, error = -ENOMEM;
        mp = kzalloc(sizeof(struct xfs_mount), GFP_KERNEL);
        if (!mp)
@@ -1428,11 +1429,11 @@ xfs_fs_fill_super(
        if (error)
                goto out_free_fsname;
-        error = -xfs_init_mount_workqueues(mp);
+        error = xfs_init_mount_workqueues(mp);
        if (error)
                goto out_close_devices;
-        error = -xfs_icsb_init_counters(mp);
+        error = xfs_icsb_init_counters(mp);
        if (error)
                goto out_destroy_workqueues;
@@ -1474,12 +1475,12 @@ xfs_fs_fill_super(
        root = igrab(VFS_I(mp->m_rootip));
        if (!root) {
-                error = ENOENT;
+                error = -ENOENT;
                goto out_unmount;
        }
        sb->s_root = d_make_root(root);
        if (!sb->s_root) {
-                error = ENOMEM;
+                error = -ENOMEM;
                goto out_unmount;
        }
@@ -1499,7 +1500,7 @@ out_destroy_workqueues:
        xfs_free_fsname(mp);
        kfree(mp);
 out:
-        return -error;
+        return error;
 out_unmount:
        xfs_filestream_unmount(mp);
@@ -1761,9 +1762,15 @@ init_xfs_fs(void)
        if (error)
                goto out_cleanup_procfs;
+        xfs_kset = kset_create_and_add("xfs", NULL, fs_kobj);
+        if (!xfs_kset) {
+                error = -ENOMEM;
+                goto out_sysctl_unregister;;
+        }
        error = xfs_qm_init();
        if (error)
-                goto out_sysctl_unregister;
+                goto out_kset_unregister;
        error = register_filesystem(&xfs_fs_type);
        if (error)
@@ -1772,6 +1779,8 @@ init_xfs_fs(void)
 out_qm_exit:
        xfs_qm_exit();
+ out_kset_unregister:
+        kset_unregister(xfs_kset);
 out_sysctl_unregister:
        xfs_sysctl_unregister();
 out_cleanup_procfs:
@@ -1793,6 +1802,7 @@ exit_xfs_fs(void)
 {
        xfs_qm_exit();
        unregister_filesystem(&xfs_fs_type);
+        kset_unregister(xfs_kset);
        xfs_sysctl_unregister();
        xfs_cleanup_procfs();
        xfs_buf_terminate();
diff --git a/fs/xfs/xfs_super.h b/fs/xfs/xfs_super.h
index bbe3d15a7904..2b830c2f322e 100644
--- a/fs/xfs/xfs_super.h
+++ b/fs/xfs/xfs_super.h
@@ -44,16 +44,6 @@ extern void xfs_qm_exit(void);
 # define XFS_REALTIME_STRING
 #endif
-#if XFS_BIG_BLKNOS
-# if XFS_BIG_INUMS
-#  define XFS_BIGFS_STRING      "large block/inode numbers, "
-# else
-#  define XFS_BIGFS_STRING      "large block numbers, "
-# endif
-#else
-# define XFS_BIGFS_STRING
-#endif
 #ifdef DEBUG
 # define XFS_DBG_STRING         "debug"
 #else
@@ -64,7 +54,6 @@ extern void xfs_qm_exit(void);
 #define XFS_BUILD_OPTIONS       XFS_ACL_STRING \
                                XFS_SECURITY_STRING \
                                XFS_REALTIME_STRING \
-                                XFS_BIGFS_STRING \
                                XFS_DBG_STRING /* DBG must be last */
 struct xfs_inode;
@@ -76,8 +65,8 @@ extern __uint64_t xfs_max_file_offset(unsigned int);
 extern void xfs_flush_inodes(struct xfs_mount *mp);
 extern void xfs_blkdev_issue_flush(struct xfs_buftarg *);
-extern xfs_agnumber_t xfs_set_inode32(struct xfs_mount *);
+extern xfs_agnumber_t xfs_set_inode32(struct xfs_mount *, xfs_agnumber_t agcount);
-extern xfs_agnumber_t xfs_set_inode64(struct xfs_mount *);
+extern xfs_agnumber_t xfs_set_inode64(struct xfs_mount *, xfs_agnumber_t agcount);
 extern const struct export_operations xfs_export_operations;
 extern const struct xattr_handler *xfs_xattr_handlers[];
diff --git a/fs/xfs/xfs_symlink.c b/fs/xfs/xfs_symlink.c
index d69363c833e1..6a944a2cd36f 100644
--- a/fs/xfs/xfs_symlink.c
+++ b/fs/xfs/xfs_symlink.c
@@ -76,15 +76,15 @@ xfs_readlink_bmap(
                bp = xfs_buf_read(mp->m_ddev_targp, d, BTOBB(byte_cnt), 0,
                                  &xfs_symlink_buf_ops);
                if (!bp)
-                        return XFS_ERROR(ENOMEM);
+                        return -ENOMEM;
                error = bp->b_error;
                if (error) {
                        xfs_buf_ioerror_alert(bp, __func__);
                        xfs_buf_relse(bp);
                        /* bad CRC means corrupted metadata */
-                        if (error == EFSBADCRC)
+                        if (error == -EFSBADCRC)
-                                error = EFSCORRUPTED;
+                                error = -EFSCORRUPTED;
                        goto out;
                }
                byte_cnt = XFS_SYMLINK_BUF_SPACE(mp, byte_cnt);
@@ -95,7 +95,7 @@ xfs_readlink_bmap(
                if (xfs_sb_version_hascrc(&mp->m_sb)) {
                        if (!xfs_symlink_hdr_ok(ip->i_ino, offset,
                                                        byte_cnt, bp)) {
-                                error = EFSCORRUPTED;
+                                error = -EFSCORRUPTED;
                                xfs_alert(mp,
 "symlink header does not match required off/len/owner (0x%x/Ox%x,0x%llx)",
                                        offset, byte_cnt, ip->i_ino);
@@ -135,7 +135,7 @@ xfs_readlink(
        trace_xfs_readlink(ip);
        if (XFS_FORCED_SHUTDOWN(mp))
-                return XFS_ERROR(EIO);
+                return -EIO;
        xfs_ilock(ip, XFS_ILOCK_SHARED);
@@ -148,7 +148,7 @@ xfs_readlink(
                         __func__, (unsigned long long) ip->i_ino,
                         (long long) pathlen);
                ASSERT(0);
-                error = XFS_ERROR(EFSCORRUPTED);
+                error = -EFSCORRUPTED;
                goto out;
        }
@@ -203,14 +203,14 @@ xfs_symlink(
        trace_xfs_symlink(dp, link_name);
        if (XFS_FORCED_SHUTDOWN(mp))
-                return XFS_ERROR(EIO);
+                return -EIO;
        /*
         * Check component lengths of the target path name.
         */
        pathlen = strlen(target_path);
        if (pathlen >= MAXPATHLEN)      /* total string too long */
-                return XFS_ERROR(ENAMETOOLONG);
+                return -ENAMETOOLONG;
        udqp = gdqp = NULL;
        prid = xfs_get_initial_prid(dp);
@@ -238,7 +238,7 @@ xfs_symlink(
                fs_blocks = xfs_symlink_blocks(mp, pathlen);
        resblks = XFS_SYMLINK_SPACE_RES(mp, link_name->len, fs_blocks);
        error = xfs_trans_reserve(tp, &M_RES(mp)->tr_symlink, resblks, 0);
-        if (error == ENOSPC && fs_blocks == 0) {
+        if (error == -ENOSPC && fs_blocks == 0) {
                resblks = 0;
                error = xfs_trans_reserve(tp, &M_RES(mp)->tr_symlink, 0, 0);
        }
@@ -254,7 +254,7 @@ xfs_symlink(
         * Check whether the directory allows new symlinks or not.
         */
        if (dp->i_d.di_flags & XFS_DIFLAG_NOSYMLINKS) {
-                error = XFS_ERROR(EPERM);
+                error = -EPERM;
                goto error_return;
        }
@@ -284,7 +284,7 @@ xfs_symlink(
        error = xfs_dir_ialloc(&tp, dp, S_IFLNK | (mode & ~S_IFMT), 1, 0,
                               prid, resblks > 0, &ip, NULL);
        if (error) {
-                if (error == ENOSPC)
+                if (error == -ENOSPC)
                        goto error_return;
                goto error1;
        }
@@ -348,7 +348,7 @@ xfs_symlink(
                        bp = xfs_trans_get_buf(tp, mp->m_ddev_targp, d,
                                               BTOBB(byte_cnt), 0);
                        if (!bp) {
-                                error = ENOMEM;
+                                error = -ENOMEM;
                                goto error2;
                        }
                        bp->b_ops = &xfs_symlink_buf_ops;
@@ -489,7 +489,7 @@ xfs_inactive_symlink_rmt(
                        XFS_FSB_TO_DADDR(mp, mval[i].br_startblock),
                        XFS_FSB_TO_BB(mp, mval[i].br_blockcount), 0);
                if (!bp) {
-                        error = ENOMEM;
+                        error = -ENOMEM;
                        goto error_bmap_cancel;
                }
                xfs_trans_binval(tp, bp);
@@ -562,7 +562,7 @@ xfs_inactive_symlink(
        trace_xfs_inactive_symlink(ip);
        if (XFS_FORCED_SHUTDOWN(mp))
-                return XFS_ERROR(EIO);
+                return -EIO;
        xfs_ilock(ip, XFS_ILOCK_EXCL);
@@ -580,7 +580,7 @@ xfs_inactive_symlink(
                         __func__, (unsigned long long)ip->i_ino, pathlen);
                xfs_iunlock(ip, XFS_ILOCK_EXCL);
                ASSERT(0);
-                return XFS_ERROR(EFSCORRUPTED);
+                return -EFSCORRUPTED;
        }
        if (ip->i_df.if_flags & XFS_IFINLINE) {
diff --git a/fs/xfs/xfs_sysfs.c b/fs/xfs/xfs_sysfs.c
new file mode 100644
index 000000000000..9835139ce1ec
--- /dev/null
+++ b/fs/xfs/xfs_sysfs.c
@@ -0,0 +1,165 @@
+/*
+ * Copyright (c) 2014 Red Hat, Inc.
+ * All Rights Reserved.
+ *
+ * This program is free software; you can redistribute it and/or
+ * modify it under the terms of the GNU General Public License as
+ * published by the Free Software Foundation.
+ *
+ * This program is distributed in the hope that it would be useful,
+ * but WITHOUT ANY WARRANTY; without even the implied warranty of
+ * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
+ * GNU General Public License for more details.
+ *
+ * You should have received a copy of the GNU General Public License
+ * along with this program; if not, write the Free Software Foundation,
+ * Inc.,  51 Franklin St, Fifth Floor, Boston, MA  02110-1301  USA
+ */
+#include "xfs.h"
+#include "xfs_sysfs.h"
+#include "xfs_log_format.h"
+#include "xfs_log.h"
+#include "xfs_log_priv.h"
+struct xfs_sysfs_attr {
+        struct attribute attr;
+        ssize_t (*show)(char *buf, void *data);
+        ssize_t (*store)(const char *buf, size_t count, void *data);
+};
+static inline struct xfs_sysfs_attr *
+to_attr(struct attribute *attr)
+{
+        return container_of(attr, struct xfs_sysfs_attr, attr);
+}
+#define XFS_SYSFS_ATTR_RW(name) \
+        static struct xfs_sysfs_attr xfs_sysfs_attr_##name = __ATTR_RW(name)
+#define XFS_SYSFS_ATTR_RO(name) \
+        static struct xfs_sysfs_attr xfs_sysfs_attr_##name = __ATTR_RO(name)
+#define ATTR_LIST(name) &xfs_sysfs_attr_##name.attr
+/*
+ * xfs_mount kobject. This currently has no attributes and thus no need for show
+ * and store helpers. The mp kobject serves as the per-mount parent object that
+ * is identified by the fsname under sysfs.
+ */
+struct kobj_type xfs_mp_ktype = {
+        .release = xfs_sysfs_release,
+};
+/* xlog */
+STATIC ssize_t
+log_head_lsn_show(
+        char    *buf,
+        void    *data)
+{
+        struct xlog *log = data;
+        int cycle;
+        int block;
+        spin_lock(&log->l_icloglock);
+        cycle = log->l_curr_cycle;
+        block = log->l_curr_block;
+        spin_unlock(&log->l_icloglock);
+        return snprintf(buf, PAGE_SIZE, "%d:%d\n", cycle, block);
+}
+XFS_SYSFS_ATTR_RO(log_head_lsn);
+STATIC ssize_t
+log_tail_lsn_show(
+        char    *buf,
+        void    *data)
+{
+        struct xlog *log = data;
+        int cycle;
+        int block;
+        xlog_crack_atomic_lsn(&log->l_tail_lsn, &cycle, &block);
+        return snprintf(buf, PAGE_SIZE, "%d:%d\n", cycle, block);
+}
+XFS_SYSFS_ATTR_RO(log_tail_lsn);
+STATIC ssize_t
+reserve_grant_head_show(
+        char    *buf,
+        void    *data)
+{
+        struct xlog *log = data;
+        int cycle;
+        int bytes;
+        xlog_crack_grant_head(&log->l_reserve_head.grant, &cycle, &bytes);
+        return snprintf(buf, PAGE_SIZE, "%d:%d\n", cycle, bytes);
+}
+XFS_SYSFS_ATTR_RO(reserve_grant_head);
+STATIC ssize_t
+write_grant_head_show(
+        char    *buf,
+        void    *data)
+{
+        struct xlog *log = data;
+        int cycle;
+        int bytes;
+        xlog_crack_grant_head(&log->l_write_head.grant, &cycle, &bytes);
+        return snprintf(buf, PAGE_SIZE, "%d:%d\n", cycle, bytes);
+}
+XFS_SYSFS_ATTR_RO(write_grant_head);
+static struct attribute *xfs_log_attrs[] = {
+        ATTR_LIST(log_head_lsn),
+        ATTR_LIST(log_tail_lsn),
+        ATTR_LIST(reserve_grant_head),
+        ATTR_LIST(write_grant_head),
+        NULL,
+};
+static inline struct xlog *
+to_xlog(struct kobject *kobject)
+{
+        struct xfs_kobj *kobj = to_kobj(kobject);
+        return container_of(kobj, struct xlog, l_kobj);
+}
+STATIC ssize_t
+xfs_log_show(
+        struct kobject          *kobject,
+        struct attribute        *attr,
+        char                    *buf)
+{
+        struct xlog *log = to_xlog(kobject);
+        struct xfs_sysfs_attr *xfs_attr = to_attr(attr);
+        return xfs_attr->show ? xfs_attr->show(buf, log) : 0;
+}
+STATIC ssize_t
+xfs_log_store(
+        struct kobject          *kobject,
+        struct attribute        *attr,
+        const char              *buf,
+        size_t                  count)
+{
+        struct xlog *log = to_xlog(kobject);
+        struct xfs_sysfs_attr *xfs_attr = to_attr(attr);
+        return xfs_attr->store ? xfs_attr->store(buf, count, log) : 0;
+}
+static struct sysfs_ops xfs_log_ops = {
+        .show = xfs_log_show,
+        .store = xfs_log_store,
+};
+struct kobj_type xfs_log_ktype = {
+        .release = xfs_sysfs_release,
+        .sysfs_ops = &xfs_log_ops,
+        .default_attrs = xfs_log_attrs,
+};
diff --git a/fs/xfs/xfs_sysfs.h b/fs/xfs/xfs_sysfs.h
new file mode 100644
index 000000000000..54a2091183c0
--- /dev/null
+++ b/fs/xfs/xfs_sysfs.h
@@ -0,0 +1,59 @@
+/*
+ * Copyright (c) 2014 Red Hat, Inc.
+ * All Rights Reserved.
+ *
+ * This program is free software; you can redistribute it and/or
+ * modify it under the terms of the GNU General Public License as
+ * published by the Free Software Foundation.
+ *
+ * This program is distributed in the hope that it would be useful,
+ * but WITHOUT ANY WARRANTY; without even the implied warranty of
+ * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
+ * GNU General Public License for more details.
+ *
+ * You should have received a copy of the GNU General Public License
+ * along with this program; if not, write the Free Software Foundation,
+ * Inc.,  51 Franklin St, Fifth Floor, Boston, MA  02110-1301  USA
+ */
+#ifndef __XFS_SYSFS_H__
+#define __XFS_SYSFS_H__
+extern struct kobj_type xfs_mp_ktype;   /* xfs_mount */
+extern struct kobj_type xfs_log_ktype;  /* xlog */
+static inline struct xfs_kobj *
+to_kobj(struct kobject *kobject)
+{
+        return container_of(kobject, struct xfs_kobj, kobject);
+}
+static inline void
+xfs_sysfs_release(struct kobject *kobject)
+{
+        struct xfs_kobj *kobj = to_kobj(kobject);
+        complete(&kobj->complete);
+}
+static inline int
+xfs_sysfs_init(
+        struct xfs_kobj         *kobj,
+        struct kobj_type        *ktype,
+        struct xfs_kobj         *parent_kobj,
+        const char              *name)
+{
+        init_completion(&kobj->complete);
+        return kobject_init_and_add(&kobj->kobject, ktype,
+                                    &parent_kobj->kobject, "%s", name);
+}
+static inline void
+xfs_sysfs_del(
+        struct xfs_kobj *kobj)
+{
+        kobject_del(&kobj->kobject);
+        kobject_put(&kobj->kobject);
+        wait_for_completion(&kobj->complete);
+}
+#endif  /* __XFS_SYSFS_H__ */
diff --git a/fs/xfs/xfs_trans.c b/fs/xfs/xfs_trans.c
index d03932564ccb..30e8e3410955 100644
--- a/fs/xfs/xfs_trans.c
+++ b/fs/xfs/xfs_trans.c
@@ -190,7 +190,7 @@ xfs_trans_reserve(
                                          -((int64_t)blocks), rsvd);
                if (error != 0) {
                        current_restore_flags_nested(&tp->t_pflags, PF_FSTRANS);
-                        return (XFS_ERROR(ENOSPC));
+                        return -ENOSPC;
                }
                tp->t_blk_res += blocks;
        }
@@ -241,7 +241,7 @@ xfs_trans_reserve(
                error = xfs_mod_incore_sb(tp->t_mountp, XFS_SBS_FREXTENTS,
                                          -((int64_t)rtextents), rsvd);
                if (error) {
-                        error = XFS_ERROR(ENOSPC);
+                        error = -ENOSPC;
                        goto undo_log;
                }
                tp->t_rtx_res += rtextents;
@@ -874,7 +874,7 @@ xfs_trans_commit(
                goto out_unreserve;
        if (XFS_FORCED_SHUTDOWN(mp)) {
-                error = XFS_ERROR(EIO);
+                error = -EIO;
                goto out_unreserve;
        }
@@ -917,7 +917,7 @@ out_unreserve:
        if (tp->t_ticket) {
                commit_lsn = xfs_log_done(mp, tp->t_ticket, NULL, log_flags);
                if (commit_lsn == -1 && !error)
-                        error = XFS_ERROR(EIO);
+                        error = -EIO;
        }
        current_restore_flags_nested(&tp->t_pflags, PF_FSTRANS);
        xfs_trans_free_items(tp, NULLCOMMITLSN, error ? XFS_TRANS_ABORT : 0);
@@ -1024,7 +1024,7 @@ xfs_trans_roll(
         */
        error = xfs_trans_commit(trans, 0);
        if (error)
-                return (error);
+                return error;
        trans = *tpp;
diff --git a/fs/xfs/xfs_trans_ail.c b/fs/xfs/xfs_trans_ail.c
index cb0f3a84cc68..859482f53b5a 100644
--- a/fs/xfs/xfs_trans_ail.c
+++ b/fs/xfs/xfs_trans_ail.c
@@ -762,7 +762,7 @@ xfs_trans_ail_init(
        ailp = kmem_zalloc(sizeof(struct xfs_ail), KM_MAYFAIL);
        if (!ailp)
-                return ENOMEM;
+                return -ENOMEM;
        ailp->xa_mount = mp;
        INIT_LIST_HEAD(&ailp->xa_ail);
@@ -781,7 +781,7 @@ xfs_trans_ail_init(
 out_free_ailp:
        kmem_free(ailp);
-        return ENOMEM;
+        return -ENOMEM;
 }
 void
diff --git a/fs/xfs/xfs_trans_buf.c b/fs/xfs/xfs_trans_buf.c
index b8eef0549f3f..96c898e7ac9a 100644
--- a/fs/xfs/xfs_trans_buf.c
+++ b/fs/xfs/xfs_trans_buf.c
@@ -166,7 +166,7 @@ xfs_trans_get_buf_map(
                ASSERT(atomic_read(&bip->bli_refcount) > 0);
                bip->bli_recur++;
                trace_xfs_trans_get_buf_recur(bip);
-                return (bp);
+                return bp;
        }
        bp = xfs_buf_get_map(target, map, nmaps, flags);
@@ -178,7 +178,7 @@ xfs_trans_get_buf_map(
        _xfs_trans_bjoin(tp, bp, 1);
        trace_xfs_trans_get_buf(bp->b_fspriv);
-        return (bp);
+        return bp;
 }
 /*
@@ -201,9 +201,8 @@ xfs_trans_getsb(xfs_trans_t	*tp,
         * Default to just trying to lock the superblock buffer
         * if tp is NULL.
         */
-        if (tp == NULL) {
+        if (tp == NULL)
-                return (xfs_getsb(mp, flags));
+                return xfs_getsb(mp, flags);
-        }
        /*
         * If the superblock buffer already has this transaction
@@ -218,7 +217,7 @@ xfs_trans_getsb(xfs_trans_t	*tp,
                ASSERT(atomic_read(&bip->bli_refcount) > 0);
                bip->bli_recur++;
                trace_xfs_trans_getsb_recur(bip);
-                return (bp);
+                return bp;
        }
        bp = xfs_getsb(mp, flags);
@@ -227,7 +226,7 @@ xfs_trans_getsb(xfs_trans_t	*tp,
        _xfs_trans_bjoin(tp, bp, 1);
        trace_xfs_trans_getsb(bp->b_fspriv);
-        return (bp);
+        return bp;
 }
 #ifdef DEBUG
@@ -267,7 +266,7 @@ xfs_trans_read_buf_map(
                bp = xfs_buf_read_map(target, map, nmaps, flags, ops);
                if (!bp)
                        return (flags & XBF_TRYLOCK) ?
-                                        EAGAIN : XFS_ERROR(ENOMEM);
+                                        -EAGAIN : -ENOMEM;
                if (bp->b_error) {
                        error = bp->b_error;
@@ -277,8 +276,8 @@ xfs_trans_read_buf_map(
                        xfs_buf_relse(bp);
                        /* bad CRC means corrupted metadata */
-                        if (error == EFSBADCRC)
+                        if (error == -EFSBADCRC)
-                                error = EFSCORRUPTED;
+                                error = -EFSCORRUPTED;
                        return error;
                }
 #ifdef DEBUG
@@ -287,7 +286,7 @@ xfs_trans_read_buf_map(
                                if (((xfs_req_num++) % xfs_error_mod) == 0) {
                                        xfs_buf_relse(bp);
                                        xfs_debug(mp, "Returning error!");
-                                        return XFS_ERROR(EIO);
+                                        return -EIO;
                                }
                        }
                }
@@ -343,8 +342,8 @@ xfs_trans_read_buf_map(
                                        xfs_force_shutdown(tp->t_mountp,
                                                        SHUTDOWN_META_IO_ERROR);
                                /* bad CRC means corrupted metadata */
-                                if (error == EFSBADCRC)
+                                if (error == -EFSBADCRC)
-                                        error = EFSCORRUPTED;
+                                        error = -EFSCORRUPTED;
                                return error;
                        }
                }
@@ -355,7 +354,7 @@ xfs_trans_read_buf_map(
                if (XFS_FORCED_SHUTDOWN(mp)) {
                        trace_xfs_trans_read_buf_shut(bp, _RET_IP_);
                        *bpp = NULL;
-                        return XFS_ERROR(EIO);
+                        return -EIO;
                }
@@ -372,7 +371,7 @@ xfs_trans_read_buf_map(
        if (bp == NULL) {
                *bpp = NULL;
                return (flags & XBF_TRYLOCK) ?
-                                        0 : XFS_ERROR(ENOMEM);
+                                        0 : -ENOMEM;
        }
        if (bp->b_error) {
                error = bp->b_error;
@@ -384,8 +383,8 @@ xfs_trans_read_buf_map(
                xfs_buf_relse(bp);
                /* bad CRC means corrupted metadata */
-                if (error == EFSBADCRC)
+                if (error == -EFSBADCRC)
-                        error = EFSCORRUPTED;
+                        error = -EFSCORRUPTED;
                return error;
        }
 #ifdef DEBUG
@@ -396,7 +395,7 @@ xfs_trans_read_buf_map(
                                                   SHUTDOWN_META_IO_ERROR);
                                xfs_buf_relse(bp);
                                xfs_debug(mp, "Returning trans error!");
-                                return XFS_ERROR(EIO);
+                                return -EIO;
                        }
                }
        }
@@ -414,7 +413,7 @@ shutdown_abort:
        trace_xfs_trans_read_buf_shut(bp, _RET_IP_);
        xfs_buf_relse(bp);
        *bpp = NULL;
-        return XFS_ERROR(EIO);
+        return -EIO;
 }
 /*
diff --git a/fs/xfs/xfs_trans_dquot.c b/fs/xfs/xfs_trans_dquot.c
index 41172861e857..846e061c2e98 100644
--- a/fs/xfs/xfs_trans_dquot.c
+++ b/fs/xfs/xfs_trans_dquot.c
@@ -722,8 +722,8 @@ xfs_trans_dqresv(
 error_return:
        xfs_dqunlock(dqp);
        if (flags & XFS_QMOPT_ENOSPC)
-                return ENOSPC;
+                return -ENOSPC;
-        return EDQUOT;
+        return -EDQUOT;
 }
diff --git a/fs/xfs/xfs_types.h b/fs/xfs/xfs_types.h
index 65c6e6650b1a..b79dc66b2ecd 100644
--- a/fs/xfs/xfs_types.h
+++ b/fs/xfs/xfs_types.h
@@ -38,43 +38,18 @@ typedef	__int32_t	xfs_tid_t;	/* transaction identifier */
 typedef __uint32_t      xfs_dablk_t;    /* dir/attr block number (in file) */
 typedef __uint32_t      xfs_dahash_t;   /* dir/attr hash value */
-/*
- * These types are 64 bits on disk but are either 32 or 64 bits in memory.
- * Disk based types:
- */
-typedef __uint64_t      xfs_dfsbno_t;   /* blockno in filesystem (agno|agbno) */
-typedef __uint64_t      xfs_drfsbno_t;  /* blockno in filesystem (raw) */
-typedef __uint64_t      xfs_drtbno_t;   /* extent (block) in realtime area */
-typedef __uint64_t      xfs_dfiloff_t;  /* block number in a file */
-typedef __uint64_t      xfs_dfilblks_t; /* number of blocks in a file */
-/*
- * Memory based types are conditional.
- */
-#if XFS_BIG_BLKNOS
 typedef __uint64_t      xfs_fsblock_t;  /* blockno in filesystem (agno|agbno) */
 typedef __uint64_t      xfs_rfsblock_t; /* blockno in filesystem (raw) */
 typedef __uint64_t      xfs_rtblock_t;  /* extent (block) in realtime area */
-typedef __int64_t       xfs_srtblock_t; /* signed version of xfs_rtblock_t */
-#else
-typedef __uint32_t      xfs_fsblock_t;  /* blockno in filesystem (agno|agbno) */
-typedef __uint32_t      xfs_rfsblock_t; /* blockno in filesystem (raw) */
-typedef __uint32_t      xfs_rtblock_t;  /* extent (block) in realtime area */
-typedef __int32_t       xfs_srtblock_t; /* signed version of xfs_rtblock_t */
-#endif
 typedef __uint64_t      xfs_fileoff_t;  /* block number in a file */
-typedef __int64_t       xfs_sfiloff_t;  /* signed block number in a file */
 typedef __uint64_t      xfs_filblks_t;  /* number of blocks in a file */
+typedef __int64_t       xfs_srtblock_t; /* signed version of xfs_rtblock_t */
+typedef __int64_t       xfs_sfiloff_t;  /* signed block number in a file */
 /*
 * Null values for the types.
 */
-#define NULLDFSBNO      ((xfs_dfsbno_t)-1)
-#define NULLDRFSBNO     ((xfs_drfsbno_t)-1)
-#define NULLDRTBNO      ((xfs_drtbno_t)-1)
-#define NULLDFILOFF     ((xfs_dfiloff_t)-1)
 #define NULLFSBLOCK     ((xfs_fsblock_t)-1)
 #define NULLRFSBLOCK    ((xfs_rfsblock_t)-1)
 #define NULLRTBLOCK     ((xfs_rtblock_t)-1)
diff --git a/fs/xfs/xfs_vnode.h b/fs/xfs/xfs_vnode.h
deleted file mode 100644
index e8a77383c0d5..000000000000
--- a/fs/xfs/xfs_vnode.h
+++ /dev/null
@@ -1,46 +0,0 @@
-/*
- * Copyright (c) 2000-2005 Silicon Graphics, Inc.
- * All Rights Reserved.
- *
- * This program is free software; you can redistribute it and/or
- * modify it under the terms of the GNU General Public License as
- * published by the Free Software Foundation.
- *
- * This program is distributed in the hope that it would be useful,
- * but WITHOUT ANY WARRANTY; without even the implied warranty of
- * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the
- * GNU General Public License for more details.
- *
- * You should have received a copy of the GNU General Public License
- * along with this program; if not, write the Free Software Foundation,
- * Inc.,  51 Franklin St, Fifth Floor, Boston, MA  02110-1301  USA
- */
-#ifndef __XFS_VNODE_H__
-#define __XFS_VNODE_H__
-#include "xfs_fs.h"
-struct file;
-struct xfs_inode;
-struct attrlist_cursor_kern;
-/*
- * Flags for read/write calls - same values as IRIX
- */
-#define IO_ISDIRECT     0x00004         /* bypass page cache */
-#define IO_INVIS        0x00020         /* don't update inode timestamps */
-#define XFS_IO_FLAGS \
-        { IO_ISDIRECT,  "DIRECT" }, \
-        { IO_INVIS,     "INVIS"}
-/*
- * Some useful predicates.
- */
-#define VN_MAPPED(vp)   mapping_mapped(vp->i_mapping)
-#define VN_CACHED(vp)   (vp->i_mapping->nrpages)
-#define VN_DIRTY(vp)    mapping_tagged(vp->i_mapping, \
-                                        PAGECACHE_TAG_DIRTY)
-#endif  /* __XFS_VNODE_H__ */
diff --git a/fs/xfs/xfs_xattr.c b/fs/xfs/xfs_xattr.c
index 78ed92a46fdd..93455b998041 100644
--- a/fs/xfs/xfs_xattr.c
+++ b/fs/xfs/xfs_xattr.c
@@ -49,7 +49,7 @@ xfs_xattr_get(struct dentry *dentry, const char *name,
                value = NULL;
        }
-        error = -xfs_attr_get(ip, (unsigned char *)name, value, &asize, xflags);
+        error = xfs_attr_get(ip, (unsigned char *)name, value, &asize, xflags);
        if (error)
                return error;
        return asize;
@@ -71,8 +71,8 @@ xfs_xattr_set(struct dentry *dentry, const char *name, const void *value,
                xflags |= ATTR_REPLACE;
        if (!value)
-                return -xfs_attr_remove(ip, (unsigned char *)name, xflags);
+                return xfs_attr_remove(ip, (unsigned char *)name, xflags);
-        return -xfs_attr_set(ip, (unsigned char *)name,
+        return xfs_attr_set(ip, (unsigned char *)name,
                                (void *)value, size, xflags);
 }