37 files changed, 685 insertions, 210 deletions
diff --git a/kernel/async.c b/kernel/async.c
index 608b32b42812..f565891f2c9b 100644
--- a/kernel/async.c
+++ b/kernel/async.c
@@ -54,6 +54,7 @@ asynchronous and synchronous parts of the kernel.
 #include <linux/sched.h>
 #include <linux/init.h>
 #include <linux/kthread.h>
+#include <linux/delay.h>
 #include <asm/atomic.h>
 static async_cookie_t next_cookie = 1;
@@ -132,21 +133,23 @@ static void run_one_entry(void)
        entry = list_first_entry(&async_pending, struct async_entry, list);
        /* 2) move it to the running queue */
-        list_del(&entry->list);
+        list_move_tail(&entry->list, entry->running);
-        list_add_tail(&entry->list, &async_running);
        spin_unlock_irqrestore(&async_lock, flags);
        /* 3) run it (and print duration)*/
        if (initcall_debug && system_state == SYSTEM_BOOTING) {
-                printk("calling  %lli_%pF @ %i\n", entry->cookie, entry->func, task_pid_nr(current));
+                printk("calling  %lli_%pF @ %i\n", (long long)entry->cookie,
+                        entry->func, task_pid_nr(current));
                calltime = ktime_get();
        }
        entry->func(entry->data, entry->cookie);
        if (initcall_debug && system_state == SYSTEM_BOOTING) {
                rettime = ktime_get();
                delta = ktime_sub(rettime, calltime);
-                printk("initcall %lli_%pF returned 0 after %lld usecs\n", entry->cookie,
+                printk("initcall %lli_%pF returned 0 after %lld usecs\n",
-                        entry->func, ktime_to_ns(delta) >> 10);
+                        (long long)entry->cookie,
+                        entry->func,
+                        (long long)ktime_to_ns(delta) >> 10);
        }
        /* 4) remove it from the running queue */
@@ -205,18 +208,44 @@ static async_cookie_t __async_schedule(async_func_ptr *ptr, void *data, struct l
        return newcookie;
 }
+/**
+ * async_schedule - schedule a function for asynchronous execution
+ * @ptr: function to execute asynchronously
+ * @data: data pointer to pass to the function
+ *
+ * Returns an async_cookie_t that may be used for checkpointing later.
+ * Note: This function may be called from atomic or non-atomic contexts.
+ */
 async_cookie_t async_schedule(async_func_ptr *ptr, void *data)
 {
-        return __async_schedule(ptr, data, &async_pending);
+        return __async_schedule(ptr, data, &async_running);
 }
 EXPORT_SYMBOL_GPL(async_schedule);
-async_cookie_t async_schedule_special(async_func_ptr *ptr, void *data, struct list_head *running)
+/**
+ * async_schedule_domain - schedule a function for asynchronous execution within a certain domain
+ * @ptr: function to execute asynchronously
+ * @data: data pointer to pass to the function
+ * @running: running list for the domain
+ *
+ * Returns an async_cookie_t that may be used for checkpointing later.
+ * @running may be used in the async_synchronize_*_domain() functions
+ * to wait within a certain synchronization domain rather than globally.
+ * A synchronization domain is specified via the running queue @running to use.
+ * Note: This function may be called from atomic or non-atomic contexts.
+ */
+async_cookie_t async_schedule_domain(async_func_ptr *ptr, void *data,
+                                     struct list_head *running)
 {
        return __async_schedule(ptr, data, running);
 }
-EXPORT_SYMBOL_GPL(async_schedule_special);
+EXPORT_SYMBOL_GPL(async_schedule_domain);
+/**
+ * async_synchronize_full - synchronize all asynchronous function calls
+ *
+ * This function waits until all asynchronous function calls have been done.
+ */
 void async_synchronize_full(void)
 {
        do {
@@ -225,13 +254,30 @@ void async_synchronize_full(void)
 }
 EXPORT_SYMBOL_GPL(async_synchronize_full);
-void async_synchronize_full_special(struct list_head *list)
+/**
+ * async_synchronize_full_domain - synchronize all asynchronous function within a certain domain
+ * @list: running list to synchronize on
+ *
+ * This function waits until all asynchronous function calls for the
+ * synchronization domain specified by the running list @list have been done.
+ */
+void async_synchronize_full_domain(struct list_head *list)
 {
-        async_synchronize_cookie_special(next_cookie, list);
+        async_synchronize_cookie_domain(next_cookie, list);
 }
-EXPORT_SYMBOL_GPL(async_synchronize_full_special);
+EXPORT_SYMBOL_GPL(async_synchronize_full_domain);
-void async_synchronize_cookie_special(async_cookie_t cookie, struct list_head *running)
+/**
+ * async_synchronize_cookie_domain - synchronize asynchronous function calls within a certain domain with cookie checkpointing
+ * @cookie: async_cookie_t to use as checkpoint
+ * @running: running list to synchronize on
+ *
+ * This function waits until all asynchronous function calls for the
+ * synchronization domain specified by the running list @list submitted
+ * prior to @cookie have been done.
+ */
+void async_synchronize_cookie_domain(async_cookie_t cookie,
+                                     struct list_head *running)
 {
        ktime_t starttime, delta, endtime;
@@ -247,14 +293,22 @@ void async_synchronize_cookie_special(async_cookie_t cookie, struct list_head *r
                delta = ktime_sub(endtime, starttime);
                printk("async_continuing @ %i after %lli usec\n",
-                        task_pid_nr(current), ktime_to_ns(delta) >> 10);
+                        task_pid_nr(current),
+                        (long long)ktime_to_ns(delta) >> 10);
        }
 }
-EXPORT_SYMBOL_GPL(async_synchronize_cookie_special);
+EXPORT_SYMBOL_GPL(async_synchronize_cookie_domain);
+/**
+ * async_synchronize_cookie - synchronize asynchronous function calls with cookie checkpointing
+ * @cookie: async_cookie_t to use as checkpoint
+ *
+ * This function waits until all asynchronous function calls prior to @cookie
+ * have been done.
+ */
 void async_synchronize_cookie(async_cookie_t cookie)
 {
-        async_synchronize_cookie_special(cookie, &async_running);
+        async_synchronize_cookie_domain(cookie, &async_running);
 }
 EXPORT_SYMBOL_GPL(async_synchronize_cookie);
@@ -315,7 +369,11 @@ static int async_manager_thread(void *unused)
                ec = atomic_read(&entry_count);
                while (tc < ec && tc < MAX_THREADS) {
-                        kthread_run(async_thread, NULL, "async/%i", tc);
+                        if (IS_ERR(kthread_run(async_thread, NULL, "async/%i",
+                                               tc))) {
+                                msleep(100);
+                                continue;
+                        }
                        atomic_inc(&thread_count);
                        tc++;
                }
@@ -330,7 +388,9 @@ static int async_manager_thread(void *unused)
 static int __init async_init(void)
 {
        if (async_enabled)
-                kthread_run(async_manager_thread, NULL, "async/mgr");
+                if (IS_ERR(kthread_run(async_manager_thread, NULL,
+                                       "async/mgr")))
+                        async_enabled = 0;
        return 0;
 }
diff --git a/kernel/cgroup.c b/kernel/cgroup.c
index c29831076e7a..e14db9c089b9 100644
--- a/kernel/cgroup.c
+++ b/kernel/cgroup.c
@@ -1115,8 +1115,10 @@ static void cgroup_kill_sb(struct super_block *sb) {
        }
        write_unlock(&css_set_lock);
-        list_del(&root->root_list);
+        if (!list_empty(&root->root_list)) {
-        root_count--;
+                list_del(&root->root_list);
+                root_count--;
+        }
        mutex_unlock(&cgroup_mutex);
@@ -2349,7 +2351,7 @@ static void cgroup_lock_hierarchy(struct cgroupfs_root *root)
        for (i = 0; i < CGROUP_SUBSYS_COUNT; i++) {
                struct cgroup_subsys *ss = subsys[i];
                if (ss->root == root)
-                        mutex_lock_nested(&ss->hierarchy_mutex, i);
+                        mutex_lock(&ss->hierarchy_mutex);
        }
 }
@@ -2434,7 +2436,9 @@ static long cgroup_create(struct cgroup *parent, struct dentry *dentry,
 err_remove:
+        cgroup_lock_hierarchy(root);
        list_del(&cgrp->sibling);
+        cgroup_unlock_hierarchy(root);
        root->number_of_cgroups--;
 err_destroy:
@@ -2507,7 +2511,7 @@ static int cgroup_clear_css_refs(struct cgroup *cgrp)
        for_each_subsys(cgrp->root, ss) {
                struct cgroup_subsys_state *css = cgrp->subsys[ss->subsys_id];
                int refcnt;
-                do {
+                while (1) {
                        /* We can only remove a CSS with a refcnt==1 */
                        refcnt = atomic_read(&css->refcnt);
                        if (refcnt > 1) {
@@ -2521,7 +2525,10 @@ static int cgroup_clear_css_refs(struct cgroup *cgrp)
                         * css_tryget() to spin until we set the
                         * CSS_REMOVED bits or abort
                         */
-                } while (atomic_cmpxchg(&css->refcnt, refcnt, 0) != refcnt);
+                        if (atomic_cmpxchg(&css->refcnt, refcnt, 0) == refcnt)
+                                break;
+                        cpu_relax();
+                }
        }
 done:
        for_each_subsys(cgrp->root, ss) {
@@ -2630,6 +2637,7 @@ static void __init cgroup_init_subsys(struct cgroup_subsys *ss)
        BUG_ON(!list_empty(&init_task.tasks));
        mutex_init(&ss->hierarchy_mutex);
+        lockdep_set_class(&ss->hierarchy_mutex, &ss->subsys_key);
        ss->active = 1;
 }
@@ -2991,20 +2999,21 @@ int cgroup_clone(struct task_struct *tsk, struct cgroup_subsys *subsys,
                mutex_unlock(&cgroup_mutex);
                return 0;
        }
-        task_lock(tsk);
-        cg = tsk->cgroups;
-        parent = task_cgroup(tsk, subsys->subsys_id);
        /* Pin the hierarchy */
-        if (!atomic_inc_not_zero(&parent->root->sb->s_active)) {
+        if (!atomic_inc_not_zero(&root->sb->s_active)) {
                /* We race with the final deactivate_super() */
                mutex_unlock(&cgroup_mutex);
                return 0;
        }
        /* Keep the cgroup alive */
+        task_lock(tsk);
+        parent = task_cgroup(tsk, subsys->subsys_id);
+        cg = tsk->cgroups;
        get_css_set(cg);
        task_unlock(tsk);
        mutex_unlock(&cgroup_mutex);
        /* Now do the VFS work to create a cgroup */
@@ -3043,7 +3052,7 @@ int cgroup_clone(struct task_struct *tsk, struct cgroup_subsys *subsys,
                mutex_unlock(&inode->i_mutex);
                put_css_set(cg);
-                deactivate_super(parent->root->sb);
+                deactivate_super(root->sb);
                /* The cgroup is still accessible in the VFS, but
                 * we're not going to try to rmdir() it at this
                 * point. */
@@ -3069,7 +3078,7 @@ int cgroup_clone(struct task_struct *tsk, struct cgroup_subsys *subsys,
        mutex_lock(&cgroup_mutex);
        put_css_set(cg);
        mutex_unlock(&cgroup_mutex);
-        deactivate_super(parent->root->sb);
+        deactivate_super(root->sb);
        return ret;
 }
diff --git a/kernel/cpuset.c b/kernel/cpuset.c
index a85678865c5e..f76db9dcaa05 100644
--- a/kernel/cpuset.c
+++ b/kernel/cpuset.c
@@ -61,6 +61,14 @@
 #include <linux/cgroup.h>
 /*
+ * Workqueue for cpuset related tasks.
+ *
+ * Using kevent workqueue may cause deadlock when memory_migrate
+ * is set. So we create a separate workqueue thread for cpuset.
+ */
+static struct workqueue_struct *cpuset_wq;
+/*
 * Tracks how many cpusets are currently defined in system.
 * When there is only one cpuset (the root cpuset) we can
 * short circuit some hooks.
@@ -831,7 +839,7 @@ static DECLARE_WORK(rebuild_sched_domains_work, do_rebuild_sched_domains);
 */
 static void async_rebuild_sched_domains(void)
 {
-        schedule_work(&rebuild_sched_domains_work);
+        queue_work(cpuset_wq, &rebuild_sched_domains_work);
 }
 /*
@@ -2111,6 +2119,9 @@ void __init cpuset_init_smp(void)
        hotcpu_notifier(cpuset_track_online_cpus, 0);
        hotplug_memory_notifier(cpuset_track_online_nodes, 10);
+        cpuset_wq = create_singlethread_workqueue("cpuset");
+        BUG_ON(!cpuset_wq);
 }
 /**
diff --git a/kernel/exit.c b/kernel/exit.c
index f80dec3f1875..167e1e3ad7c6 100644
--- a/kernel/exit.c
+++ b/kernel/exit.c
@@ -118,6 +118,8 @@ static void __exit_signal(struct task_struct *tsk)
                 * We won't ever get here for the group leader, since it
                 * will have been the last reference on the signal_struct.
                 */
+                sig->utime = cputime_add(sig->utime, task_utime(tsk));
+                sig->stime = cputime_add(sig->stime, task_stime(tsk));
                sig->gtime = cputime_add(sig->gtime, task_gtime(tsk));
                sig->min_flt += tsk->min_flt;
                sig->maj_flt += tsk->maj_flt;
@@ -126,6 +128,7 @@ static void __exit_signal(struct task_struct *tsk)
                sig->inblock += task_io_get_inblock(tsk);
                sig->oublock += task_io_get_oublock(tsk);
                task_io_accounting_add(&sig->ioac, &tsk->ioac);
+                sig->sum_sched_runtime += tsk->se.sum_exec_runtime;
                sig = NULL; /* Marker for below. */
        }
@@ -977,12 +980,9 @@ static void check_stack_usage(void)
 {
        static DEFINE_SPINLOCK(low_water_lock);
        static int lowest_to_date = THREAD_SIZE;
-        unsigned long *n = end_of_stack(current);
        unsigned long free;
-        while (*n == 0)
+        free = stack_not_used(current);
-                n++;
-        free = (unsigned long)n - (unsigned long)end_of_stack(current);
        if (free >= lowest_to_date)
                return;
diff --git a/kernel/fork.c b/kernel/fork.c
index 242a706e7721..8de303bdd4e5 100644
--- a/kernel/fork.c
+++ b/kernel/fork.c
@@ -61,6 +61,7 @@
 #include <linux/proc_fs.h>
 #include <linux/blkdev.h>
 #include <trace/sched.h>
+#include <linux/magic.h>
 #include <asm/pgtable.h>
 #include <asm/pgalloc.h>
@@ -212,6 +213,8 @@ static struct task_struct *dup_task_struct(struct task_struct *orig)
 {
        struct task_struct *tsk;
        struct thread_info *ti;
+        unsigned long *stackend;
        int err;
        prepare_to_copy(orig);
@@ -237,6 +240,8 @@ static struct task_struct *dup_task_struct(struct task_struct *orig)
                goto out;
        setup_thread_stack(tsk, orig);
+        stackend = end_of_stack(tsk);
+        *stackend = STACK_END_MAGIC;    /* for overflow detection */
 #ifdef CONFIG_CC_STACKPROTECTOR
        tsk->stack_canary = get_random_int();
@@ -851,13 +856,14 @@ static int copy_signal(unsigned long clone_flags, struct task_struct *tsk)
        sig->tty_old_pgrp = NULL;
        sig->tty = NULL;
-        sig->cutime = sig->cstime = cputime_zero;
+        sig->utime = sig->stime = sig->cutime = sig->cstime = cputime_zero;
        sig->gtime = cputime_zero;
        sig->cgtime = cputime_zero;
        sig->nvcsw = sig->nivcsw = sig->cnvcsw = sig->cnivcsw = 0;
        sig->min_flt = sig->maj_flt = sig->cmin_flt = sig->cmaj_flt = 0;
        sig->inblock = sig->oublock = sig->cinblock = sig->coublock = 0;
        task_io_accounting_init(&sig->ioac);
+        sig->sum_sched_runtime = 0;
        taskstats_tgid_init(sig);
        task_lock(current->group_leader);
@@ -1005,6 +1011,7 @@ static struct task_struct *copy_process(unsigned long clone_flags,
         * triggers too late. This doesn't hurt, the check is only there
         * to stop root fork bombs.
         */
+        retval = -EAGAIN;
        if (nr_threads >= max_threads)
                goto bad_fork_cleanup_count;
@@ -1093,7 +1100,7 @@ static struct task_struct *copy_process(unsigned long clone_flags,
 #ifdef CONFIG_DEBUG_MUTEXES
        p->blocked_on = NULL; /* not blocked yet */
 #endif
-        if (unlikely(ptrace_reparented(current)))
+        if (unlikely(current->ptrace))
                ptrace_fork(p, clone_flags);
        /* Perform scheduler related setup. Assign this task to a CPU. */
diff --git a/kernel/hrtimer.c b/kernel/hrtimer.c
index f33afb0407bc..f394d2a42ca3 100644
--- a/kernel/hrtimer.c
+++ b/kernel/hrtimer.c
@@ -501,6 +501,13 @@ static void hrtimer_force_reprogram(struct hrtimer_cpu_base *cpu_base)
                        continue;
                timer = rb_entry(base->first, struct hrtimer, node);
                expires = ktime_sub(hrtimer_get_expires(timer), base->offset);
+                /*
+                 * clock_was_set() has changed base->offset so the
+                 * result might be negative. Fix it up to prevent a
+                 * false positive in clockevents_program_event()
+                 */
+                if (expires.tv64 < 0)
+                        expires.tv64 = 0;
                if (expires.tv64 < cpu_base->expires_next.tv64)
                        cpu_base->expires_next = expires;
        }
@@ -1158,6 +1165,29 @@ static void __run_hrtimer(struct hrtimer *timer)
 #ifdef CONFIG_HIGH_RES_TIMERS
+static int force_clock_reprogram;
+/*
+ * After 5 iteration's attempts, we consider that hrtimer_interrupt()
+ * is hanging, which could happen with something that slows the interrupt
+ * such as the tracing. Then we force the clock reprogramming for each future
+ * hrtimer interrupts to avoid infinite loops and use the min_delta_ns
+ * threshold that we will overwrite.
+ * The next tick event will be scheduled to 3 times we currently spend on
+ * hrtimer_interrupt(). This gives a good compromise, the cpus will spend
+ * 1/4 of their time to process the hrtimer interrupts. This is enough to
+ * let it running without serious starvation.
+ */
+static inline void
+hrtimer_interrupt_hanging(struct clock_event_device *dev,
+                        ktime_t try_time)
+{
+        force_clock_reprogram = 1;
+        dev->min_delta_ns = (unsigned long)try_time.tv64 * 3;
+        printk(KERN_WARNING "hrtimer: interrupt too slow, "
+                "forcing clock min delta to %lu ns\n", dev->min_delta_ns);
+}
 /*
 * High resolution timer interrupt
 * Called with interrupts disabled
@@ -1167,6 +1197,7 @@ void hrtimer_interrupt(struct clock_event_device *dev)
        struct hrtimer_cpu_base *cpu_base = &__get_cpu_var(hrtimer_bases);
        struct hrtimer_clock_base *base;
        ktime_t expires_next, now;
+        int nr_retries = 0;
        int i;
        BUG_ON(!cpu_base->hres_active);
@@ -1174,6 +1205,10 @@ void hrtimer_interrupt(struct clock_event_device *dev)
        dev->next_event.tv64 = KTIME_MAX;
 retry:
+        /* 5 retries is enough to notice a hang */
+        if (!(++nr_retries % 5))
+                hrtimer_interrupt_hanging(dev, ktime_sub(ktime_get(), now));
        now = ktime_get();
        expires_next.tv64 = KTIME_MAX;
@@ -1226,7 +1261,7 @@ void hrtimer_interrupt(struct clock_event_device *dev)
        /* Reprogramming necessary ? */
        if (expires_next.tv64 != KTIME_MAX) {
-                if (tick_program_event(expires_next, 0))
+                if (tick_program_event(expires_next, force_clock_reprogram))
                        goto retry;
        }
 }
@@ -1580,6 +1615,10 @@ static int __cpuinit hrtimer_cpu_notify(struct notifier_block *self,
                break;
 #ifdef CONFIG_HOTPLUG_CPU
+        case CPU_DYING:
+        case CPU_DYING_FROZEN:
+                clockevents_notify(CLOCK_EVT_NOTIFY_CPU_DYING, &scpu);
+                break;
        case CPU_DEAD:
        case CPU_DEAD_FROZEN:
        {
diff --git a/kernel/irq/chip.c b/kernel/irq/chip.c
index f63c706d25e1..122fef4b0bd3 100644
--- a/kernel/irq/chip.c
+++ b/kernel/irq/chip.c
@@ -46,7 +46,10 @@ void dynamic_irq_init(unsigned int irq)
        desc->irq_count = 0;
        desc->irqs_unhandled = 0;
 #ifdef CONFIG_SMP
-        cpumask_setall(&desc->affinity);
+        cpumask_setall(desc->affinity);
+#ifdef CONFIG_GENERIC_PENDING_IRQ
+        cpumask_clear(desc->pending_mask);
+#endif
 #endif
        spin_unlock_irqrestore(&desc->lock, flags);
 }
@@ -383,6 +386,7 @@ handle_level_irq(unsigned int irq, struct irq_desc *desc)
 out_unlock:
        spin_unlock(&desc->lock);
 }
+EXPORT_SYMBOL_GPL(handle_level_irq);
 /**
 *      handle_fasteoi_irq - irq handler for transparent controllers
@@ -593,6 +597,7 @@ __set_irq_handler(unsigned int irq, irq_flow_handler_t handle, int is_chained,
        }
        spin_unlock_irqrestore(&desc->lock, flags);
 }
+EXPORT_SYMBOL_GPL(__set_irq_handler);
 void
 set_irq_chip_and_handler(unsigned int irq, struct irq_chip *chip,
diff --git a/kernel/irq/handle.c b/kernel/irq/handle.c
index c20db0be9173..f51eaee921b6 100644
--- a/kernel/irq/handle.c
+++ b/kernel/irq/handle.c
@@ -17,6 +17,7 @@
 #include <linux/kernel_stat.h>
 #include <linux/rculist.h>
 #include <linux/hash.h>
+#include <linux/bootmem.h>
 #include "internals.h"
@@ -39,6 +40,18 @@ void handle_bad_irq(unsigned int irq, struct irq_desc *desc)
        ack_bad_irq(irq);
 }
+#if defined(CONFIG_SMP) && defined(CONFIG_GENERIC_HARDIRQS)
+static void __init init_irq_default_affinity(void)
+{
+        alloc_bootmem_cpumask_var(&irq_default_affinity);
+        cpumask_setall(irq_default_affinity);
+}
+#else
+static void __init init_irq_default_affinity(void)
+{
+}
+#endif
 /*
 * Linux has a controller-independent interrupt architecture.
 * Every controller has a 'controller-template', that is used
@@ -57,6 +70,7 @@ int nr_irqs = NR_IRQS;
 EXPORT_SYMBOL_GPL(nr_irqs);
 #ifdef CONFIG_SPARSE_IRQ
 static struct irq_desc irq_desc_init = {
        .irq        = -1,
        .status     = IRQ_DISABLED,
@@ -64,9 +78,6 @@ static struct irq_desc irq_desc_init = {
        .handle_irq = handle_bad_irq,
        .depth      = 1,
        .lock       = __SPIN_LOCK_UNLOCKED(irq_desc_init.lock),
-#ifdef CONFIG_SMP
-        .affinity   = CPU_MASK_ALL
-#endif
 };
 void init_kstat_irqs(struct irq_desc *desc, int cpu, int nr)
@@ -101,6 +112,10 @@ static void init_one_irq_desc(int irq, struct irq_desc *desc, int cpu)
                printk(KERN_ERR "can not alloc kstat_irqs\n");
                BUG_ON(1);
        }
+        if (!init_alloc_desc_masks(desc, cpu, false)) {
+                printk(KERN_ERR "can not alloc irq_desc cpumasks\n");
+                BUG_ON(1);
+        }
        arch_init_chip_data(desc, cpu);
 }
@@ -109,7 +124,7 @@ static void init_one_irq_desc(int irq, struct irq_desc *desc, int cpu)
 */
 DEFINE_SPINLOCK(sparse_irq_lock);
-struct irq_desc *irq_desc_ptrs[NR_IRQS] __read_mostly;
+struct irq_desc **irq_desc_ptrs __read_mostly;
 static struct irq_desc irq_desc_legacy[NR_IRQS_LEGACY] __cacheline_aligned_in_smp = {
        [0 ... NR_IRQS_LEGACY-1] = {
@@ -119,14 +134,10 @@ static struct irq_desc irq_desc_legacy[NR_IRQS_LEGACY] __cacheline_aligned_in_sm
                .handle_irq = handle_bad_irq,
                .depth      = 1,
                .lock       = __SPIN_LOCK_UNLOCKED(irq_desc_init.lock),
-#ifdef CONFIG_SMP
-                .affinity   = CPU_MASK_ALL
-#endif
        }
 };
-/* FIXME: use bootmem alloc ...*/
+static unsigned int *kstat_irqs_legacy;
-static unsigned int kstat_irqs_legacy[NR_IRQS_LEGACY][NR_CPUS];
 int __init early_irq_init(void)
 {
@@ -134,18 +145,32 @@ int __init early_irq_init(void)
        int legacy_count;
        int i;
+        init_irq_default_affinity();
+         /* initialize nr_irqs based on nr_cpu_ids */
+        arch_probe_nr_irqs();
+        printk(KERN_INFO "NR_IRQS:%d nr_irqs:%d\n", NR_IRQS, nr_irqs);
        desc = irq_desc_legacy;
        legacy_count = ARRAY_SIZE(irq_desc_legacy);
+        /* allocate irq_desc_ptrs array based on nr_irqs */
+        irq_desc_ptrs = alloc_bootmem(nr_irqs * sizeof(void *));
+        /* allocate based on nr_cpu_ids */
+        /* FIXME: invert kstat_irgs, and it'd be a per_cpu_alloc'd thing */
+        kstat_irqs_legacy = alloc_bootmem(NR_IRQS_LEGACY * nr_cpu_ids *
+                                          sizeof(int));
        for (i = 0; i < legacy_count; i++) {
                desc[i].irq = i;
-                desc[i].kstat_irqs = kstat_irqs_legacy[i];
+                desc[i].kstat_irqs = kstat_irqs_legacy + i * nr_cpu_ids;
                lockdep_set_class(&desc[i].lock, &irq_desc_lock_class);
+                init_alloc_desc_masks(&desc[i], 0, true);
                irq_desc_ptrs[i] = desc + i;
        }
-        for (i = legacy_count; i < NR_IRQS; i++)
+        for (i = legacy_count; i < nr_irqs; i++)
                irq_desc_ptrs[i] = NULL;
        return arch_early_irq_init();
@@ -153,7 +178,10 @@ int __init early_irq_init(void)
 struct irq_desc *irq_to_desc(unsigned int irq)
 {
-        return (irq < NR_IRQS) ? irq_desc_ptrs[irq] : NULL;
+        if (irq_desc_ptrs && irq < nr_irqs)
+                return irq_desc_ptrs[irq];
+        return NULL;
 }
 struct irq_desc *irq_to_desc_alloc_cpu(unsigned int irq, int cpu)
@@ -162,10 +190,9 @@ struct irq_desc *irq_to_desc_alloc_cpu(unsigned int irq, int cpu)
        unsigned long flags;
        int node;
-        if (irq >= NR_IRQS) {
+        if (irq >= nr_irqs) {
-                printk(KERN_WARNING "irq >= NR_IRQS in irq_to_desc_alloc: %d %d\n",
+                WARN(1, "irq (%d) >= nr_irqs (%d) in irq_to_desc_alloc\n",
-                                irq, NR_IRQS);
+                        irq, nr_irqs);
-                WARN_ON(1);
                return NULL;
        }
@@ -207,9 +234,6 @@ struct irq_desc irq_desc[NR_IRQS] __cacheline_aligned_in_smp = {
                .handle_irq = handle_bad_irq,
                .depth = 1,
                .lock = __SPIN_LOCK_UNLOCKED(irq_desc->lock),
-#ifdef CONFIG_SMP
-                .affinity = CPU_MASK_ALL
-#endif
        }
 };
@@ -219,12 +243,17 @@ int __init early_irq_init(void)
        int count;
        int i;
+        init_irq_default_affinity();
+        printk(KERN_INFO "NR_IRQS:%d\n", NR_IRQS);
        desc = irq_desc;
        count = ARRAY_SIZE(irq_desc);
-        for (i = 0; i < count; i++)
+        for (i = 0; i < count; i++) {
                desc[i].irq = i;
+                init_alloc_desc_masks(&desc[i], 0, true);
+        }
        return arch_early_irq_init();
 }
diff --git a/kernel/irq/internals.h b/kernel/irq/internals.h
index e6d0a43cc125..40416a81a0f5 100644
--- a/kernel/irq/internals.h
+++ b/kernel/irq/internals.h
@@ -16,7 +16,14 @@ extern int __irq_set_trigger(struct irq_desc *desc, unsigned int irq,
 extern struct lock_class_key irq_desc_lock_class;
 extern void init_kstat_irqs(struct irq_desc *desc, int cpu, int nr);
 extern spinlock_t sparse_irq_lock;
+#ifdef CONFIG_SPARSE_IRQ
+/* irq_desc_ptrs allocated at boot time */
+extern struct irq_desc **irq_desc_ptrs;
+#else
+/* irq_desc_ptrs is a fixed size array */
 extern struct irq_desc *irq_desc_ptrs[NR_IRQS];
+#endif
 #ifdef CONFIG_PROC_FS
 extern void register_irq_proc(unsigned int irq, struct irq_desc *desc);
diff --git a/kernel/irq/manage.c b/kernel/irq/manage.c
index cd0cd8dcb345..a3a5dc9ef346 100644
--- a/kernel/irq/manage.c
+++ b/kernel/irq/manage.c
@@ -15,17 +15,9 @@
 #include "internals.h"
-#ifdef CONFIG_SMP
+#if defined(CONFIG_SMP) && defined(CONFIG_GENERIC_HARDIRQS)
 cpumask_var_t irq_default_affinity;
-static int init_irq_default_affinity(void)
-{
-        alloc_cpumask_var(&irq_default_affinity, GFP_KERNEL);
-        cpumask_setall(irq_default_affinity);
-        return 0;
-}
-core_initcall(init_irq_default_affinity);
 /**
 *      synchronize_irq - wait for pending IRQ handlers (on other CPUs)
 *      @irq: interrupt number to wait for
@@ -98,14 +90,14 @@ int irq_set_affinity(unsigned int irq, const struct cpumask *cpumask)
 #ifdef CONFIG_GENERIC_PENDING_IRQ
        if (desc->status & IRQ_MOVE_PCNTXT || desc->status & IRQ_DISABLED) {
-                cpumask_copy(&desc->affinity, cpumask);
+                cpumask_copy(desc->affinity, cpumask);
                desc->chip->set_affinity(irq, cpumask);
        } else {
                desc->status |= IRQ_MOVE_PENDING;
-                cpumask_copy(&desc->pending_mask, cpumask);
+                cpumask_copy(desc->pending_mask, cpumask);
        }
 #else
-        cpumask_copy(&desc->affinity, cpumask);
+        cpumask_copy(desc->affinity, cpumask);
        desc->chip->set_affinity(irq, cpumask);
 #endif
        desc->status |= IRQ_AFFINITY_SET;
@@ -127,16 +119,16 @@ int do_irq_select_affinity(unsigned int irq, struct irq_desc *desc)
         * one of the targets is online.
         */
        if (desc->status & (IRQ_AFFINITY_SET | IRQ_NO_BALANCING)) {
-                if (cpumask_any_and(&desc->affinity, cpu_online_mask)
+                if (cpumask_any_and(desc->affinity, cpu_online_mask)
                    < nr_cpu_ids)
                        goto set_affinity;
                else
                        desc->status &= ~IRQ_AFFINITY_SET;
        }
-        cpumask_and(&desc->affinity, cpu_online_mask, irq_default_affinity);
+        cpumask_and(desc->affinity, cpu_online_mask, irq_default_affinity);
 set_affinity:
-        desc->chip->set_affinity(irq, &desc->affinity);
+        desc->chip->set_affinity(irq, desc->affinity);
        return 0;
 }
diff --git a/kernel/irq/migration.c b/kernel/irq/migration.c
index bd72329e630c..e05ad9be43b7 100644
--- a/kernel/irq/migration.c
+++ b/kernel/irq/migration.c
@@ -18,7 +18,7 @@ void move_masked_irq(int irq)
        desc->status &= ~IRQ_MOVE_PENDING;
-        if (unlikely(cpumask_empty(&desc->pending_mask)))
+        if (unlikely(cpumask_empty(desc->pending_mask)))
                return;
        if (!desc->chip->set_affinity)
@@ -38,13 +38,13 @@ void move_masked_irq(int irq)
         * For correct operation this depends on the caller
         * masking the irqs.
         */
-        if (likely(cpumask_any_and(&desc->pending_mask, cpu_online_mask)
+        if (likely(cpumask_any_and(desc->pending_mask, cpu_online_mask)
                   < nr_cpu_ids)) {
-                cpumask_and(&desc->affinity,
+                cpumask_and(desc->affinity,
-                            &desc->pending_mask, cpu_online_mask);
+                            desc->pending_mask, cpu_online_mask);
-                desc->chip->set_affinity(irq, &desc->affinity);
+                desc->chip->set_affinity(irq, desc->affinity);
        }
-        cpumask_clear(&desc->pending_mask);
+        cpumask_clear(desc->pending_mask);
 }
 void move_native_irq(int irq)
diff --git a/kernel/irq/numa_migrate.c b/kernel/irq/numa_migrate.c
index ecf765c6a77a..7f9b80434e32 100644
--- a/kernel/irq/numa_migrate.c
+++ b/kernel/irq/numa_migrate.c
@@ -38,15 +38,22 @@ static void free_kstat_irqs(struct irq_desc *old_desc, struct irq_desc *desc)
        old_desc->kstat_irqs = NULL;
 }
-static void init_copy_one_irq_desc(int irq, struct irq_desc *old_desc,
+static bool init_copy_one_irq_desc(int irq, struct irq_desc *old_desc,
                 struct irq_desc *desc, int cpu)
 {
        memcpy(desc, old_desc, sizeof(struct irq_desc));
+        if (!init_alloc_desc_masks(desc, cpu, false)) {
+                printk(KERN_ERR "irq %d: can not get new irq_desc cpumask "
+                                "for migration.\n", irq);
+                return false;
+        }
        spin_lock_init(&desc->lock);
        desc->cpu = cpu;
        lockdep_set_class(&desc->lock, &irq_desc_lock_class);
        init_copy_kstat_irqs(old_desc, desc, cpu, nr_cpu_ids);
+        init_copy_desc_masks(old_desc, desc);
        arch_init_copy_chip_data(old_desc, desc, cpu);
+        return true;
 }
 static void free_one_irq_desc(struct irq_desc *old_desc, struct irq_desc *desc)
@@ -71,23 +78,34 @@ static struct irq_desc *__real_move_irq_desc(struct irq_desc *old_desc,
        desc = irq_desc_ptrs[irq];
        if (desc && old_desc != desc)
-                        goto out_unlock;
+                goto out_unlock;
        node = cpu_to_node(cpu);
        desc = kzalloc_node(sizeof(*desc), GFP_ATOMIC, node);
        if (!desc) {
-                printk(KERN_ERR "irq %d: can not get new irq_desc for migration.\n", irq);
+                printk(KERN_ERR "irq %d: can not get new irq_desc "
+                                "for migration.\n", irq);
                /* still use old one */
                desc = old_desc;
                goto out_unlock;
        }
-        init_copy_one_irq_desc(irq, old_desc, desc, cpu);
+        if (!init_copy_one_irq_desc(irq, old_desc, desc, cpu)) {
+                /* still use old one */
+                kfree(desc);
+                desc = old_desc;
+                goto out_unlock;
+        }
        irq_desc_ptrs[irq] = desc;
+        spin_unlock_irqrestore(&sparse_irq_lock, flags);
        /* free the old one */
        free_one_irq_desc(old_desc, desc);
+        spin_unlock(&old_desc->lock);
        kfree(old_desc);
+        spin_lock(&desc->lock);
+        return desc;
 out_unlock:
        spin_unlock_irqrestore(&sparse_irq_lock, flags);
diff --git a/kernel/irq/proc.c b/kernel/irq/proc.c
index aae3f742bcec..692363dd591f 100644
--- a/kernel/irq/proc.c
+++ b/kernel/irq/proc.c
@@ -20,11 +20,11 @@ static struct proc_dir_entry *root_irq_dir;
 static int irq_affinity_proc_show(struct seq_file *m, void *v)
 {
        struct irq_desc *desc = irq_to_desc((long)m->private);
-        const struct cpumask *mask = &desc->affinity;
+        const struct cpumask *mask = desc->affinity;
 #ifdef CONFIG_GENERIC_PENDING_IRQ
        if (desc->status & IRQ_MOVE_PENDING)
-                mask = &desc->pending_mask;
+                mask = desc->pending_mask;
 #endif
        seq_cpumask(m, mask);
        seq_putc(m, '\n');
diff --git a/kernel/itimer.c b/kernel/itimer.c
index 6a5fe93dd8bd..58762f7077ec 100644
--- a/kernel/itimer.c
+++ b/kernel/itimer.c
@@ -62,7 +62,7 @@ int do_getitimer(int which, struct itimerval *value)
                        struct task_cputime cputime;
                        cputime_t utime;
-                        thread_group_cputime(tsk, &cputime);
+                        thread_group_cputimer(tsk, &cputime);
                        utime = cputime.utime;
                        if (cputime_le(cval, utime)) { /* about to fire */
                                cval = jiffies_to_cputime(1);
@@ -82,7 +82,7 @@ int do_getitimer(int which, struct itimerval *value)
                        struct task_cputime times;
                        cputime_t ptime;
-                        thread_group_cputime(tsk, &times);
+                        thread_group_cputimer(tsk, &times);
                        ptime = cputime_add(times.utime, times.stime);
                        if (cputime_le(cval, ptime)) { /* about to fire */
                                cval = jiffies_to_cputime(1);
diff --git a/kernel/kexec.c b/kernel/kexec.c
index 8a6d7b08864e..795e7b67a228 100644
--- a/kernel/kexec.c
+++ b/kernel/kexec.c
@@ -1130,7 +1130,7 @@ void crash_save_cpu(struct pt_regs *regs, int cpu)
                return;
        memset(&prstatus, 0, sizeof(prstatus));
        prstatus.pr_pid = current->pid;
-        elf_core_copy_regs(&prstatus.pr_reg, regs);
+        elf_core_copy_kernel_regs(&prstatus.pr_reg, regs);
        buf = append_elf_note(buf, KEXEC_CORE_NOTE_NAME, NT_PRSTATUS,
                              &prstatus, sizeof(prstatus));
        final_note(buf);
diff --git a/kernel/module.c b/kernel/module.c
index e8b51d41dd72..ba22484a987e 100644
--- a/kernel/module.c
+++ b/kernel/module.c
@@ -573,13 +573,13 @@ static char last_unloaded_module[MODULE_NAME_LEN+1];
 /* Init the unload section of the module. */
 static void module_unload_init(struct module *mod)
 {
-        unsigned int i;
+        int cpu;
        INIT_LIST_HEAD(&mod->modules_which_use_me);
-        for (i = 0; i < NR_CPUS; i++)
+        for_each_possible_cpu(cpu)
-                local_set(&mod->ref[i].count, 0);
+                local_set(__module_ref_addr(mod, cpu), 0);
        /* Hold reference count during initialization. */
-        local_set(&mod->ref[raw_smp_processor_id()].count, 1);
+        local_set(__module_ref_addr(mod, raw_smp_processor_id()), 1);
        /* Backwards compatibility macros put refcount during init. */
        mod->waiter = current;
 }
@@ -717,10 +717,11 @@ static int try_stop_module(struct module *mod, int flags, int *forced)
 unsigned int module_refcount(struct module *mod)
 {
-        unsigned int i, total = 0;
+        unsigned int total = 0;
+        int cpu;
-        for (i = 0; i < NR_CPUS; i++)
+        for_each_possible_cpu(cpu)
-                total += local_read(&mod->ref[i].count);
+                total += local_read(__module_ref_addr(mod, cpu));
        return total;
 }
 EXPORT_SYMBOL(module_refcount);
@@ -894,7 +895,7 @@ void module_put(struct module *module)
 {
        if (module) {
                unsigned int cpu = get_cpu();
-                local_dec(&module->ref[cpu].count);
+                local_dec(__module_ref_addr(module, cpu));
                /* Maybe they're waiting for us to drop reference? */
                if (unlikely(!module_is_live(module)))
                        wake_up_process(module->waiter);
@@ -1464,7 +1465,10 @@ static void free_module(struct module *mod)
        kfree(mod->args);
        if (mod->percpu)
                percpu_modfree(mod->percpu);
+#if defined(CONFIG_MODULE_UNLOAD) && defined(CONFIG_SMP)
+        if (mod->refptr)
+                percpu_modfree(mod->refptr);
+#endif
        /* Free lock-classes: */
        lockdep_free_key_range(mod->module_core, mod->core_size);
@@ -2011,6 +2015,14 @@ static noinline struct module *load_module(void __user *umod,
        if (err < 0)
                goto free_mod;
+#if defined(CONFIG_MODULE_UNLOAD) && defined(CONFIG_SMP)
+        mod->refptr = percpu_modalloc(sizeof(local_t), __alignof__(local_t),
+                                      mod->name);
+        if (!mod->refptr) {
+                err = -ENOMEM;
+                goto free_mod;
+        }
+#endif
        if (pcpuindex) {
                /* We have a special allocation for this section. */
                percpu = percpu_modalloc(sechdrs[pcpuindex].sh_size,
@@ -2018,7 +2030,7 @@ static noinline struct module *load_module(void __user *umod,
                                         mod->name);
                if (!percpu) {
                        err = -ENOMEM;
-                        goto free_mod;
+                        goto free_percpu;
                }
                sechdrs[pcpuindex].sh_flags &= ~(unsigned long)SHF_ALLOC;
                mod->percpu = percpu;
@@ -2282,6 +2294,9 @@ static noinline struct module *load_module(void __user *umod,
 free_percpu:
        if (percpu)
                percpu_modfree(percpu);
+#if defined(CONFIG_MODULE_UNLOAD) && defined(CONFIG_SMP)
+        percpu_modfree(mod->refptr);
+#endif
 free_mod:
        kfree(args);
 free_hdr:
diff --git a/kernel/panic.c b/kernel/panic.c
index 2a2ff36ff44d..32fe4eff1b89 100644
--- a/kernel/panic.c
+++ b/kernel/panic.c
@@ -74,6 +74,9 @@ NORET_TYPE void panic(const char * fmt, ...)
        vsnprintf(buf, sizeof(buf), fmt, args);
        va_end(args);
        printk(KERN_EMERG "Kernel panic - not syncing: %s\n",buf);
+#ifdef CONFIG_DEBUG_BUGVERBOSE
+        dump_stack();
+#endif
        bust_spinlocks(0);
        /*
@@ -355,15 +358,18 @@ EXPORT_SYMBOL(warn_slowpath);
 #endif
 #ifdef CONFIG_CC_STACKPROTECTOR
 /*
 * Called when gcc's -fstack-protector feature is used, and
 * gcc detects corruption of the on-stack canary value
 */
 void __stack_chk_fail(void)
 {
-        panic("stack-protector: Kernel stack is corrupted");
+        panic("stack-protector: Kernel stack is corrupted in: %p\n",
+                __builtin_return_address(0));
 }
 EXPORT_SYMBOL(__stack_chk_fail);
 #endif
 core_param(panic, panic_timeout, int, 0644);
diff --git a/kernel/posix-cpu-timers.c b/kernel/posix-cpu-timers.c
index fa07da94d7be..2313a4cc14ea 100644
--- a/kernel/posix-cpu-timers.c
+++ b/kernel/posix-cpu-timers.c
@@ -230,6 +230,71 @@ static int cpu_clock_sample(const clockid_t which_clock, struct task_struct *p,
        return 0;
 }
+void thread_group_cputime(struct task_struct *tsk, struct task_cputime *times)
+{
+        struct sighand_struct *sighand;
+        struct signal_struct *sig;
+        struct task_struct *t;
+        *times = INIT_CPUTIME;
+        rcu_read_lock();
+        sighand = rcu_dereference(tsk->sighand);
+        if (!sighand)
+                goto out;
+        sig = tsk->signal;
+        t = tsk;
+        do {
+                times->utime = cputime_add(times->utime, t->utime);
+                times->stime = cputime_add(times->stime, t->stime);
+                times->sum_exec_runtime += t->se.sum_exec_runtime;
+                t = next_thread(t);
+        } while (t != tsk);
+        times->utime = cputime_add(times->utime, sig->utime);
+        times->stime = cputime_add(times->stime, sig->stime);
+        times->sum_exec_runtime += sig->sum_sched_runtime;
+out:
+        rcu_read_unlock();
+}
+static void update_gt_cputime(struct task_cputime *a, struct task_cputime *b)
+{
+        if (cputime_gt(b->utime, a->utime))
+                a->utime = b->utime;
+        if (cputime_gt(b->stime, a->stime))
+                a->stime = b->stime;
+        if (b->sum_exec_runtime > a->sum_exec_runtime)
+                a->sum_exec_runtime = b->sum_exec_runtime;
+}
+void thread_group_cputimer(struct task_struct *tsk, struct task_cputime *times)
+{
+        struct thread_group_cputimer *cputimer = &tsk->signal->cputimer;
+        struct task_cputime sum;
+        unsigned long flags;
+        spin_lock_irqsave(&cputimer->lock, flags);
+        if (!cputimer->running) {
+                cputimer->running = 1;
+                /*
+                 * The POSIX timer interface allows for absolute time expiry
+                 * values through the TIMER_ABSTIME flag, therefore we have
+                 * to synchronize the timer to the clock every time we start
+                 * it.
+                 */
+                thread_group_cputime(tsk, &sum);
+                update_gt_cputime(&cputimer->cputime, &sum);
+        }
+        *times = cputimer->cputime;
+        spin_unlock_irqrestore(&cputimer->lock, flags);
+}
 /*
 * Sample a process (thread group) clock for the given group_leader task.
 * Must be called with tasklist_lock held for reading.
@@ -457,7 +522,7 @@ void posix_cpu_timers_exit_group(struct task_struct *tsk)
 {
        struct task_cputime cputime;
-        thread_group_cputime(tsk, &cputime);
+        thread_group_cputimer(tsk, &cputime);
        cleanup_timers(tsk->signal->cpu_timers,
                       cputime.utime, cputime.stime, cputime.sum_exec_runtime);
 }
@@ -964,6 +1029,19 @@ static void check_thread_timers(struct task_struct *tsk,
        }
 }
+static void stop_process_timers(struct task_struct *tsk)
+{
+        struct thread_group_cputimer *cputimer = &tsk->signal->cputimer;
+        unsigned long flags;
+        if (!cputimer->running)
+                return;
+        spin_lock_irqsave(&cputimer->lock, flags);
+        cputimer->running = 0;
+        spin_unlock_irqrestore(&cputimer->lock, flags);
+}
 /*
 * Check for any per-thread CPU timers that have fired and move them
 * off the tsk->*_timers list onto the firing list.  Per-thread timers
@@ -987,13 +1065,15 @@ static void check_process_timers(struct task_struct *tsk,
            sig->rlim[RLIMIT_CPU].rlim_cur == RLIM_INFINITY &&
            list_empty(&timers[CPUCLOCK_VIRT]) &&
            cputime_eq(sig->it_virt_expires, cputime_zero) &&
-            list_empty(&timers[CPUCLOCK_SCHED]))
+            list_empty(&timers[CPUCLOCK_SCHED])) {
+                stop_process_timers(tsk);
                return;
+        }
        /*
         * Collect the current process totals.
         */
-        thread_group_cputime(tsk, &cputime);
+        thread_group_cputimer(tsk, &cputime);
        utime = cputime.utime;
        ptime = cputime_add(utime, cputime.stime);
        sum_sched_runtime = cputime.sum_exec_runtime;
@@ -1259,7 +1339,7 @@ static inline int fastpath_timer_check(struct task_struct *tsk)
        if (!task_cputime_zero(&sig->cputime_expires)) {
                struct task_cputime group_sample;
-                thread_group_cputime(tsk, &group_sample);
+                thread_group_cputimer(tsk, &group_sample);
                if (task_cputime_expired(&group_sample, &sig->cputime_expires))
                        return 1;
        }
@@ -1329,6 +1409,33 @@ void run_posix_cpu_timers(struct task_struct *tsk)
 }
 /*
+ * Sample a process (thread group) timer for the given group_leader task.
+ * Must be called with tasklist_lock held for reading.
+ */
+static int cpu_timer_sample_group(const clockid_t which_clock,
+                                  struct task_struct *p,
+                                  union cpu_time_count *cpu)
+{
+        struct task_cputime cputime;
+        thread_group_cputimer(p, &cputime);
+        switch (CPUCLOCK_WHICH(which_clock)) {
+        default:
+                return -EINVAL;
+        case CPUCLOCK_PROF:
+                cpu->cpu = cputime_add(cputime.utime, cputime.stime);
+                break;
+        case CPUCLOCK_VIRT:
+                cpu->cpu = cputime.utime;
+                break;
+        case CPUCLOCK_SCHED:
+                cpu->sched = cputime.sum_exec_runtime + task_delta_exec(p);
+                break;
+        }
+        return 0;
+}
+/*
 * Set one of the process-wide special case CPU timers.
 * The tsk->sighand->siglock must be held by the caller.
 * The *newval argument is relative and we update it to be absolute, *oldval
@@ -1341,7 +1448,7 @@ void set_process_cpu_timer(struct task_struct *tsk, unsigned int clock_idx,
        struct list_head *head;
        BUG_ON(clock_idx == CPUCLOCK_SCHED);
-        cpu_clock_sample_group(clock_idx, tsk, &now);
+        cpu_timer_sample_group(clock_idx, tsk, &now);
        if (oldval) {
                if (!cputime_eq(*oldval, cputime_zero)) {
diff --git a/kernel/power/disk.c b/kernel/power/disk.c
index 45e8541ab7e3..432ee575c9ee 100644
--- a/kernel/power/disk.c
+++ b/kernel/power/disk.c
@@ -71,6 +71,14 @@ void hibernation_set_ops(struct platform_hibernation_ops *ops)
        mutex_unlock(&pm_mutex);
 }
+static bool entering_platform_hibernation;
+bool system_entering_hibernation(void)
+{
+        return entering_platform_hibernation;
+}
+EXPORT_SYMBOL(system_entering_hibernation);
 #ifdef CONFIG_PM_DEBUG
 static void hibernation_debug_sleep(void)
 {
@@ -411,6 +419,7 @@ int hibernation_platform_enter(void)
        if (error)
                goto Close;
+        entering_platform_hibernation = true;
        suspend_console();
        error = device_suspend(PMSG_HIBERNATE);
        if (error) {
@@ -445,6 +454,7 @@ int hibernation_platform_enter(void)
 Finish:
        hibernation_ops->finish();
 Resume_devices:
+        entering_platform_hibernation = false;
        device_resume(PMSG_RESTORE);
        resume_console();
 Close:
diff --git a/kernel/power/main.c b/kernel/power/main.c
index 239988873971..b4d219016b6c 100644
--- a/kernel/power/main.c
+++ b/kernel/power/main.c
@@ -57,16 +57,6 @@ int pm_notifier_call_chain(unsigned long val)
 #ifdef CONFIG_PM_DEBUG
 int pm_test_level = TEST_NONE;
-static int suspend_test(int level)
-{
-        if (pm_test_level == level) {
-                printk(KERN_INFO "suspend debug: Waiting for 5 seconds.\n");
-                mdelay(5000);
-                return 1;
-        }
-        return 0;
-}
 static const char * const pm_tests[__TEST_AFTER_LAST] = {
        [TEST_NONE] = "none",
        [TEST_CORE] = "core",
@@ -125,14 +115,24 @@ static ssize_t pm_test_store(struct kobject *kobj, struct kobj_attribute *attr,
 }
 power_attr(pm_test);
-#else /* !CONFIG_PM_DEBUG */
+#endif /* CONFIG_PM_DEBUG */
-static inline int suspend_test(int level) { return 0; }
-#endif /* !CONFIG_PM_DEBUG */
 #endif /* CONFIG_PM_SLEEP */
 #ifdef CONFIG_SUSPEND
+static int suspend_test(int level)
+{
+#ifdef CONFIG_PM_DEBUG
+        if (pm_test_level == level) {
+                printk(KERN_INFO "suspend debug: Waiting for 5 seconds.\n");
+                mdelay(5000);
+                return 1;
+        }
+#endif /* !CONFIG_PM_DEBUG */
+        return 0;
+}
 #ifdef CONFIG_PM_TEST_SUSPEND
 /*
diff --git a/kernel/profile.c b/kernel/profile.c
index 784933acf5b8..7724e0409bae 100644
--- a/kernel/profile.c
+++ b/kernel/profile.c
@@ -114,12 +114,15 @@ int __ref profile_init(void)
        if (!slab_is_available()) {
                prof_buffer = alloc_bootmem(buffer_bytes);
                alloc_bootmem_cpumask_var(&prof_cpu_mask);
+                cpumask_copy(prof_cpu_mask, cpu_possible_mask);
                return 0;
        }
        if (!alloc_cpumask_var(&prof_cpu_mask, GFP_KERNEL))
                return -ENOMEM;
+        cpumask_copy(prof_cpu_mask, cpu_possible_mask);
        prof_buffer = kzalloc(buffer_bytes, GFP_KERNEL);
        if (prof_buffer)
                return 0;
diff --git a/kernel/sched.c b/kernel/sched.c
index 52bbf1c842a8..61245b8d0f16 100644
--- a/kernel/sched.c
+++ b/kernel/sched.c
@@ -3880,19 +3880,24 @@ int select_nohz_load_balancer(int stop_tick)
        int cpu = smp_processor_id();
        if (stop_tick) {
-                cpumask_set_cpu(cpu, nohz.cpu_mask);
                cpu_rq(cpu)->in_nohz_recently = 1;
-                /*
+                if (!cpu_active(cpu)) {
-                 * If we are going offline and still the leader, give up!
+                        if (atomic_read(&nohz.load_balancer) != cpu)
-                 */
+                                return 0;
-                if (!cpu_active(cpu) &&
-                    atomic_read(&nohz.load_balancer) == cpu) {
+                        /*
+                         * If we are going offline and still the leader,
+                         * give up!
+                         */
                        if (atomic_cmpxchg(&nohz.load_balancer, cpu, -1) != cpu)
                                BUG();
                        return 0;
                }
+                cpumask_set_cpu(cpu, nohz.cpu_mask);
                /* time for ilb owner also to sleep */
                if (cpumask_weight(nohz.cpu_mask) == num_online_cpus()) {
                        if (atomic_read(&nohz.load_balancer) == cpu)
@@ -4687,8 +4692,8 @@ EXPORT_SYMBOL(default_wake_function);
 * started to run but is not in state TASK_RUNNING. try_to_wake_up() returns
 * zero in this (rare) case, and we handle it by continuing to scan the queue.
 */
-static void __wake_up_common(wait_queue_head_t *q, unsigned int mode,
+void __wake_up_common(wait_queue_head_t *q, unsigned int mode,
-                             int nr_exclusive, int sync, void *key)
+                        int nr_exclusive, int sync, void *key)
 {
        wait_queue_t *curr, *next;
@@ -5939,12 +5944,7 @@ void sched_show_task(struct task_struct *p)
                printk(KERN_CONT " %016lx ", thread_saved_pc(p));
 #endif
 #ifdef CONFIG_DEBUG_STACK_USAGE
-        {
+        free = stack_not_used(p);
-                unsigned long *n = end_of_stack(p);
-                while (!*n)
-                        n++;
-                free = (unsigned long)n - (unsigned long)end_of_stack(p);
-        }
 #endif
        printk(KERN_CONT "%5lu %5d %6d\n", free,
                task_pid_nr(p), task_pid_nr(p->real_parent));
diff --git a/kernel/sched_fair.c b/kernel/sched_fair.c
index 5cc1c162044f..0566f2a03c42 100644
--- a/kernel/sched_fair.c
+++ b/kernel/sched_fair.c
@@ -719,7 +719,7 @@ enqueue_entity(struct cfs_rq *cfs_rq, struct sched_entity *se, int wakeup)
                __enqueue_entity(cfs_rq, se);
 }
-static void clear_buddies(struct cfs_rq *cfs_rq, struct sched_entity *se)
+static void __clear_buddies(struct cfs_rq *cfs_rq, struct sched_entity *se)
 {
        if (cfs_rq->last == se)
                cfs_rq->last = NULL;
@@ -728,6 +728,12 @@ static void clear_buddies(struct cfs_rq *cfs_rq, struct sched_entity *se)
                cfs_rq->next = NULL;
 }
+static void clear_buddies(struct cfs_rq *cfs_rq, struct sched_entity *se)
+{
+        for_each_sched_entity(se)
+                __clear_buddies(cfs_rq_of(se), se);
+}
 static void
 dequeue_entity(struct cfs_rq *cfs_rq, struct sched_entity *se, int sleep)
 {
@@ -768,8 +774,14 @@ check_preempt_tick(struct cfs_rq *cfs_rq, struct sched_entity *curr)
        ideal_runtime = sched_slice(cfs_rq, curr);
        delta_exec = curr->sum_exec_runtime - curr->prev_sum_exec_runtime;
-        if (delta_exec > ideal_runtime)
+        if (delta_exec > ideal_runtime) {
                resched_task(rq_of(cfs_rq)->curr);
+                /*
+                 * The current task ran long enough, ensure it doesn't get
+                 * re-elected due to buddy favours.
+                 */
+                clear_buddies(cfs_rq, curr);
+        }
 }
 static void
@@ -1452,6 +1464,11 @@ static struct task_struct *pick_next_task_fair(struct rq *rq)
        do {
                se = pick_next_entity(cfs_rq);
+                /*
+                 * If se was a buddy, clear it so that it will have to earn
+                 * the favour again.
+                 */
+                __clear_buddies(cfs_rq, se);
                set_next_entity(cfs_rq, se);
                cfs_rq = group_cfs_rq(se);
        } while (cfs_rq);
diff --git a/kernel/sched_rt.c b/kernel/sched_rt.c
index 954e1a81b796..da932f4c8524 100644
--- a/kernel/sched_rt.c
+++ b/kernel/sched_rt.c
@@ -960,16 +960,17 @@ static struct task_struct *pick_next_highest_task_rt(struct rq *rq, int cpu)
 static DEFINE_PER_CPU(cpumask_var_t, local_cpu_mask);
-static inline int pick_optimal_cpu(int this_cpu, cpumask_t *mask)
+static inline int pick_optimal_cpu(int this_cpu,
+                                   const struct cpumask *mask)
 {
        int first;
        /* "this_cpu" is cheaper to preempt than a remote processor */
-        if ((this_cpu != -1) && cpu_isset(this_cpu, *mask))
+        if ((this_cpu != -1) && cpumask_test_cpu(this_cpu, mask))
                return this_cpu;
-        first = first_cpu(*mask);
+        first = cpumask_first(mask);
-        if (first != NR_CPUS)
+        if (first < nr_cpu_ids)
                return first;
        return -1;
@@ -981,6 +982,7 @@ static int find_lowest_rq(struct task_struct *task)
        struct cpumask *lowest_mask = __get_cpu_var(local_cpu_mask);
        int this_cpu = smp_processor_id();
        int cpu      = task_cpu(task);
+        cpumask_var_t domain_mask;
        if (task->rt.nr_cpus_allowed == 1)
                return -1; /* No other targets possible */
@@ -1013,19 +1015,25 @@ static int find_lowest_rq(struct task_struct *task)
        if (this_cpu == cpu)
                this_cpu = -1; /* Skip this_cpu opt if the same */
-        for_each_domain(cpu, sd) {
+        if (alloc_cpumask_var(&domain_mask, GFP_ATOMIC)) {
-                if (sd->flags & SD_WAKE_AFFINE) {
+                for_each_domain(cpu, sd) {
-                        cpumask_t domain_mask;
+                        if (sd->flags & SD_WAKE_AFFINE) {
-                        int       best_cpu;
+                                int best_cpu;
-                        cpumask_and(&domain_mask, sched_domain_span(sd),
+                                cpumask_and(domain_mask,
-                                    lowest_mask);
+                                            sched_domain_span(sd),
+                                            lowest_mask);
-                        best_cpu = pick_optimal_cpu(this_cpu,
+                                best_cpu = pick_optimal_cpu(this_cpu,
-                                                    &domain_mask);
+                                                            domain_mask);
-                        if (best_cpu != -1)
-                                return best_cpu;
+                                if (best_cpu != -1) {
+                                        free_cpumask_var(domain_mask);
+                                        return best_cpu;
+                                }
+                        }
                }
+                free_cpumask_var(domain_mask);
        }
        /*
diff --git a/kernel/sched_stats.h b/kernel/sched_stats.h
index 8ab0cef8ecab..a8f93dd374e1 100644
--- a/kernel/sched_stats.h
+++ b/kernel/sched_stats.h
@@ -296,19 +296,21 @@ sched_info_switch(struct task_struct *prev, struct task_struct *next)
 static inline void account_group_user_time(struct task_struct *tsk,
                                           cputime_t cputime)
 {
-        struct task_cputime *times;
+        struct thread_group_cputimer *cputimer;
-        struct signal_struct *sig;
        /* tsk == current, ensure it is safe to use ->signal */
        if (unlikely(tsk->exit_state))
                return;
-        sig = tsk->signal;
+        cputimer = &tsk->signal->cputimer;
-        times = &sig->cputime.totals;
-        spin_lock(&times->lock);
+        if (!cputimer->running)
-        times->utime = cputime_add(times->utime, cputime);
+                return;
-        spin_unlock(&times->lock);
+        spin_lock(&cputimer->lock);
+        cputimer->cputime.utime =
+                cputime_add(cputimer->cputime.utime, cputime);
+        spin_unlock(&cputimer->lock);
 }
 /**
@@ -324,19 +326,21 @@ static inline void account_group_user_time(struct task_struct *tsk,
 static inline void account_group_system_time(struct task_struct *tsk,
                                             cputime_t cputime)
 {
-        struct task_cputime *times;
+        struct thread_group_cputimer *cputimer;
-        struct signal_struct *sig;
        /* tsk == current, ensure it is safe to use ->signal */
        if (unlikely(tsk->exit_state))
                return;
-        sig = tsk->signal;
+        cputimer = &tsk->signal->cputimer;
-        times = &sig->cputime.totals;
+        if (!cputimer->running)
+                return;
-        spin_lock(&times->lock);
+        spin_lock(&cputimer->lock);
-        times->stime = cputime_add(times->stime, cputime);
+        cputimer->cputime.stime =
-        spin_unlock(&times->lock);
+                cputime_add(cputimer->cputime.stime, cputime);
+        spin_unlock(&cputimer->lock);
 }
 /**
@@ -352,7 +356,7 @@ static inline void account_group_system_time(struct task_struct *tsk,
 static inline void account_group_exec_runtime(struct task_struct *tsk,
                                              unsigned long long ns)
 {
-        struct task_cputime *times;
+        struct thread_group_cputimer *cputimer;
        struct signal_struct *sig;
        sig = tsk->signal;
@@ -361,9 +365,12 @@ static inline void account_group_exec_runtime(struct task_struct *tsk,
        if (unlikely(!sig))
                return;
-        times = &sig->cputime.totals;
+        cputimer = &sig->cputimer;
+        if (!cputimer->running)
+                return;
-        spin_lock(&times->lock);
+        spin_lock(&cputimer->lock);
-        times->sum_exec_runtime += ns;
+        cputimer->cputime.sum_exec_runtime += ns;
-        spin_unlock(&times->lock);
+        spin_unlock(&cputimer->lock);
 }
diff --git a/kernel/signal.c b/kernel/signal.c
index e73759783dc8..2a74fe87c0dd 100644
--- a/kernel/signal.c
+++ b/kernel/signal.c
@@ -909,7 +909,9 @@ static void print_fatal_signal(struct pt_regs *regs, int signr)
        }
 #endif
        printk("\n");
+        preempt_disable();
        show_regs(regs);
+        preempt_enable();
 }
 static int __init setup_print_fatal_signals(char *str)
@@ -1365,7 +1367,6 @@ int do_notify_parent(struct task_struct *tsk, int sig)
        struct siginfo info;
        unsigned long flags;
        struct sighand_struct *psig;
-        struct task_cputime cputime;
        int ret = sig;
        BUG_ON(sig == -1);
@@ -1395,9 +1396,10 @@ int do_notify_parent(struct task_struct *tsk, int sig)
        info.si_uid = __task_cred(tsk)->uid;
        rcu_read_unlock();
-        thread_group_cputime(tsk, &cputime);
+        info.si_utime = cputime_to_clock_t(cputime_add(tsk->utime,
-        info.si_utime = cputime_to_jiffies(cputime.utime);
+                                tsk->signal->utime));
-        info.si_stime = cputime_to_jiffies(cputime.stime);
+        info.si_stime = cputime_to_clock_t(cputime_add(tsk->stime,
+                                tsk->signal->stime));
        info.si_status = tsk->exit_code & 0x7f;
        if (tsk->exit_code & 0x80)
diff --git a/kernel/smp.c b/kernel/smp.c
index 5cfa0e5e3e88..bbedbb7efe32 100644
--- a/kernel/smp.c
+++ b/kernel/smp.c
@@ -18,6 +18,7 @@ __cacheline_aligned_in_smp DEFINE_SPINLOCK(call_function_lock);
 enum {
        CSD_FLAG_WAIT           = 0x01,
        CSD_FLAG_ALLOC          = 0x02,
+        CSD_FLAG_LOCK           = 0x04,
 };
 struct call_function_data {
@@ -186,6 +187,9 @@ void generic_smp_call_function_single_interrupt(void)
                        if (data_flags & CSD_FLAG_WAIT) {
                                smp_wmb();
                                data->flags &= ~CSD_FLAG_WAIT;
+                        } else if (data_flags & CSD_FLAG_LOCK) {
+                                smp_wmb();
+                                data->flags &= ~CSD_FLAG_LOCK;
                        } else if (data_flags & CSD_FLAG_ALLOC)
                                kfree(data);
                }
@@ -196,6 +200,8 @@ void generic_smp_call_function_single_interrupt(void)
        }
 }
+static DEFINE_PER_CPU(struct call_single_data, csd_data);
 /*
 * smp_call_function_single - Run a function on a specific CPU
 * @func: The function to run. This must be fast and non-blocking.
@@ -224,14 +230,38 @@ int smp_call_function_single(int cpu, void (*func) (void *info), void *info,
                func(info);
                local_irq_restore(flags);
        } else if ((unsigned)cpu < nr_cpu_ids && cpu_online(cpu)) {
-                struct call_single_data *data = NULL;
+                struct call_single_data *data;
                if (!wait) {
+                        /*
+                         * We are calling a function on a single CPU
+                         * and we are not going to wait for it to finish.
+                         * We first try to allocate the data, but if we
+                         * fail, we fall back to use a per cpu data to pass
+                         * the information to that CPU. Since all callers
+                         * of this code will use the same data, we must
+                         * synchronize the callers to prevent a new caller
+                         * from corrupting the data before the callee
+                         * can access it.
+                         *
+                         * The CSD_FLAG_LOCK is used to let us know when
+                         * the IPI handler is done with the data.
+                         * The first caller will set it, and the callee
+                         * will clear it. The next caller must wait for
+                         * it to clear before we set it again. This
+                         * will make sure the callee is done with the
+                         * data before a new caller will use it.
+                         */
                        data = kmalloc(sizeof(*data), GFP_ATOMIC);
                        if (data)
                                data->flags = CSD_FLAG_ALLOC;
-                }
+                        else {
-                if (!data) {
+                                data = &per_cpu(csd_data, me);
+                                while (data->flags & CSD_FLAG_LOCK)
+                                        cpu_relax();
+                                data->flags = CSD_FLAG_LOCK;
+                        }
+                } else {
                        data = &d;
                        data->flags = CSD_FLAG_WAIT;
                }
diff --git a/kernel/softirq.c b/kernel/softirq.c
index bdbe9de9cd8d..0365b4899a3d 100644
--- a/kernel/softirq.c
+++ b/kernel/softirq.c
@@ -795,6 +795,11 @@ int __init __weak early_irq_init(void)
        return 0;
 }
+int __init __weak arch_probe_nr_irqs(void)
+{
+        return 0;
+}
 int __init __weak arch_early_irq_init(void)
 {
        return 0;
diff --git a/kernel/sys.c b/kernel/sys.c
index e7dc0e10a485..f145c415bc16 100644
--- a/kernel/sys.c
+++ b/kernel/sys.c
@@ -1525,22 +1525,14 @@ SYSCALL_DEFINE2(setrlimit, unsigned int, resource, struct rlimit __user *, rlim)
                return -EINVAL;
        if (copy_from_user(&new_rlim, rlim, sizeof(*rlim)))
                return -EFAULT;
+        if (new_rlim.rlim_cur > new_rlim.rlim_max)
+                return -EINVAL;
        old_rlim = current->signal->rlim + resource;
        if ((new_rlim.rlim_max > old_rlim->rlim_max) &&
            !capable(CAP_SYS_RESOURCE))
                return -EPERM;
+        if (resource == RLIMIT_NOFILE && new_rlim.rlim_max > sysctl_nr_open)
-        if (resource == RLIMIT_NOFILE) {
+                return -EPERM;
-                if (new_rlim.rlim_max == RLIM_INFINITY)
-                        new_rlim.rlim_max = sysctl_nr_open;
-                if (new_rlim.rlim_cur == RLIM_INFINITY)
-                        new_rlim.rlim_cur = sysctl_nr_open;
-                if (new_rlim.rlim_max > sysctl_nr_open)
-                        return -EPERM;
-        }
-        if (new_rlim.rlim_cur > new_rlim.rlim_max)
-                return -EINVAL;
        retval = security_task_setrlimit(resource, &new_rlim);
        if (retval)
diff --git a/kernel/sysctl.c b/kernel/sysctl.c
index 790f9d785663..c5ef44ff850f 100644
--- a/kernel/sysctl.c
+++ b/kernel/sysctl.c
@@ -101,6 +101,7 @@ static int two = 2;
 static int zero;
 static int one = 1;
+static unsigned long one_ul = 1;
 static int one_hundred = 100;
 /* this is needed for the proc_dointvec_minmax for [fs_]overflow UID and GID */
@@ -974,7 +975,7 @@ static struct ctl_table vm_table[] = {
                .mode           = 0644,
                .proc_handler   = &dirty_background_bytes_handler,
                .strategy       = &sysctl_intvec,
-                .extra1         = &one,
+                .extra1         = &one_ul,
        },
        {
                .ctl_name       = VM_DIRTY_RATIO,
@@ -995,7 +996,7 @@ static struct ctl_table vm_table[] = {
                .mode           = 0644,
                .proc_handler   = &dirty_bytes_handler,
                .strategy       = &sysctl_intvec,
-                .extra1         = &one,
+                .extra1         = &one_ul,
        },
        {
                .procname       = "dirty_writeback_centisecs",
diff --git a/kernel/time/tick-common.c b/kernel/time/tick-common.c
index 63e05d423a09..21a5ca849514 100644
--- a/kernel/time/tick-common.c
+++ b/kernel/time/tick-common.c
@@ -274,6 +274,21 @@ out_bc:
 }
 /*
+ * Transfer the do_timer job away from a dying cpu.
+ *
+ * Called with interrupts disabled.
+ */
+static void tick_handover_do_timer(int *cpup)
+{
+        if (*cpup == tick_do_timer_cpu) {
+                int cpu = cpumask_first(cpu_online_mask);
+                tick_do_timer_cpu = (cpu < nr_cpu_ids) ? cpu :
+                        TICK_DO_TIMER_NONE;
+        }
+}
+/*
 * Shutdown an event device on a given cpu:
 *
 * This is called on a life CPU, when a CPU is dead. So we cannot
@@ -297,13 +312,6 @@ static void tick_shutdown(unsigned int *cpup)
                clockevents_exchange_device(dev, NULL);
                td->evtdev = NULL;
        }
-        /* Transfer the do_timer job away from this cpu */
-        if (*cpup == tick_do_timer_cpu) {
-                int cpu = cpumask_first(cpu_online_mask);
-                tick_do_timer_cpu = (cpu < nr_cpu_ids) ? cpu :
-                        TICK_DO_TIMER_NONE;
-        }
        spin_unlock_irqrestore(&tick_device_lock, flags);
 }
@@ -357,6 +365,10 @@ static int tick_notify(struct notifier_block *nb, unsigned long reason,
                tick_broadcast_oneshot_control(reason);
                break;
+        case CLOCK_EVT_NOTIFY_CPU_DYING:
+                tick_handover_do_timer(dev);
+                break;
        case CLOCK_EVT_NOTIFY_CPU_DEAD:
                tick_shutdown_broadcast_oneshot(dev);
                tick_shutdown_broadcast(dev);
diff --git a/kernel/trace/ftrace.c b/kernel/trace/ftrace.c
index 2f32969c09df..9a236ffe2aa4 100644
--- a/kernel/trace/ftrace.c
+++ b/kernel/trace/ftrace.c
@@ -17,6 +17,7 @@
 #include <linux/clocksource.h>
 #include <linux/kallsyms.h>
 #include <linux/seq_file.h>
+#include <linux/suspend.h>
 #include <linux/debugfs.h>
 #include <linux/hardirq.h>
 #include <linux/kthread.h>
@@ -1736,9 +1737,12 @@ static void clear_ftrace_pid(struct pid *pid)
 {
        struct task_struct *p;
+        rcu_read_lock();
        do_each_pid_task(pid, PIDTYPE_PID, p) {
                clear_tsk_trace_trace(p);
        } while_each_pid_task(pid, PIDTYPE_PID, p);
+        rcu_read_unlock();
        put_pid(pid);
 }
@@ -1746,9 +1750,11 @@ static void set_ftrace_pid(struct pid *pid)
 {
        struct task_struct *p;
+        rcu_read_lock();
        do_each_pid_task(pid, PIDTYPE_PID, p) {
                set_tsk_trace_trace(p);
        } while_each_pid_task(pid, PIDTYPE_PID, p);
+        rcu_read_unlock();
 }
 static void clear_ftrace_pid_task(struct pid **pid)
@@ -1965,6 +1971,7 @@ ftrace_enable_sysctl(struct ctl_table *table, int write,
 #ifdef CONFIG_FUNCTION_GRAPH_TRACER
 static atomic_t ftrace_graph_active;
+static struct notifier_block ftrace_suspend_notifier;
 int ftrace_graph_entry_stub(struct ftrace_graph_ent *trace)
 {
@@ -2043,6 +2050,27 @@ static int start_graph_tracing(void)
        return ret;
 }
+/*
+ * Hibernation protection.
+ * The state of the current task is too much unstable during
+ * suspend/restore to disk. We want to protect against that.
+ */
+static int
+ftrace_suspend_notifier_call(struct notifier_block *bl, unsigned long state,
+                                                        void *unused)
+{
+        switch (state) {
+        case PM_HIBERNATION_PREPARE:
+                pause_graph_tracing();
+                break;
+        case PM_POST_HIBERNATION:
+                unpause_graph_tracing();
+                break;
+        }
+        return NOTIFY_DONE;
+}
 int register_ftrace_graph(trace_func_graph_ret_t retfunc,
                        trace_func_graph_ent_t entryfunc)
 {
@@ -2050,6 +2078,9 @@ int register_ftrace_graph(trace_func_graph_ret_t retfunc,
        mutex_lock(&ftrace_sysctl_lock);
+        ftrace_suspend_notifier.notifier_call = ftrace_suspend_notifier_call;
+        register_pm_notifier(&ftrace_suspend_notifier);
        atomic_inc(&ftrace_graph_active);
        ret = start_graph_tracing();
        if (ret) {
@@ -2075,6 +2106,7 @@ void unregister_ftrace_graph(void)
        ftrace_graph_return = (trace_func_graph_ret_t)ftrace_stub;
        ftrace_graph_entry = ftrace_graph_entry_stub;
        ftrace_shutdown(FTRACE_STOP_FUNC_RET);
+        unregister_pm_notifier(&ftrace_suspend_notifier);
        mutex_unlock(&ftrace_sysctl_lock);
 }
diff --git a/kernel/trace/ring_buffer.c b/kernel/trace/ring_buffer.c
index 8b0daf0662ef..bd38c5cfd8ad 100644
--- a/kernel/trace/ring_buffer.c
+++ b/kernel/trace/ring_buffer.c
@@ -246,7 +246,7 @@ static inline int test_time_stamp(u64 delta)
        return 0;
 }
-#define BUF_PAGE_SIZE (PAGE_SIZE - sizeof(struct buffer_data_page))
+#define BUF_PAGE_SIZE (PAGE_SIZE - offsetof(struct buffer_data_page, data))
 /*
 * head_page == tail_page && head == tail then buffer is empty.
@@ -1025,12 +1025,8 @@ __rb_reserve_next(struct ring_buffer_per_cpu *cpu_buffer,
                }
                if (next_page == head_page) {
-                        if (!(buffer->flags & RB_FL_OVERWRITE)) {
+                        if (!(buffer->flags & RB_FL_OVERWRITE))
-                                /* reset write */
-                                if (tail <= BUF_PAGE_SIZE)
-                                        local_set(&tail_page->write, tail);
                                goto out_unlock;
-                        }
                        /* tail_page has not moved yet? */
                        if (tail_page == cpu_buffer->tail_page) {
@@ -1105,6 +1101,10 @@ __rb_reserve_next(struct ring_buffer_per_cpu *cpu_buffer,
        return event;
 out_unlock:
+        /* reset write */
+        if (tail <= BUF_PAGE_SIZE)
+                local_set(&tail_page->write, tail);
        __raw_spin_unlock(&cpu_buffer->lock);
        local_irq_restore(flags);
        return NULL;
@@ -2174,6 +2174,9 @@ rb_reset_cpu(struct ring_buffer_per_cpu *cpu_buffer)
        cpu_buffer->overrun = 0;
        cpu_buffer->entries = 0;
+        cpu_buffer->write_stamp = 0;
+        cpu_buffer->read_stamp = 0;
 }
 /**
diff --git a/kernel/trace/trace.c b/kernel/trace/trace.c
index c580233add95..17bb88d86ac2 100644
--- a/kernel/trace/trace.c
+++ b/kernel/trace/trace.c
@@ -40,7 +40,7 @@
 #define TRACE_BUFFER_FLAGS      (RB_FL_OVERWRITE)
-unsigned long __read_mostly     tracing_max_latency = (cycle_t)ULONG_MAX;
+unsigned long __read_mostly     tracing_max_latency;
 unsigned long __read_mostly     tracing_thresh;
 /*
@@ -3736,7 +3736,7 @@ static struct notifier_block trace_die_notifier = {
 * it if we decide to change what log level the ftrace dump
 * should be at.
 */
-#define KERN_TRACE              KERN_INFO
+#define KERN_TRACE              KERN_EMERG
 static void
 trace_printk_seq(struct trace_seq *s)
@@ -3770,6 +3770,7 @@ void ftrace_dump(void)
        dump_ran = 1;
        /* No turning back! */
+        tracing_off();
        ftrace_kill();
        for_each_tracing_cpu(cpu) {
diff --git a/kernel/trace/trace_irqsoff.c b/kernel/trace/trace_irqsoff.c
index 7c2e326bbc8b..62a78d943534 100644
--- a/kernel/trace/trace_irqsoff.c
+++ b/kernel/trace/trace_irqsoff.c
@@ -380,6 +380,7 @@ static void stop_irqsoff_tracer(struct trace_array *tr)
 static void __irqsoff_tracer_init(struct trace_array *tr)
 {
+        tracing_max_latency = 0;
        irqsoff_trace = tr;
        /* make sure that the tracer is visible */
        smp_wmb();
diff --git a/kernel/trace/trace_sched_wakeup.c b/kernel/trace/trace_sched_wakeup.c
index 43586b689e31..42ae1e77b6b3 100644
--- a/kernel/trace/trace_sched_wakeup.c
+++ b/kernel/trace/trace_sched_wakeup.c
@@ -333,6 +333,7 @@ static void stop_wakeup_tracer(struct trace_array *tr)
 static int wakeup_tracer_init(struct trace_array *tr)
 {
+        tracing_max_latency = 0;
        wakeup_trace = tr;
        start_wakeup_tracer(tr);
        return 0;
diff --git a/kernel/wait.c b/kernel/wait.c
index cd87131f2fc2..42a2dbc181c8 100644
--- a/kernel/wait.c
+++ b/kernel/wait.c
@@ -91,6 +91,15 @@ prepare_to_wait_exclusive(wait_queue_head_t *q, wait_queue_t *wait, int state)
 }
 EXPORT_SYMBOL(prepare_to_wait_exclusive);
+/*
+ * finish_wait - clean up after waiting in a queue
+ * @q: waitqueue waited on
+ * @wait: wait descriptor
+ *
+ * Sets current thread back to running state and removes
+ * the wait descriptor from the given waitqueue if still
+ * queued.
+ */
 void finish_wait(wait_queue_head_t *q, wait_queue_t *wait)
 {
        unsigned long flags;
@@ -117,6 +126,39 @@ void finish_wait(wait_queue_head_t *q, wait_queue_t *wait)
 }
 EXPORT_SYMBOL(finish_wait);
+/*
+ * abort_exclusive_wait - abort exclusive waiting in a queue
+ * @q: waitqueue waited on
+ * @wait: wait descriptor
+ * @state: runstate of the waiter to be woken
+ * @key: key to identify a wait bit queue or %NULL
+ *
+ * Sets current thread back to running state and removes
+ * the wait descriptor from the given waitqueue if still
+ * queued.
+ *
+ * Wakes up the next waiter if the caller is concurrently
+ * woken up through the queue.
+ *
+ * This prevents waiter starvation where an exclusive waiter
+ * aborts and is woken up concurrently and noone wakes up
+ * the next waiter.
+ */
+void abort_exclusive_wait(wait_queue_head_t *q, wait_queue_t *wait,
+                        unsigned int mode, void *key)
+{
+        unsigned long flags;
+        __set_current_state(TASK_RUNNING);
+        spin_lock_irqsave(&q->lock, flags);
+        if (!list_empty(&wait->task_list))
+                list_del_init(&wait->task_list);
+        else if (waitqueue_active(q))
+                __wake_up_common(q, mode, 1, 0, key);
+        spin_unlock_irqrestore(&q->lock, flags);
+}
+EXPORT_SYMBOL(abort_exclusive_wait);
 int autoremove_wake_function(wait_queue_t *wait, unsigned mode, int sync, void *key)
 {
        int ret = default_wake_function(wait, mode, sync, key);
@@ -177,17 +219,20 @@ int __sched
 __wait_on_bit_lock(wait_queue_head_t *wq, struct wait_bit_queue *q,
                        int (*action)(void *), unsigned mode)
 {
-        int ret = 0;
        do {
+                int ret;
                prepare_to_wait_exclusive(wq, &q->wait, mode);
-                if (test_bit(q->key.bit_nr, q->key.flags)) {
+                if (!test_bit(q->key.bit_nr, q->key.flags))
-                        if ((ret = (*action)(q->key.flags)))
+                        continue;
-                                break;
+                ret = action(q->key.flags);
-                }
+                if (!ret)
+                        continue;
+                abort_exclusive_wait(wq, &q->wait, mode, &q->key);
+                return ret;
        } while (test_and_set_bit(q->key.bit_nr, q->key.flags));
        finish_wait(wq, &q->wait);
-        return ret;
+        return 0;
 }
 EXPORT_SYMBOL(__wait_on_bit_lock);