sched/loadavg: Avoid loadavg spikes caused by delayed NO_HZ accounting

[pandora-kernel.git] / kernel / sched.c
diff --git a/kernel/sched.c b/kernel/sched.c

index a409d81..4b3e12e 100644 (file)
--- a/kernel/sched.c
+++ b/kernel/sched.c
@@ -746,22 +746,19 @@ static inline int cpu_of(struct rq *rq)
  /*
   * Return the group to which this tasks belongs.
   *
- * We use task_subsys_state_check() and extend the RCU verification with
- * pi->lock and rq->lock because cpu_cgroup_attach() holds those locks for each
- * task it moves into the cgroup. Therefore by holding either of those locks,
- * we pin the task to the current cgroup.
+ * We cannot use task_subsys_state() and friends because the cgroup
+ * subsystem changes that value before the cgroup_subsys::attach() method
+ * is called, therefore we cannot pin it and might observe the wrong value.
+ *
+ * The same is true for autogroup's p->signal->autogroup->tg, the autogroup
+ * core changes this before calling sched_move_task().
+ *
+ * Instead we use a 'copy' which is updated from sched_move_task() while
+ * holding both task_struct::pi_lock and rq::lock.
   */
  static inline struct task_group *task_group(struct task_struct *p)
  {
-       struct task_group *tg;
-       struct cgroup_subsys_state *css;
-
-       css = task_subsys_state_check(p, cpu_cgroup_subsys_id,
-                       lockdep_is_held(&p->pi_lock) ||
-                       lockdep_is_held(&task_rq(p)->lock));
-       tg = container_of(css, struct task_group, css);
-
-       return autogroup_task_group(p, tg);
+       return p->sched_task_group;
  }
  
  /* Change a task's cfs_rq and parent entity if it moves across CPUs/groups */
@@ -1019,8 +1016,10 @@ static inline void finish_lock_switch(struct rq *rq, struct task_struct *prev)
          * After ->on_cpu is cleared, the task can be moved to a different CPU.
          * We must ensure this doesn't happen until the switch is completely
          * finished.
+        *
+        * Pairs with the control dependency and rmb in try_to_wake_up().
          */
-       smp_wmb();
+       smp_mb();
         prev->on_cpu = 0;
  #endif
  #ifdef CONFIG_DEBUG_SPINLOCK
@@ -2192,6 +2191,10 @@ static int irqtime_account_si_update(void)
  
  #endif
  
+#ifdef CONFIG_SMP
+static void unthrottle_offline_cfs_rqs(struct rq *rq);
+#endif
+
  #include "sched_idletask.c"
  #include "sched_fair.c"
  #include "sched_rt.c"
@@ -2372,7 +2375,7 @@ void set_task_cpu(struct task_struct *p, unsigned int new_cpu)
          * a task's CPU. ->pi_lock for waking tasks, rq->lock for runnable tasks.
          *
          * sched_move_task() holds both and thus holding either pins the cgroup,
-        * see set_task_rq().
+        * see task_group().
          *
          * Furthermore, all task_rq users should acquire both locks, see
          * task_rq_lock().
@@ -2830,6 +2833,28 @@ try_to_wake_up(struct task_struct *p, unsigned int state, int wake_flags)
         success = 1; /* we're going to change ->state */
         cpu = task_cpu(p);
  
+       /*
+        * Ensure we load p->on_rq _after_ p->state, otherwise it would
+        * be possible to, falsely, observe p->on_rq == 0 and get stuck
+        * in smp_cond_load_acquire() below.
+        *
+        * sched_ttwu_pending()                 try_to_wake_up()
+        *   [S] p->on_rq = 1;                  [L] P->state
+        *       UNLOCK rq->lock  -----.
+        *                              \
+        *                               +---   RMB
+        * schedule()                   /
+        *       LOCK rq->lock    -----'
+        *       UNLOCK rq->lock
+        *
+        * [task p]
+        *   [S] p->state = UNINTERRUPTIBLE     [L] p->on_rq
+        *
+        * Pairs with the UNLOCK+LOCK on rq->lock from the
+        * last wakeup of our task and the schedule that got our task
+        * current.
+        */
+       smp_rmb();
         if (p->on_rq && ttwu_remote(p, wake_flags))
                 goto stat;
  
@@ -2892,8 +2917,10 @@ static void try_to_wake_up_local(struct task_struct *p)
  {
         struct rq *rq = task_rq(p);
  
-       BUG_ON(rq != this_rq());
-       BUG_ON(p == current);
+       if (WARN_ON_ONCE(rq != this_rq()) ||
+           WARN_ON_ONCE(p == current))
+               return;
+
         lockdep_assert_held(&rq->lock);
  
         if (!raw_spin_trylock(&p->pi_lock)) {
@@ -2927,7 +2954,7 @@ out:
   */
  int wake_up_process(struct task_struct *p)
  {
-       return try_to_wake_up(p, TASK_ALL, 0);
+       return try_to_wake_up(p, TASK_NORMAL, 0);
  }
  EXPORT_SYMBOL(wake_up_process);
  
@@ -3187,11 +3214,11 @@ static void finish_task_switch(struct rq *rq, struct task_struct *prev)
          * If a task dies, then it sets TASK_DEAD in tsk->state and calls
          * schedule one last time. The schedule call will never return, and
          * the scheduled task must drop that reference.
-        * The test for TASK_DEAD must occur while the runqueue locks are
-        * still held, otherwise prev could be scheduled on another cpu, die
-        * there before we look at prev->state, and then the reference would
-        * be dropped twice.
-        *              Manfred Spraul <manfred@colorfullife.com>
+        *
+        * We must observe prev->state before clearing prev->on_cpu (in
+        * finish_lock_switch), otherwise a concurrent wakeup can get prev
+        * running on another CPU and we could rave with its RUNNING -> DEAD
+        * transition, resulting in a double drop.
          */
         prev_state = prev->state;
         finish_arch_switch(prev);
@@ -3489,10 +3516,13 @@ static long calc_load_fold_active(struct rq *this_rq)
  static unsigned long
  calc_load(unsigned long load, unsigned long exp, unsigned long active)
  {
-       load *= exp;
-       load += active * (FIXED_1 - exp);
-       load += 1UL << (FSHIFT - 1);
-       return load >> FSHIFT;
+       unsigned long newload;
+
+       newload = load * exp + active * (FIXED_1 - exp);
+       if (active >= load)
+               newload += FIXED_1-1;
+
+       return newload / FIXED_1;
  }
  
  #ifdef CONFIG_NO_HZ
@@ -3587,8 +3617,9 @@ void calc_load_exit_idle(void)
         struct rq *this_rq = this_rq();
  
         /*
-        * If we're still before the sample window, we're done.
+        * If we're still before the pending sample window, we're done.
          */
+       this_rq->calc_load_update = calc_load_update;
         if (time_before(jiffies, this_rq->calc_load_update))
                 return;
  
@@ -3597,7 +3628,6 @@ void calc_load_exit_idle(void)
          * accounted through the nohz accounting, so skip the entire deal and
          * sync up for the next window.
          */
-       this_rq->calc_load_update = calc_load_update;
         if (time_before(jiffies, this_rq->calc_load_update + 10))
                 this_rq->calc_load_update += LOAD_FREQ;
  }
@@ -3886,25 +3916,32 @@ static void __update_cpu_load(struct rq *this_rq, unsigned long this_load,
         sched_avg_update(this_rq);
  }
  
+#ifdef CONFIG_NO_HZ
+/*
+ * There is no sane way to deal with nohz on smp when using jiffies because the
+ * cpu doing the jiffies update might drift wrt the cpu doing the jiffy reading
+ * causing off-by-one errors in observed deltas; {0,2} instead of {1,1}.
+ *
+ * Therefore we cannot use the delta approach from the regular tick since that
+ * would seriously skew the load calculation. However we'll make do for those
+ * updates happening while idle (nohz_idle_balance) or coming out of idle
+ * (tick_nohz_idle_exit).
+ *
+ * This means we might still be one tick off for nohz periods.
+ */
+
  /*
   * Called from nohz_idle_balance() to update the load ratings before doing the
   * idle balance.
   */
  static void update_idle_cpu_load(struct rq *this_rq)
  {
-       unsigned long curr_jiffies = jiffies;
+       unsigned long curr_jiffies = ACCESS_ONCE(jiffies);
         unsigned long load = this_rq->load.weight;
         unsigned long pending_updates;
  
         /*
-        * Bloody broken means of dealing with nohz, but better than nothing..
-        * jiffies is updated by one cpu, another cpu can drift wrt the jiffy
-        * update and see 0 difference the one time and 2 the next, even though
-        * we ticked at roughtly the same rate.
-        *
-        * Hence we only use this from nohz_idle_balance() and skip this
-        * nonsense when called from the scheduler_tick() since that's
-        * guaranteed a stable rate.
+        * bail if there's load or we're actually up-to-date.
          */
         if (load || curr_jiffies == this_rq->last_load_update_tick)
                 return;
@@ -3915,13 +3952,39 @@ static void update_idle_cpu_load(struct rq *this_rq)
         __update_cpu_load(this_rq, load, pending_updates);
  }
  
+/*
+ * Called from tick_nohz_idle_exit() -- try and fix up the ticks we missed.
+ */
+void update_cpu_load_nohz(void)
+{
+       struct rq *this_rq = this_rq();
+       unsigned long curr_jiffies = ACCESS_ONCE(jiffies);
+       unsigned long pending_updates;
+
+       if (curr_jiffies == this_rq->last_load_update_tick)
+               return;
+
+       raw_spin_lock(&this_rq->lock);
+       pending_updates = curr_jiffies - this_rq->last_load_update_tick;
+       if (pending_updates) {
+               this_rq->last_load_update_tick = curr_jiffies;
+               /*
+                * We were idle, this means load 0, the current load might be
+                * !0 due to remote wakeups and the sort.
+                */
+               __update_cpu_load(this_rq, 0, pending_updates);
+       }
+       raw_spin_unlock(&this_rq->lock);
+}
+#endif /* CONFIG_NO_HZ */
+
  /*
   * Called from scheduler_tick()
   */
  static void update_cpu_load_active(struct rq *this_rq)
  {
         /*
-        * See the mess in update_idle_cpu_load().
+        * See the mess around update_idle_cpu_load() / update_cpu_load_nohz().
          */
         this_rq->last_load_update_tick = jiffies;
         __update_cpu_load(this_rq, this_rq->load.weight, 1);
@@ -4325,6 +4388,20 @@ void thread_group_times(struct task_struct *p, cputime_t *ut, cputime_t *st)
  # define nsecs_to_cputime(__nsecs)     nsecs_to_jiffies(__nsecs)
  #endif
  
+static cputime_t scale_utime(cputime_t utime, cputime_t rtime, cputime_t total)
+{
+       u64 temp = (__force u64) rtime;
+
+       temp *= (__force u64) utime;
+
+       if (sizeof(cputime_t) == 4)
+               temp = div_u64(temp, (__force u32) total);
+       else
+               temp = div64_u64(temp, (__force u64) total);
+
+       return (__force cputime_t) temp;
+}
+
  void task_times(struct task_struct *p, cputime_t *ut, cputime_t *st)
  {
         cputime_t rtime, utime = p->utime, total = cputime_add(utime, p->stime);
@@ -4334,13 +4411,9 @@ void task_times(struct task_struct *p, cputime_t *ut, cputime_t *st)
          */
         rtime = nsecs_to_cputime(p->se.sum_exec_runtime);
  
-       if (total) {
-               u64 temp = rtime;
-
-               temp *= utime;
-               do_div(temp, total);
-               utime = (cputime_t)temp;
-       } else
+       if (total)
+               utime = scale_utime(utime, rtime, total);
+       else
                 utime = rtime;
  
         /*
@@ -4367,13 +4440,9 @@ void thread_group_times(struct task_struct *p, cputime_t *ut, cputime_t *st)
         total = cputime_add(cputime.utime, cputime.stime);
         rtime = nsecs_to_cputime(cputime.sum_exec_runtime);
  
-       if (total) {
-               u64 temp = rtime;
-
-               temp *= cputime.utime;
-               do_div(temp, total);
-               utime = (cputime_t)temp;
-       } else
+       if (total)
+               utime = scale_utime(cputime.utime, rtime, total);
+       else
                 utime = rtime;
  
         sig->prev_utime = max(sig->prev_utime, utime);
@@ -5181,8 +5250,11 @@ void rt_mutex_setprio(struct task_struct *p, int prio)
  
         if (rt_prio(prio))
                 p->sched_class = &rt_sched_class;
-       else
+       else {
+               if (rt_prio(oldprio))
+                       p->rt.timeout = 0;
                 p->sched_class = &fair_sched_class;
+       }
  
         p->prio = prio;
  
@@ -6197,14 +6269,16 @@ void show_state_filter(unsigned long state_filter)
                 /*
                  * reset the NMI-timeout, listing all files on a slow
                  * console might take a lot of time:
+                * Also, reset softlockup watchdogs on all CPUs, because
+                * another CPU might be blocked waiting for us to process
+                * an IPI.
                  */
                 touch_nmi_watchdog();
+               touch_all_softlockup_watchdogs();
                 if (!state_filter || (p->state & state_filter))
                         sched_show_task(p);
         } while_each_thread(g, p);
  
-       touch_all_softlockup_watchdogs();
-
  #ifdef CONFIG_SCHED_DEBUG
         sysrq_sched_debug_show();
  #endif
@@ -6527,8 +6601,6 @@ static void unthrottle_offline_cfs_rqs(struct rq *rq)
                         unthrottle_cfs_rq(cfs_rq);
         }
  }
-#else
-static void unthrottle_offline_cfs_rqs(struct rq *rq) {}
  #endif
  
  /*
@@ -6556,9 +6628,6 @@ static void migrate_tasks(unsigned int dead_cpu)
          */
         rq->stop = NULL;
  
-       /* Ensure any throttled groups are reachable by pick_next_task */
-       unthrottle_offline_cfs_rqs(rq);
-
         for ( ; ; ) {
                 /*
                  * There's this thread running, bail when that's the only
@@ -6585,6 +6654,10 @@ static void migrate_tasks(unsigned int dead_cpu)
  
  #endif /* CONFIG_HOTPLUG_CPU */
  
+#if !defined(CONFIG_HOTPLUG_CPU) || !defined(CONFIG_CFS_BANDWIDTH)
+static void unthrottle_offline_cfs_rqs(struct rq *rq) {}
+#endif
+
  #if defined(CONFIG_SCHED_DEBUG) && defined(CONFIG_SYSCTL)
  
  static struct ctl_table sd_ctl_dir[] = {
@@ -6633,16 +6706,25 @@ static void sd_free_ctl_entry(struct ctl_table **tablep)
         *tablep = NULL;
  }
  
+static int min_load_idx = 0;
+static int max_load_idx = CPU_LOAD_IDX_MAX-1;
+
  static void
  set_table_entry(struct ctl_table *entry,
                 const char *procname, void *data, int maxlen,
-               mode_t mode, proc_handler *proc_handler)
+               mode_t mode, proc_handler *proc_handler,
+               bool load_idx)
  {
         entry->procname = procname;
         entry->data = data;
         entry->maxlen = maxlen;
         entry->mode = mode;
         entry->proc_handler = proc_handler;
+
+       if (load_idx) {
+               entry->extra1 = &min_load_idx;
+               entry->extra2 = &max_load_idx;
+       }
  }
  
  static struct ctl_table *
@@ -6654,30 +6736,30 @@ sd_alloc_ctl_domain_table(struct sched_domain *sd)
                 return NULL;
  
         set_table_entry(&table[0], "min_interval", &sd->min_interval,
-               sizeof(long), 0644, proc_doulongvec_minmax);
+               sizeof(long), 0644, proc_doulongvec_minmax, false);
         set_table_entry(&table[1], "max_interval", &sd->max_interval,
-               sizeof(long), 0644, proc_doulongvec_minmax);
+               sizeof(long), 0644, proc_doulongvec_minmax, false);
         set_table_entry(&table[2], "busy_idx", &sd->busy_idx,
-               sizeof(int), 0644, proc_dointvec_minmax);
+               sizeof(int), 0644, proc_dointvec_minmax, true);
         set_table_entry(&table[3], "idle_idx", &sd->idle_idx,
-               sizeof(int), 0644, proc_dointvec_minmax);
+               sizeof(int), 0644, proc_dointvec_minmax, true);
         set_table_entry(&table[4], "newidle_idx", &sd->newidle_idx,
-               sizeof(int), 0644, proc_dointvec_minmax);
+               sizeof(int), 0644, proc_dointvec_minmax, true);
         set_table_entry(&table[5], "wake_idx", &sd->wake_idx,
-               sizeof(int), 0644, proc_dointvec_minmax);
+               sizeof(int), 0644, proc_dointvec_minmax, true);
         set_table_entry(&table[6], "forkexec_idx", &sd->forkexec_idx,
-               sizeof(int), 0644, proc_dointvec_minmax);
+               sizeof(int), 0644, proc_dointvec_minmax, true);
         set_table_entry(&table[7], "busy_factor", &sd->busy_factor,
-               sizeof(int), 0644, proc_dointvec_minmax);
+               sizeof(int), 0644, proc_dointvec_minmax, false);
         set_table_entry(&table[8], "imbalance_pct", &sd->imbalance_pct,
-               sizeof(int), 0644, proc_dointvec_minmax);
+               sizeof(int), 0644, proc_dointvec_minmax, false);
         set_table_entry(&table[9], "cache_nice_tries",
                 &sd->cache_nice_tries,
-               sizeof(int), 0644, proc_dointvec_minmax);
+               sizeof(int), 0644, proc_dointvec_minmax, false);
         set_table_entry(&table[10], "flags", &sd->flags,
-               sizeof(int), 0644, proc_dointvec_minmax);
+               sizeof(int), 0644, proc_dointvec_minmax, false);
         set_table_entry(&table[11], "name", sd->name,
-               CORENAME_MAX_SIZE, 0444, proc_dostring);
+               CORENAME_MAX_SIZE, 0444, proc_dostring, false);
         /* &table[12] is terminator */
  
         return table;
@@ -7115,11 +7197,11 @@ static int init_rootdomain(struct root_domain *rd)
  {
         memset(rd, 0, sizeof(*rd));
  
-       if (!alloc_cpumask_var(&rd->span, GFP_KERNEL))
+       if (!zalloc_cpumask_var(&rd->span, GFP_KERNEL))
                 goto out;
-       if (!alloc_cpumask_var(&rd->online, GFP_KERNEL))
+       if (!zalloc_cpumask_var(&rd->online, GFP_KERNEL))
                 goto free_span;
-       if (!alloc_cpumask_var(&rd->rto_mask, GFP_KERNEL))
+       if (!zalloc_cpumask_var(&rd->rto_mask, GFP_KERNEL))
                 goto free_online;
  
         if (cpupri_init(&rd->cpupri) != 0)
@@ -8156,34 +8238,66 @@ int __init sched_create_sysfs_power_savings_entries(struct sysdev_class *cls)
  }
  #endif /* CONFIG_SCHED_MC || CONFIG_SCHED_SMT */
  
+static int num_cpus_frozen;    /* used to mark begin/end of suspend/resume */
+
  /*
   * Update cpusets according to cpu_active mask.  If cpusets are
   * disabled, cpuset_update_active_cpus() becomes a simple wrapper
   * around partition_sched_domains().
+ *
+ * If we come here as part of a suspend/resume, don't touch cpusets because we
+ * want to restore it back to its original state upon resume anyway.
   */
  static int cpuset_cpu_active(struct notifier_block *nfb, unsigned long action,
                              void *hcpu)
  {
-       switch (action & ~CPU_TASKS_FROZEN) {
+       switch (action) {
+       case CPU_ONLINE_FROZEN:
+       case CPU_DOWN_FAILED_FROZEN:
+
+               /*
+                * num_cpus_frozen tracks how many CPUs are involved in suspend
+                * resume sequence. As long as this is not the last online
+                * operation in the resume sequence, just build a single sched
+                * domain, ignoring cpusets.
+                */
+               num_cpus_frozen--;
+               if (likely(num_cpus_frozen)) {
+                       partition_sched_domains(1, NULL, NULL);
+                       break;
+               }
+
+               /*
+                * This is the last CPU online operation. So fall through and
+                * restore the original sched domains by considering the
+                * cpuset configurations.
+                */
+
         case CPU_ONLINE:
         case CPU_DOWN_FAILED:
                 cpuset_update_active_cpus();
-               return NOTIFY_OK;
+               break;
         default:
                 return NOTIFY_DONE;
         }
+       return NOTIFY_OK;
  }
  
  static int cpuset_cpu_inactive(struct notifier_block *nfb, unsigned long action,
                                void *hcpu)
  {
-       switch (action & ~CPU_TASKS_FROZEN) {
+       switch (action) {
         case CPU_DOWN_PREPARE:
                 cpuset_update_active_cpus();
-               return NOTIFY_OK;
+               break;
+       case CPU_DOWN_PREPARE_FROZEN:
+               num_cpus_frozen++;
+               partition_sched_domains(1, NULL, NULL);
+               break;
         default:
                 return NOTIFY_DONE;
         }
+       return NOTIFY_OK;
  }
  
  static int update_runtime(struct notifier_block *nfb,
@@ -8919,6 +9033,7 @@ void sched_destroy_group(struct task_group *tg)
   */
  void sched_move_task(struct task_struct *tsk)
  {
+       struct task_group *tg;
         int on_rq, running;
         unsigned long flags;
         struct rq *rq;
@@ -8933,6 +9048,12 @@ void sched_move_task(struct task_struct *tsk)
         if (unlikely(running))
                 tsk->sched_class->put_prev_task(rq, tsk);
  
+       tg = container_of(task_subsys_state_check(tsk, cpu_cgroup_subsys_id,
+                               lockdep_is_held(&tsk->sighand->siglock)),
+                         struct task_group, css);
+       tg = autogroup_task_group(tsk, tg);
+       tsk->sched_task_group = tg;
+
  #ifdef CONFIG_FAIR_GROUP_SCHED
         if (tsk->sched_class->task_move_group)
                 tsk->sched_class->task_move_group(tsk, on_rq);
@@ -9014,6 +9135,12 @@ static inline int tg_has_rt_tasks(struct task_group *tg)
  {
         struct task_struct *g, *p;
  
+       /*
+        * Autogroups do not have RT tasks; see autogroup_create().
+        */
+       if (task_group_is_autogroup(tg))
+               return 0;
+
         do_each_thread(g, p) {
                 if (rt_task(p) && rt_rq_of_se(&p->rt)->tg == tg)
                         return 1;