sched: remove PREEMPT_RESTRICT

[powerpc.git] / kernel / sched.c
diff --git a/kernel/sched.c b/kernel/sched.c

index b4fbbc4..2a107e4 100644 (file)
--- a/kernel/sched.c
+++ b/kernel/sched.c
@@ -75,7 +75,7 @@
   */
  unsigned long long __attribute__((weak)) sched_clock(void)
  {
-       return (unsigned long long)jiffies * (1000000000 / HZ);
+       return (unsigned long long)jiffies * (NSEC_PER_SEC / HZ);
  }
  
  /*
@@ -99,8 +99,8 @@ unsigned long long __attribute__((weak)) sched_clock(void)
  /*
   * Some helpers for converting nanosecond timing to jiffy resolution
   */
-#define NS_TO_JIFFIES(TIME)    ((unsigned long)(TIME) / (1000000000 / HZ))
-#define JIFFIES_TO_NS(TIME)    ((TIME) * (1000000000 / HZ))
+#define NS_TO_JIFFIES(TIME)    ((unsigned long)(TIME) / (NSEC_PER_SEC / HZ))
+#define JIFFIES_TO_NS(TIME)    ((TIME) * (NSEC_PER_SEC / HZ))
  
  #define NICE_0_LOAD            SCHED_LOAD_SCALE
  #define NICE_0_SHIFT           SCHED_LOAD_SHIFT
@@ -172,6 +172,7 @@ struct task_group {
         unsigned long shares;
         /* spinlock to serialize modification to shares */
         spinlock_t lock;
+       struct rcu_head rcu;
  };
  
  /* Default task group's sched entity on each cpu */
@@ -258,7 +259,6 @@ struct cfs_rq {
          */
         struct list_head leaf_cfs_rq_list; /* Better name : task_cfs_rq_list? */
         struct task_group *tg;    /* group that "owns" this runqueue */
-       struct rcu_head rcu;
  #endif
  };
  
@@ -460,7 +460,6 @@ enum {
         SCHED_FEAT_TREE_AVG             = 4,
         SCHED_FEAT_APPROX_AVG           = 8,
         SCHED_FEAT_WAKEUP_PREEMPT       = 16,
-       SCHED_FEAT_PREEMPT_RESTRICT     = 32,
  };
  
  const_debug unsigned int sysctl_sched_features =
@@ -468,8 +467,7 @@ const_debug unsigned int sysctl_sched_features =
                 SCHED_FEAT_START_DEBIT          * 1 |
                 SCHED_FEAT_TREE_AVG             * 0 |
                 SCHED_FEAT_APPROX_AVG           * 0 |
-               SCHED_FEAT_WAKEUP_PREEMPT       * 1 |
-               SCHED_FEAT_PREEMPT_RESTRICT     * 1;
+               SCHED_FEAT_WAKEUP_PREEMPT       * 1;
  
  #define sched_feat(x) (sysctl_sched_features & SCHED_FEAT_##x)
  
@@ -3355,7 +3353,7 @@ void account_user_time(struct task_struct *p, cputime_t cputime)
   * @p: the process that the cpu time gets accounted to
   * @cputime: the cpu time spent in virtual machine since the last update
   */
-void account_guest_time(struct task_struct *p, cputime_t cputime)
+static void account_guest_time(struct task_struct *p, cputime_t cputime)
  {
         cputime64_t tmp;
         struct cpu_usage_stat *cpustat = &kstat_this_cpu.cpustat;
@@ -4992,6 +4990,32 @@ void __cpuinit init_idle(struct task_struct *idle, int cpu)
   */
  cpumask_t nohz_cpu_mask = CPU_MASK_NONE;
  
+/*
+ * Increase the granularity value when there are more CPUs,
+ * because with more CPUs the 'effective latency' as visible
+ * to users decreases. But the relationship is not linear,
+ * so pick a second-best guess by going with the log2 of the
+ * number of CPUs.
+ *
+ * This idea comes from the SD scheduler of Con Kolivas:
+ */
+static inline void sched_init_granularity(void)
+{
+       unsigned int factor = 1 + ilog2(num_online_cpus());
+       const unsigned long limit = 200000000;
+
+       sysctl_sched_min_granularity *= factor;
+       if (sysctl_sched_min_granularity > limit)
+               sysctl_sched_min_granularity = limit;
+
+       sysctl_sched_latency *= factor;
+       if (sysctl_sched_latency > limit)
+               sysctl_sched_latency = limit;
+
+       sysctl_sched_wakeup_granularity *= factor;
+       sysctl_sched_batch_wakeup_granularity *= factor;
+}
+
  #ifdef CONFIG_SMP
  /*
   * This is how migration works:
@@ -5365,7 +5389,7 @@ static struct ctl_table sd_ctl_dir[] = {
                 .procname       = "sched_domain",
                 .mode           = 0555,
         },
-       {0,},
+       {0, },
  };
  
  static struct ctl_table sd_ctl_root[] = {
@@ -5375,7 +5399,7 @@ static struct ctl_table sd_ctl_root[] = {
                 .mode           = 0555,
                 .child          = sd_ctl_dir,
         },
-       {0,},
+       {0, },
  };
  
  static struct ctl_table *sd_alloc_ctl_entry(int n)
@@ -6688,10 +6712,12 @@ void __init sched_init_smp(void)
         /* Move init over to a non-isolated CPU */
         if (set_cpus_allowed(current, non_isolated_cpus) < 0)
                 BUG();
+       sched_init_granularity();
  }
  #else
  void __init sched_init_smp(void)
  {
+       sched_init_granularity();
  }
  #endif /* CONFIG_SMP */
  
@@ -7019,8 +7045,8 @@ err:
  /* rcu callback to free various structures associated with a task group */
  static void free_sched_group(struct rcu_head *rhp)
  {
-       struct cfs_rq *cfs_rq = container_of(rhp, struct cfs_rq, rcu);
-       struct task_group *tg = cfs_rq->tg;
+       struct task_group *tg = container_of(rhp, struct task_group, rcu);
+       struct cfs_rq *cfs_rq;
         struct sched_entity *se;
         int i;
  
@@ -7041,7 +7067,7 @@ static void free_sched_group(struct rcu_head *rhp)
  /* Destroy runqueue etc associated with a task group */
  void sched_destroy_group(struct task_group *tg)
  {
-       struct cfs_rq *cfs_rq;
+       struct cfs_rq *cfs_rq = NULL;
         int i;
  
         for_each_possible_cpu(i) {
@@ -7049,10 +7075,10 @@ void sched_destroy_group(struct task_group *tg)
                 list_del_rcu(&cfs_rq->leaf_cfs_rq_list);
         }
  
-       cfs_rq = tg->cfs_rq[0];
+       BUG_ON(!cfs_rq);
  
         /* wait for possible concurrent references to cfs_rqs complete */
-       call_rcu(&cfs_rq->rcu, free_sched_group);
+       call_rcu(&tg->rcu, free_sched_group);
  }
  
  /* change task's runqueue when it moves between groups.
@@ -7211,25 +7237,53 @@ static u64 cpu_shares_read_uint(struct cgroup *cgrp, struct cftype *cft)
         return (u64) tg->shares;
  }
  
-static struct cftype cpu_shares = {
-       .name = "shares",
-       .read_uint = cpu_shares_read_uint,
-       .write_uint = cpu_shares_write_uint,
+static u64 cpu_usage_read(struct cgroup *cgrp, struct cftype *cft)
+{
+       struct task_group *tg = cgroup_tg(cgrp);
+       unsigned long flags;
+       u64 res = 0;
+       int i;
+
+       for_each_possible_cpu(i) {
+               /*
+                * Lock to prevent races with updating 64-bit counters
+                * on 32-bit arches.
+                */
+               spin_lock_irqsave(&cpu_rq(i)->lock, flags);
+               res += tg->se[i]->sum_exec_runtime;
+               spin_unlock_irqrestore(&cpu_rq(i)->lock, flags);
+       }
+       /* Convert from ns to ms */
+       do_div(res, NSEC_PER_MSEC);
+
+       return res;
+}
+
+static struct cftype cpu_files[] = {
+       {
+               .name = "shares",
+               .read_uint = cpu_shares_read_uint,
+               .write_uint = cpu_shares_write_uint,
+       },
+       {
+               .name = "usage",
+               .read_uint = cpu_usage_read,
+       },
  };
  
  static int cpu_cgroup_populate(struct cgroup_subsys *ss, struct cgroup *cont)
  {
-       return cgroup_add_file(cont, ss, &cpu_shares);
+       return cgroup_add_files(cont, ss, cpu_files, ARRAY_SIZE(cpu_files));
  }
  
  struct cgroup_subsys cpu_cgroup_subsys = {
-       .name           = "cpu",
-       .create         = cpu_cgroup_create,
-       .destroy        = cpu_cgroup_destroy,
-       .can_attach     = cpu_cgroup_can_attach,
-       .attach         = cpu_cgroup_attach,
-       .populate       = cpu_cgroup_populate,
-       .subsys_id      = cpu_cgroup_subsys_id,
+       .name           = "cpu",
+       .create         = cpu_cgroup_create,
+       .destroy        = cpu_cgroup_destroy,
+       .can_attach     = cpu_cgroup_can_attach,
+       .attach         = cpu_cgroup_attach,
+       .populate       = cpu_cgroup_populate,
+       .subsys_id      = cpu_cgroup_subsys_id,
         .early_init     = 1,
  };