ipv4: avoid parallel route cache gc executions

[pandora-kernel.git] / kernel / cgroup.c
diff --git a/kernel/cgroup.c b/kernel/cgroup.c

index d9d5648..ffcf896 100644 (file)
--- a/kernel/cgroup.c
+++ b/kernel/cgroup.c
@@ -361,12 +361,20 @@ static void __put_css_set(struct css_set *cg, int taskexit)
                 struct cgroup *cgrp = link->cgrp;
                 list_del(&link->cg_link_list);
                 list_del(&link->cgrp_link_list);
+
+               /*
+                * We may not be holding cgroup_mutex, and if cgrp->count is
+                * dropped to 0 the cgroup can be destroyed at any time, hence
+                * rcu_read_lock is used to keep it alive.
+                */
+               rcu_read_lock();
                 if (atomic_dec_and_test(&cgrp->count) &&
                     notify_on_release(cgrp)) {
                         if (taskexit)
                                 set_bit(CGRP_RELEASABLE, &cgrp->flags);
                         check_for_release(cgrp);
                 }
+               rcu_read_unlock();
  
                 kfree(link);
         }
@@ -1175,10 +1183,10 @@ static int parse_cgroupfs_options(char *data, struct cgroup_sb_opts *opts)
  
         /*
          * If the 'all' option was specified select all the subsystems,
-        * otherwise 'all, 'none' and a subsystem name options were not
-        * specified, let's default to 'all'
+        * otherwise if 'none', 'name=' and a subsystem name options
+        * were not specified, let's default to 'all'
          */
-       if (all_ss || (!all_ss && !one_ss && !opts->none)) {
+       if (all_ss || (!one_ss && !opts->none && !opts->name)) {
                 for (i = 0; i < CGROUP_SUBSYS_COUNT; i++) {
                         struct cgroup_subsys *ss = subsys[i];
                         if (ss == NULL)
@@ -1803,9 +1811,8 @@ static int cgroup_task_migrate(struct cgroup *cgrp, struct cgroup *oldcgrp,
          * trading it for newcg is protected by cgroup_mutex, we're safe to drop
          * it here; it will be freed under RCU.
          */
-       put_css_set(oldcg);
-
         set_bit(CGRP_RELEASABLE, &oldcgrp->flags);
+       put_css_set(oldcg);
         return 0;
  }
  
@@ -2022,7 +2029,7 @@ int cgroup_attach_proc(struct cgroup *cgrp, struct task_struct *leader)
         if (!group)
                 return -ENOMEM;
         /* pre-allocate to guarantee space while iterating in rcu read-side. */
-       retval = flex_array_prealloc(group, 0, group_size - 1, GFP_KERNEL);
+       retval = flex_array_prealloc(group, 0, group_size, GFP_KERNEL);
         if (retval)
                 goto out_free_group_list;
  
@@ -2098,11 +2105,6 @@ int cgroup_attach_proc(struct cgroup *cgrp, struct task_struct *leader)
                         continue;
                 /* get old css_set pointer */
                 task_lock(tsk);
-               if (tsk->flags & PF_EXITING) {
-                       /* ignore this task if it's going away */
-                       task_unlock(tsk);
-                       continue;
-               }
                 oldcg = tsk->cgroups;
                 get_css_set(oldcg);
                 task_unlock(tsk);
@@ -2642,9 +2644,7 @@ static int cgroup_create_dir(struct cgroup *cgrp, struct dentry *dentry,
                 dentry->d_fsdata = cgrp;
                 inc_nlink(parent->d_inode);
                 rcu_assign_pointer(cgrp->dentry, dentry);
-               dget(dentry);
         }
-       dput(dentry);
  
         return error;
  }
@@ -2785,9 +2785,14 @@ static void cgroup_enable_task_cg_lists(void)
                  * We should check if the process is exiting, otherwise
                  * it will race with cgroup_exit() in that the list
                  * entry won't be deleted though the process has exited.
+                * Do it while holding siglock so that we don't end up
+                * racing against cgroup_exit().
                  */
+               spin_lock_irq(&p->sighand->siglock);
                 if (!(p->flags & PF_EXITING) && list_empty(&p->cg_list))
                         list_add(&p->cg_list, &p->cgroups->tasks);
+               spin_unlock_irq(&p->sighand->siglock);
+
                 task_unlock(p);
         } while_each_thread(g, p);
         write_unlock(&css_set_lock);
@@ -3504,6 +3509,7 @@ static int cgroup_write_event_control(struct cgroup *cgrp, struct cftype *cft,
                                       const char *buffer)
  {
         struct cgroup_event *event = NULL;
+       struct cgroup *cgrp_cfile;
         unsigned int efd, cfd;
         struct file *efile = NULL;
         struct file *cfile = NULL;
@@ -3559,6 +3565,16 @@ static int cgroup_write_event_control(struct cgroup *cgrp, struct cftype *cft,
                 goto fail;
         }
  
+       /*
+        * The file to be monitored must be in the same cgroup as
+        * cgroup.event_control is.
+        */
+       cgrp_cfile = __d_cgrp(cfile->f_dentry->d_parent);
+       if (cgrp_cfile != cgrp) {
+               ret = -EINVAL;
+               goto fail;
+       }
+
         if (!event->cft->register_event || !event->cft->unregister_event) {
                 ret = -EINVAL;
                 goto fail;
@@ -3855,6 +3871,11 @@ static int cgroup_mkdir(struct inode *dir, struct dentry *dentry, int mode)
  {
         struct cgroup *c_parent = dentry->d_parent->d_fsdata;
  
+       /* Do not accept '\n' to prevent making /proc/<pid>/cgroup unparsable.
+        */
+       if (strchr(dentry->d_name.name, '\n'))
+               return -EINVAL;
+
         /* the vfs holds inode->i_mutex already */
         return cgroup_create(c_parent, dentry, mode | S_IFDIR);
  }
@@ -4513,42 +4534,20 @@ void cgroup_fork(struct task_struct *child)
         INIT_LIST_HEAD(&child->cg_list);
  }
  
-/**
- * cgroup_fork_callbacks - run fork callbacks
- * @child: the new task
- *
- * Called on a new task very soon before adding it to the
- * tasklist. No need to take any locks since no-one can
- * be operating on this task.
- */
-void cgroup_fork_callbacks(struct task_struct *child)
-{
-       if (need_forkexit_callback) {
-               int i;
-               /*
-                * forkexit callbacks are only supported for builtin
-                * subsystems, and the builtin section of the subsys array is
-                * immutable, so we don't need to lock the subsys array here.
-                */
-               for (i = 0; i < CGROUP_BUILTIN_SUBSYS_COUNT; i++) {
-                       struct cgroup_subsys *ss = subsys[i];
-                       if (ss->fork)
-                               ss->fork(ss, child);
-               }
-       }
-}
-
  /**
   * cgroup_post_fork - called on a new task after adding it to the task list
   * @child: the task in question
   *
- * Adds the task to the list running through its css_set if necessary.
- * Has to be after the task is visible on the task list in case we race
- * with the first call to cgroup_iter_start() - to guarantee that the
- * new task ends up on its list.
+ * Adds the task to the list running through its css_set if necessary and
+ * call the subsystem fork() callbacks.  Has to be after the task is
+ * visible on the task list in case we race with the first call to
+ * cgroup_iter_start() - to guarantee that the new task ends up on its
+ * list.
   */
  void cgroup_post_fork(struct task_struct *child)
  {
+       int i;
+
         if (use_task_css_set_links) {
                 write_lock(&css_set_lock);
                 task_lock(child);
@@ -4557,7 +4556,21 @@ void cgroup_post_fork(struct task_struct *child)
                 task_unlock(child);
                 write_unlock(&css_set_lock);
         }
+
+       /*
+        * Call ss->fork().  This must happen after @child is linked on
+        * css_set; otherwise, @child might change state between ->fork()
+        * and addition to css_set.
+        */
+       if (need_forkexit_callback) {
+               for (i = 0; i < CGROUP_BUILTIN_SUBSYS_COUNT; i++) {
+                       struct cgroup_subsys *ss = subsys[i];
+                       if (ss->fork)
+                               ss->fork(ss, child);
+               }
+       }
  }
+
  /**
   * cgroup_exit - detach cgroup from exiting task
   * @tsk: pointer to task_struct of exiting process