mfd: twl4030: fix ELF section mismatch...

[pandora-kernel.git] / kernel / cgroup.c
diff --git a/kernel/cgroup.c b/kernel/cgroup.c

index 8ba6809..ca83b73 100644 (file)
--- a/kernel/cgroup.c
+++ b/kernel/cgroup.c
@@ -49,6 +49,8 @@
  #include <linux/namei.h>
  #include <linux/smp_lock.h>
  #include <linux/pid_namespace.h>
+#include <linux/idr.h>
+#include <linux/vmalloc.h> /* TODO: replace with more sophisticated array */
  
  #include <asm/atomic.h>
  
@@ -77,6 +79,9 @@ struct cgroupfs_root {
          */
         unsigned long subsys_bits;
  
+       /* Unique id for this hierarchy. */
+       int hierarchy_id;
+
         /* The bitmask of subsystems currently attached to this hierarchy */
         unsigned long actual_subsys_bits;
  
@@ -147,6 +152,10 @@ struct css_id {
  static LIST_HEAD(roots);
  static int root_count;
  
+static DEFINE_IDA(hierarchy_ida);
+static int next_hierarchy_id;
+static DEFINE_SPINLOCK(hierarchy_id_lock);
+
  /* dummytop is a shorthand for the dummy hierarchy's top cgroup */
  #define dummytop (&rootnode.top_cgroup)
  
@@ -258,48 +267,22 @@ static struct hlist_head *css_set_hash(struct cgroup_subsys_state *css[])
         return &css_set_table[index];
  }
  
+static void free_css_set_rcu(struct rcu_head *obj)
+{
+       struct css_set *cg = container_of(obj, struct css_set, rcu_head);
+       kfree(cg);
+}
+
  /* We don't maintain the lists running through each css_set to its
   * task until after the first call to cgroup_iter_start(). This
   * reduces the fork()/exit() overhead for people who have cgroups
   * compiled into their kernel but not actually in use */
  static int use_task_css_set_links __read_mostly;
  
-/* When we create or destroy a css_set, the operation simply
- * takes/releases a reference count on all the cgroups referenced
- * by subsystems in this css_set. This can end up multiple-counting
- * some cgroups, but that's OK - the ref-count is just a
- * busy/not-busy indicator; ensuring that we only count each cgroup
- * once would require taking a global lock to ensure that no
- * subsystems moved between hierarchies while we were doing so.
- *
- * Possible TODO: decide at boot time based on the number of
- * registered subsystems and the number of CPUs or NUMA nodes whether
- * it's better for performance to ref-count every subsystem, or to
- * take a global lock and only add one ref count to each hierarchy.
- */
-
-/*
- * unlink a css_set from the list and free it
- */
-static void unlink_css_set(struct css_set *cg)
+static void __put_css_set(struct css_set *cg, int taskexit)
  {
         struct cg_cgroup_link *link;
         struct cg_cgroup_link *saved_link;
-
-       hlist_del(&cg->hlist);
-       css_set_count--;
-
-       list_for_each_entry_safe(link, saved_link, &cg->cg_links,
-                                cg_link_list) {
-               list_del(&link->cg_link_list);
-               list_del(&link->cgrp_link_list);
-               kfree(link);
-       }
-}
-
-static void __put_css_set(struct css_set *cg, int taskexit)
-{
-       int i;
         /*
          * Ensure that the refcount doesn't hit zero while any readers
          * can see it. Similar to atomic_dec_and_lock(), but for an
@@ -312,21 +295,28 @@ static void __put_css_set(struct css_set *cg, int taskexit)
                 write_unlock(&css_set_lock);
                 return;
         }
-       unlink_css_set(cg);
-       write_unlock(&css_set_lock);
  
-       rcu_read_lock();
-       for (i = 0; i < CGROUP_SUBSYS_COUNT; i++) {
-               struct cgroup *cgrp = rcu_dereference(cg->subsys[i]->cgroup);
+       /* This css_set is dead. unlink it and release cgroup refcounts */
+       hlist_del(&cg->hlist);
+       css_set_count--;
+
+       list_for_each_entry_safe(link, saved_link, &cg->cg_links,
+                                cg_link_list) {
+               struct cgroup *cgrp = link->cgrp;
+               list_del(&link->cg_link_list);
+               list_del(&link->cgrp_link_list);
                 if (atomic_dec_and_test(&cgrp->count) &&
                     notify_on_release(cgrp)) {
                         if (taskexit)
                                 set_bit(CGRP_RELEASABLE, &cgrp->flags);
                         check_for_release(cgrp);
                 }
+
+               kfree(link);
         }
-       rcu_read_unlock();
-       kfree(cg);
+
+       write_unlock(&css_set_lock);
+       call_rcu(&cg->rcu_head, free_css_set_rcu);
  }
  
  /*
@@ -519,6 +509,7 @@ static void link_css_set(struct list_head *tmp_cg_links,
                                 cgrp_link_list);
         link->cg = cg;
         link->cgrp = cgrp;
+       atomic_inc(&cgrp->count);
         list_move(&link->cgrp_link_list, &cgrp->css_sets);
         /*
          * Always add links to the tail of the list so that the list
@@ -539,7 +530,6 @@ static struct css_set *find_css_set(
  {
         struct css_set *res;
         struct cgroup_subsys_state *template[CGROUP_SUBSYS_COUNT];
-       int i;
  
         struct list_head tmp_cg_links;
  
@@ -578,10 +568,6 @@ static struct css_set *find_css_set(
  
         write_lock(&css_set_lock);
         /* Add reference counts and links from the new css_set. */
-       for (i = 0; i < CGROUP_SUBSYS_COUNT; i++) {
-               struct cgroup *cgrp = res->subsys[i]->cgroup;
-               atomic_inc(&cgrp->count);
-       }
         list_for_each_entry(link, &oldcg->cg_links, cg_link_list) {
                 struct cgroup *c = link->cgrp;
                 if (c->root == cgrp->root)
@@ -717,7 +703,7 @@ static int cgroup_mkdir(struct inode *dir, struct dentry *dentry, int mode);
  static int cgroup_rmdir(struct inode *unused_dir, struct dentry *dentry);
  static int cgroup_populate_dir(struct cgroup *cgrp);
  static const struct inode_operations cgroup_dir_inode_operations;
-static struct file_operations proc_cgroupstats_operations;
+static const struct file_operations proc_cgroupstats_operations;
  
  static struct backing_dev_info cgroup_backing_dev_info = {
         .name           = "cgroup",
@@ -797,6 +783,12 @@ static void cgroup_diput(struct dentry *dentry, struct inode *inode)
                  */
                 deactivate_super(cgrp->root->sb);
  
+               /*
+                * if we're getting rid of the cgroup, refcount should ensure
+                * that there are no pidlists left.
+                */
+               BUG_ON(!list_empty(&cgrp->pidlists));
+
                 call_rcu(&cgrp->rcu_head, free_cgroup_rcu);
         }
         iput(inode);
@@ -972,8 +964,11 @@ struct cgroup_sb_opts {
         unsigned long flags;
         char *release_agent;
         char *name;
+       /* User explicitly requested empty subsystem */
+       bool none;
  
         struct cgroupfs_root *new_root;
+
  };
  
  /* Convert a hierarchy specifier into a bitmask of subsystems and
@@ -1002,6 +997,9 @@ static int parse_cgroupfs_options(char *data,
                                 if (!ss->disabled)
                                         opts->subsys_bits |= 1ul << i;
                         }
+               } else if (!strcmp(token, "none")) {
+                       /* Explicitly have no subsystems */
+                       opts->none = true;
                 } else if (!strcmp(token, "noprefix")) {
                         set_bit(ROOT_NOPREFIX, &opts->flags);
                 } else if (!strncmp(token, "release_agent=", 14)) {
@@ -1051,6 +1049,8 @@ static int parse_cgroupfs_options(char *data,
                 }
         }
  
+       /* Consistency checks */
+
         /*
          * Option noprefix was introduced just for backward compatibility
          * with the old cpuset, so we allow noprefix only if mounting just
@@ -1060,7 +1060,15 @@ static int parse_cgroupfs_options(char *data,
             (opts->subsys_bits & mask))
                 return -EINVAL;
  
-       /* We can't have an empty hierarchy */
+
+       /* Can't specify "none" and some subsystems */
+       if (opts->subsys_bits && opts->none)
+               return -EINVAL;
+
+       /*
+        * We either have to specify by name or by subsystems. (So all
+        * empty hierarchies must have a name).
+        */
         if (!opts->subsys_bits && !opts->name)
                 return -EINVAL;
  
@@ -1126,8 +1134,8 @@ static void init_cgroup_housekeeping(struct cgroup *cgrp)
         INIT_LIST_HEAD(&cgrp->children);
         INIT_LIST_HEAD(&cgrp->css_sets);
         INIT_LIST_HEAD(&cgrp->release_list);
-       INIT_LIST_HEAD(&cgrp->pids_list);
-       init_rwsem(&cgrp->pids_mutex);
+       INIT_LIST_HEAD(&cgrp->pidlists);
+       mutex_init(&cgrp->pidlist_mutex);
  }
  
  static void init_cgroup_root(struct cgroupfs_root *root)
@@ -1141,6 +1149,31 @@ static void init_cgroup_root(struct cgroupfs_root *root)
         init_cgroup_housekeeping(cgrp);
  }
  
+static bool init_root_id(struct cgroupfs_root *root)
+{
+       int ret = 0;
+
+       do {
+               if (!ida_pre_get(&hierarchy_ida, GFP_KERNEL))
+                       return false;
+               spin_lock(&hierarchy_id_lock);
+               /* Try to allocate the next unused ID */
+               ret = ida_get_new_above(&hierarchy_ida, next_hierarchy_id,
+                                       &root->hierarchy_id);
+               if (ret == -ENOSPC)
+                       /* Try again starting from 0 */
+                       ret = ida_get_new(&hierarchy_ida, &root->hierarchy_id);
+               if (!ret) {
+                       next_hierarchy_id = root->hierarchy_id + 1;
+               } else if (ret != -EAGAIN) {
+                       /* Can only get here if the 31-bit IDR is full ... */
+                       BUG_ON(ret);
+               }
+               spin_unlock(&hierarchy_id_lock);
+       } while (ret);
+       return true;
+}
+
  static int cgroup_test_super(struct super_block *sb, void *data)
  {
         struct cgroup_sb_opts *opts = data;
@@ -1150,8 +1183,12 @@ static int cgroup_test_super(struct super_block *sb, void *data)
         if (opts->name && strcmp(opts->name, root->name))
                 return 0;
  
-       /* If we asked for subsystems then they must match */
-       if (opts->subsys_bits && (opts->subsys_bits != root->subsys_bits))
+       /*
+        * If we asked for subsystems (or explicitly for no
+        * subsystems) then they must match
+        */
+       if ((opts->subsys_bits || opts->none)
+           && (opts->subsys_bits != root->subsys_bits))
                 return 0;
  
         return 1;
@@ -1161,15 +1198,19 @@ static struct cgroupfs_root *cgroup_root_from_opts(struct cgroup_sb_opts *opts)
  {
         struct cgroupfs_root *root;
  
-       /* Empty hierarchies aren't supported */
-       if (!opts->subsys_bits)
+       if (!opts->subsys_bits && !opts->none)
                 return NULL;
  
         root = kzalloc(sizeof(*root), GFP_KERNEL);
         if (!root)
                 return ERR_PTR(-ENOMEM);
  
+       if (!init_root_id(root)) {
+               kfree(root);
+               return ERR_PTR(-ENOMEM);
+       }
         init_cgroup_root(root);
+
         root->subsys_bits = opts->subsys_bits;
         root->flags = opts->flags;
         if (opts->release_agent)
@@ -1179,6 +1220,18 @@ static struct cgroupfs_root *cgroup_root_from_opts(struct cgroup_sb_opts *opts)
         return root;
  }
  
+static void cgroup_drop_root(struct cgroupfs_root *root)
+{
+       if (!root)
+               return;
+
+       BUG_ON(!root->hierarchy_id);
+       spin_lock(&hierarchy_id_lock);
+       ida_remove(&hierarchy_ida, root->hierarchy_id);
+       spin_unlock(&hierarchy_id_lock);
+       kfree(root);
+}
+
  static int cgroup_set_super(struct super_block *sb, void *data)
  {
         int ret;
@@ -1188,7 +1241,7 @@ static int cgroup_set_super(struct super_block *sb, void *data)
         if (!opts->new_root)
                 return -EINVAL;
  
-       BUG_ON(!opts->subsys_bits);
+       BUG_ON(!opts->subsys_bits && !opts->none);
  
         ret = set_anon_super(sb, NULL);
         if (ret)
@@ -1257,7 +1310,7 @@ static int cgroup_get_sb(struct file_system_type *fs_type,
         sb = sget(fs_type, cgroup_test_super, cgroup_set_super, &opts);
         if (IS_ERR(sb)) {
                 ret = PTR_ERR(sb);
-               kfree(opts.new_root);
+               cgroup_drop_root(opts.new_root);
                 goto out_err;
         }
  
@@ -1351,7 +1404,7 @@ static int cgroup_get_sb(struct file_system_type *fs_type,
                  * We re-used an existing hierarchy - the new root (if
                  * any) is not needed
                  */
-               kfree(opts.new_root);
+               cgroup_drop_root(opts.new_root);
         }
  
         simple_set_mnt(mnt, sb);
@@ -1410,7 +1463,7 @@ static void cgroup_kill_sb(struct super_block *sb) {
         mutex_unlock(&cgroup_mutex);
  
         kill_litter_super(sb);
-       kfree(root);
+       cgroup_drop_root(root);
  }
  
  static struct file_system_type cgroup_fs_type = {
@@ -1499,7 +1552,7 @@ int cgroup_attach_task(struct cgroup *cgrp, struct task_struct *tsk)
  
         for_each_subsys(root, ss) {
                 if (ss->can_attach) {
-                       retval = ss->can_attach(ss, cgrp, tsk);
+                       retval = ss->can_attach(ss, cgrp, tsk, false);
                         if (retval)
                                 return retval;
                 }
@@ -1537,7 +1590,7 @@ int cgroup_attach_task(struct cgroup *cgrp, struct task_struct *tsk)
  
         for_each_subsys(root, ss) {
                 if (ss->attach)
-                       ss->attach(ss, cgrp, oldcgrp, tsk);
+                       ss->attach(ss, cgrp, oldcgrp, tsk, false);
         }
         set_bit(CGRP_RELEASABLE, &oldcgrp->flags);
         synchronize_rcu();
@@ -1598,15 +1651,6 @@ static int cgroup_tasks_write(struct cgroup *cgrp, struct cftype *cft, u64 pid)
         return ret;
  }
  
-/* The various types of files and directories in a cgroup file system */
-enum cgroup_filetype {
-       FILE_ROOT,
-       FILE_DIR,
-       FILE_TASKLIST,
-       FILE_NOTIFY_ON_RELEASE,
-       FILE_RELEASE_AGENT,
-};
-
  /**
   * cgroup_lock_live_group - take cgroup_mutex and check that cgrp is alive.
   * @cgrp: the cgroup to be checked for liveness
@@ -1819,7 +1863,7 @@ static int cgroup_seqfile_release(struct inode *inode, struct file *file)
         return single_release(inode, file);
  }
  
-static struct file_operations cgroup_seqfile_operations = {
+static const struct file_operations cgroup_seqfile_operations = {
         .read = seq_read,
         .write = cgroup_file_write,
         .llseek = seq_lseek,
@@ -1878,7 +1922,7 @@ static int cgroup_rename(struct inode *old_dir, struct dentry *old_dentry,
         return simple_rename(old_dir, old_dentry, new_dir, new_dentry);
  }
  
-static struct file_operations cgroup_file_operations = {
+static const struct file_operations cgroup_file_operations = {
         .read = cgroup_file_read,
         .write = cgroup_file_write,
         .llseek = generic_file_llseek,
@@ -2304,7 +2348,7 @@ int cgroup_scan_tasks(struct cgroup_scanner *scan)
  }
  
  /*
- * Stuff for reading the 'tasks' file.
+ * Stuff for reading the 'tasks'/'procs' files.
   *
   * Reading this file can return large amounts of data if a cgroup has
   * *lots* of attached tasks. So it may need several calls to read(),
@@ -2314,27 +2358,196 @@ int cgroup_scan_tasks(struct cgroup_scanner *scan)
   */
  
  /*
- * Load into 'pidarray' up to 'npids' of the tasks using cgroup
- * 'cgrp'.  Return actual number of pids loaded.  No need to
- * task_lock(p) when reading out p->cgroup, since we're in an RCU
- * read section, so the css_set can't go away, and is
- * immutable after creation.
+ * The following two functions "fix" the issue where there are more pids
+ * than kmalloc will give memory for; in such cases, we use vmalloc/vfree.
+ * TODO: replace with a kernel-wide solution to this problem
+ */
+#define PIDLIST_TOO_LARGE(c) ((c) * sizeof(pid_t) > (PAGE_SIZE * 2))
+static void *pidlist_allocate(int count)
+{
+       if (PIDLIST_TOO_LARGE(count))
+               return vmalloc(count * sizeof(pid_t));
+       else
+               return kmalloc(count * sizeof(pid_t), GFP_KERNEL);
+}
+static void pidlist_free(void *p)
+{
+       if (is_vmalloc_addr(p))
+               vfree(p);
+       else
+               kfree(p);
+}
+static void *pidlist_resize(void *p, int newcount)
+{
+       void *newlist;
+       /* note: if new alloc fails, old p will still be valid either way */
+       if (is_vmalloc_addr(p)) {
+               newlist = vmalloc(newcount * sizeof(pid_t));
+               if (!newlist)
+                       return NULL;
+               memcpy(newlist, p, newcount * sizeof(pid_t));
+               vfree(p);
+       } else {
+               newlist = krealloc(p, newcount * sizeof(pid_t), GFP_KERNEL);
+       }
+       return newlist;
+}
+
+/*
+ * pidlist_uniq - given a kmalloc()ed list, strip out all duplicate entries
+ * If the new stripped list is sufficiently smaller and there's enough memory
+ * to allocate a new buffer, will let go of the unneeded memory. Returns the
+ * number of unique elements.
   */
-static int pid_array_load(pid_t *pidarray, int npids, struct cgroup *cgrp)
+/* is the size difference enough that we should re-allocate the array? */
+#define PIDLIST_REALLOC_DIFFERENCE(old, new) ((old) - PAGE_SIZE >= (new))
+static int pidlist_uniq(pid_t **p, int length)
  {
-       int n = 0, pid;
+       int src, dest = 1;
+       pid_t *list = *p;
+       pid_t *newlist;
+
+       /*
+        * we presume the 0th element is unique, so i starts at 1. trivial
+        * edge cases first; no work needs to be done for either
+        */
+       if (length == 0 || length == 1)
+               return length;
+       /* src and dest walk down the list; dest counts unique elements */
+       for (src = 1; src < length; src++) {
+               /* find next unique element */
+               while (list[src] == list[src-1]) {
+                       src++;
+                       if (src == length)
+                               goto after;
+               }
+               /* dest always points to where the next unique element goes */
+               list[dest] = list[src];
+               dest++;
+       }
+after:
+       /*
+        * if the length difference is large enough, we want to allocate a
+        * smaller buffer to save memory. if this fails due to out of memory,
+        * we'll just stay with what we've got.
+        */
+       if (PIDLIST_REALLOC_DIFFERENCE(length, dest)) {
+               newlist = pidlist_resize(list, dest);
+               if (newlist)
+                       *p = newlist;
+       }
+       return dest;
+}
+
+static int cmppid(const void *a, const void *b)
+{
+       return *(pid_t *)a - *(pid_t *)b;
+}
+
+/*
+ * find the appropriate pidlist for our purpose (given procs vs tasks)
+ * returns with the lock on that pidlist already held, and takes care
+ * of the use count, or returns NULL with no locks held if we're out of
+ * memory.
+ */
+static struct cgroup_pidlist *cgroup_pidlist_find(struct cgroup *cgrp,
+                                                 enum cgroup_filetype type)
+{
+       struct cgroup_pidlist *l;
+       /* don't need task_nsproxy() if we're looking at ourself */
+       struct pid_namespace *ns = get_pid_ns(current->nsproxy->pid_ns);
+       /*
+        * We can't drop the pidlist_mutex before taking the l->mutex in case
+        * the last ref-holder is trying to remove l from the list at the same
+        * time. Holding the pidlist_mutex precludes somebody taking whichever
+        * list we find out from under us - compare release_pid_array().
+        */
+       mutex_lock(&cgrp->pidlist_mutex);
+       list_for_each_entry(l, &cgrp->pidlists, links) {
+               if (l->key.type == type && l->key.ns == ns) {
+                       /* found a matching list - drop the extra refcount */
+                       put_pid_ns(ns);
+                       /* make sure l doesn't vanish out from under us */
+                       down_write(&l->mutex);
+                       mutex_unlock(&cgrp->pidlist_mutex);
+                       l->use_count++;
+                       return l;
+               }
+       }
+       /* entry not found; create a new one */
+       l = kmalloc(sizeof(struct cgroup_pidlist), GFP_KERNEL);
+       if (!l) {
+               mutex_unlock(&cgrp->pidlist_mutex);
+               put_pid_ns(ns);
+               return l;
+       }
+       init_rwsem(&l->mutex);
+       down_write(&l->mutex);
+       l->key.type = type;
+       l->key.ns = ns;
+       l->use_count = 0; /* don't increment here */
+       l->list = NULL;
+       l->owner = cgrp;
+       list_add(&l->links, &cgrp->pidlists);
+       mutex_unlock(&cgrp->pidlist_mutex);
+       return l;
+}
+
+/*
+ * Load a cgroup's pidarray with either procs' tgids or tasks' pids
+ */
+static int pidlist_array_load(struct cgroup *cgrp, enum cgroup_filetype type,
+                             struct cgroup_pidlist **lp)
+{
+       pid_t *array;
+       int length;
+       int pid, n = 0; /* used for populating the array */
         struct cgroup_iter it;
         struct task_struct *tsk;
+       struct cgroup_pidlist *l;
+
+       /*
+        * If cgroup gets more users after we read count, we won't have
+        * enough space - tough.  This race is indistinguishable to the
+        * caller from the case that the additional cgroup users didn't
+        * show up until sometime later on.
+        */
+       length = cgroup_task_count(cgrp);
+       array = pidlist_allocate(length);
+       if (!array)
+               return -ENOMEM;
+       /* now, populate the array */
         cgroup_iter_start(cgrp, &it);
         while ((tsk = cgroup_iter_next(cgrp, &it))) {
-               if (unlikely(n == npids))
+               if (unlikely(n == length))
                         break;
-               pid = task_pid_vnr(tsk);
-               if (pid > 0)
-                       pidarray[n++] = pid;
+               /* get tgid or pid for procs or tasks file respectively */
+               if (type == CGROUP_FILE_PROCS)
+                       pid = task_tgid_vnr(tsk);
+               else
+                       pid = task_pid_vnr(tsk);
+               if (pid > 0) /* make sure to only use valid results */
+                       array[n++] = pid;
         }
         cgroup_iter_end(cgrp, &it);
-       return n;
+       length = n;
+       /* now sort & (if procs) strip out duplicates */
+       sort(array, length, sizeof(pid_t), cmppid, NULL);
+       if (type == CGROUP_FILE_PROCS)
+               length = pidlist_uniq(&array, length);
+       l = cgroup_pidlist_find(cgrp, type);
+       if (!l) {
+               pidlist_free(array);
+               return -ENOMEM;
+       }
+       /* store array, freeing old if necessary - lock already held */
+       pidlist_free(l->list);
+       l->list = array;
+       l->length = length;
+       l->use_count++;
+       up_write(&l->mutex);
+       *lp = l;
+       return 0;
  }
  
  /**
@@ -2391,37 +2604,14 @@ err:
         return ret;
  }
  
-/*
- * Cache pids for all threads in the same pid namespace that are
- * opening the same "tasks" file.
- */
-struct cgroup_pids {
-       /* The node in cgrp->pids_list */
-       struct list_head list;
-       /* The cgroup those pids belong to */
-       struct cgroup *cgrp;
-       /* The namepsace those pids belong to */
-       struct pid_namespace *ns;
-       /* Array of process ids in the cgroup */
-       pid_t *tasks_pids;
-       /* How many files are using the this tasks_pids array */
-       int use_count;
-       /* Length of the current tasks_pids array */
-       int length;
-};
-
-static int cmppid(const void *a, const void *b)
-{
-       return *(pid_t *)a - *(pid_t *)b;
-}
  
  /*
- * seq_file methods for the "tasks" file. The seq_file position is the
+ * seq_file methods for the tasks/procs files. The seq_file position is the
   * next pid to display; the seq_file iterator is a pointer to the pid
- * in the cgroup->tasks_pids array.
+ * in the cgroup->l->list array.
   */
  
-static void *cgroup_tasks_start(struct seq_file *s, loff_t *pos)
+static void *cgroup_pidlist_start(struct seq_file *s, loff_t *pos)
  {
         /*
          * Initially we receive a position value that corresponds to
@@ -2429,48 +2619,45 @@ static void *cgroup_tasks_start(struct seq_file *s, loff_t *pos)
          * after a seek to the start). Use a binary-search to find the
          * next pid to display, if any
          */
-       struct cgroup_pids *cp = s->private;
-       struct cgroup *cgrp = cp->cgrp;
+       struct cgroup_pidlist *l = s->private;
         int index = 0, pid = *pos;
         int *iter;
  
-       down_read(&cgrp->pids_mutex);
+       down_read(&l->mutex);
         if (pid) {
-               int end = cp->length;
+               int end = l->length;
  
                 while (index < end) {
                         int mid = (index + end) / 2;
-                       if (cp->tasks_pids[mid] == pid) {
+                       if (l->list[mid] == pid) {
                                 index = mid;
                                 break;
-                       } else if (cp->tasks_pids[mid] <= pid)
+                       } else if (l->list[mid] <= pid)
                                 index = mid + 1;
                         else
                                 end = mid;
                 }
         }
         /* If we're off the end of the array, we're done */
-       if (index >= cp->length)
+       if (index >= l->length)
                 return NULL;
         /* Update the abstract position to be the actual pid that we found */
-       iter = cp->tasks_pids + index;
+       iter = l->list + index;
         *pos = *iter;
         return iter;
  }
  
-static void cgroup_tasks_stop(struct seq_file *s, void *v)
+static void cgroup_pidlist_stop(struct seq_file *s, void *v)
  {
-       struct cgroup_pids *cp = s->private;
-       struct cgroup *cgrp = cp->cgrp;
-       up_read(&cgrp->pids_mutex);
+       struct cgroup_pidlist *l = s->private;
+       up_read(&l->mutex);
  }
  
-static void *cgroup_tasks_next(struct seq_file *s, void *v, loff_t *pos)
+static void *cgroup_pidlist_next(struct seq_file *s, void *v, loff_t *pos)
  {
-       struct cgroup_pids *cp = s->private;
-       int *p = v;
-       int *end = cp->tasks_pids + cp->length;
-
+       struct cgroup_pidlist *l = s->private;
+       pid_t *p = v;
+       pid_t *end = l->list + l->length;
         /*
          * Advance to the next pid in the array. If this goes off the
          * end, we're done
@@ -2484,124 +2671,107 @@ static void *cgroup_tasks_next(struct seq_file *s, void *v, loff_t *pos)
         }
  }
  
-static int cgroup_tasks_show(struct seq_file *s, void *v)
+static int cgroup_pidlist_show(struct seq_file *s, void *v)
  {
         return seq_printf(s, "%d\n", *(int *)v);
  }
  
-static const struct seq_operations cgroup_tasks_seq_operations = {
-       .start = cgroup_tasks_start,
-       .stop = cgroup_tasks_stop,
-       .next = cgroup_tasks_next,
-       .show = cgroup_tasks_show,
+/*
+ * seq_operations functions for iterating on pidlists through seq_file -
+ * independent of whether it's tasks or procs
+ */
+static const struct seq_operations cgroup_pidlist_seq_operations = {
+       .start = cgroup_pidlist_start,
+       .stop = cgroup_pidlist_stop,
+       .next = cgroup_pidlist_next,
+       .show = cgroup_pidlist_show,
  };
  
-static void release_cgroup_pid_array(struct cgroup_pids *cp)
+static void cgroup_release_pid_array(struct cgroup_pidlist *l)
  {
-       struct cgroup *cgrp = cp->cgrp;
-
-       down_write(&cgrp->pids_mutex);
-       BUG_ON(!cp->use_count);
-       if (!--cp->use_count) {
-               list_del(&cp->list);
-               put_pid_ns(cp->ns);
-               kfree(cp->tasks_pids);
-               kfree(cp);
+       /*
+        * the case where we're the last user of this particular pidlist will
+        * have us remove it from the cgroup's list, which entails taking the
+        * mutex. since in pidlist_find the pidlist->lock depends on cgroup->
+        * pidlist_mutex, we have to take pidlist_mutex first.
+        */
+       mutex_lock(&l->owner->pidlist_mutex);
+       down_write(&l->mutex);
+       BUG_ON(!l->use_count);
+       if (!--l->use_count) {
+               /* we're the last user if refcount is 0; remove and free */
+               list_del(&l->links);
+               mutex_unlock(&l->owner->pidlist_mutex);
+               pidlist_free(l->list);
+               put_pid_ns(l->key.ns);
+               up_write(&l->mutex);
+               kfree(l);
+               return;
         }
-       up_write(&cgrp->pids_mutex);
+       mutex_unlock(&l->owner->pidlist_mutex);
+       up_write(&l->mutex);
  }
  
-static int cgroup_tasks_release(struct inode *inode, struct file *file)
+static int cgroup_pidlist_release(struct inode *inode, struct file *file)
  {
-       struct seq_file *seq;
-       struct cgroup_pids *cp;
-
+       struct cgroup_pidlist *l;
         if (!(file->f_mode & FMODE_READ))
                 return 0;
-
-       seq = file->private_data;
-       cp = seq->private;
-
-       release_cgroup_pid_array(cp);
+       /*
+        * the seq_file will only be initialized if the file was opened for
+        * reading; hence we check if it's not null only in that case.
+        */
+       l = ((struct seq_file *)file->private_data)->private;
+       cgroup_release_pid_array(l);
         return seq_release(inode, file);
  }
  
-static struct file_operations cgroup_tasks_operations = {
+static const struct file_operations cgroup_pidlist_operations = {
         .read = seq_read,
         .llseek = seq_lseek,
         .write = cgroup_file_write,
-       .release = cgroup_tasks_release,
+       .release = cgroup_pidlist_release,
  };
  
  /*
- * Handle an open on 'tasks' file.  Prepare an array containing the
- * process id's of tasks currently attached to the cgroup being opened.
+ * The following functions handle opens on a file that displays a pidlist
+ * (tasks or procs). Prepare an array of the process/thread IDs of whoever's
+ * in the cgroup.
   */
-
-static int cgroup_tasks_open(struct inode *unused, struct file *file)
+/* helper function for the two below it */
+static int cgroup_pidlist_open(struct file *file, enum cgroup_filetype type)
  {
         struct cgroup *cgrp = __d_cgrp(file->f_dentry->d_parent);
-       struct pid_namespace *ns = current->nsproxy->pid_ns;
-       struct cgroup_pids *cp;
-       pid_t *pidarray;
-       int npids;
+       struct cgroup_pidlist *l;
         int retval;
  
         /* Nothing to do for write-only files */
         if (!(file->f_mode & FMODE_READ))
                 return 0;
  
-       /*
-        * If cgroup gets more users after we read count, we won't have
-        * enough space - tough.  This race is indistinguishable to the
-        * caller from the case that the additional cgroup users didn't
-        * show up until sometime later on.
-        */
-       npids = cgroup_task_count(cgrp);
-       pidarray = kmalloc(npids * sizeof(pid_t), GFP_KERNEL);
-       if (!pidarray)
-               return -ENOMEM;
-       npids = pid_array_load(pidarray, npids, cgrp);
-       sort(pidarray, npids, sizeof(pid_t), cmppid, NULL);
-
-       /*
-        * Store the array in the cgroup, freeing the old
-        * array if necessary
-        */
-       down_write(&cgrp->pids_mutex);
-
-       list_for_each_entry(cp, &cgrp->pids_list, list) {
-               if (ns == cp->ns)
-                       goto found;
-       }
-
-       cp = kzalloc(sizeof(*cp), GFP_KERNEL);
-       if (!cp) {
-               up_write(&cgrp->pids_mutex);
-               kfree(pidarray);
-               return -ENOMEM;
-       }
-       cp->cgrp = cgrp;
-       cp->ns = ns;
-       get_pid_ns(ns);
-       list_add(&cp->list, &cgrp->pids_list);
-found:
-       kfree(cp->tasks_pids);
-       cp->tasks_pids = pidarray;
-       cp->length = npids;
-       cp->use_count++;
-       up_write(&cgrp->pids_mutex);
-
-       file->f_op = &cgroup_tasks_operations;
+       /* have the array populated */
+       retval = pidlist_array_load(cgrp, type, &l);
+       if (retval)
+               return retval;
+       /* configure file information */
+       file->f_op = &cgroup_pidlist_operations;
  
-       retval = seq_open(file, &cgroup_tasks_seq_operations);
+       retval = seq_open(file, &cgroup_pidlist_seq_operations);
         if (retval) {
-               release_cgroup_pid_array(cp);
+               cgroup_release_pid_array(l);
                 return retval;
         }
-       ((struct seq_file *)file->private_data)->private = cp;
+       ((struct seq_file *)file->private_data)->private = l;
         return 0;
  }
+static int cgroup_tasks_open(struct inode *unused, struct file *file)
+{
+       return cgroup_pidlist_open(file, CGROUP_FILE_TASKS);
+}
+static int cgroup_procs_open(struct inode *unused, struct file *file)
+{
+       return cgroup_pidlist_open(file, CGROUP_FILE_PROCS);
+}
  
  static u64 cgroup_read_notify_on_release(struct cgroup *cgrp,
                                             struct cftype *cft)
@@ -2624,21 +2794,27 @@ static int cgroup_write_notify_on_release(struct cgroup *cgrp,
  /*
   * for the common functions, 'private' gives the type of file
   */
+/* for hysterical raisins, we can't put this on the older files */
+#define CGROUP_FILE_GENERIC_PREFIX "cgroup."
  static struct cftype files[] = {
         {
                 .name = "tasks",
                 .open = cgroup_tasks_open,
                 .write_u64 = cgroup_tasks_write,
-               .release = cgroup_tasks_release,
-               .private = FILE_TASKLIST,
+               .release = cgroup_pidlist_release,
                 .mode = S_IRUGO | S_IWUSR,
         },
-
+       {
+               .name = CGROUP_FILE_GENERIC_PREFIX "procs",
+               .open = cgroup_procs_open,
+               /* .write_u64 = cgroup_procs_write, TODO */
+               .release = cgroup_pidlist_release,
+               .mode = S_IRUGO,
+       },
         {
                 .name = "notify_on_release",
                 .read_u64 = cgroup_read_notify_on_release,
                 .write_u64 = cgroup_write_notify_on_release,
-               .private = FILE_NOTIFY_ON_RELEASE,
         },
  };
  
@@ -2647,7 +2823,6 @@ static struct cftype cft_release_agent = {
         .read_seq_string = cgroup_release_agent_show,
         .write_string = cgroup_release_agent_write,
         .max_write_len = PATH_MAX,
-       .private = FILE_RELEASE_AGENT,
  };
  
  static int cgroup_populate_dir(struct cgroup *cgrp)
@@ -3109,7 +3284,7 @@ int __init cgroup_init(void)
         /* Add init_css_set to the hash table */
         hhead = css_set_hash(init_css_set.subsys);
         hlist_add_head(&init_css_set.hlist, hhead);
-
+       BUG_ON(!init_root_id(&rootnode));
         err = register_filesystem(&cgroup_fs_type);
         if (err < 0)
                 goto out;
@@ -3164,7 +3339,7 @@ static int proc_cgroup_show(struct seq_file *m, void *v)
                 struct cgroup *cgrp;
                 int count = 0;
  
-               seq_printf(m, "%lu:", root->subsys_bits);
+               seq_printf(m, "%d:", root->hierarchy_id);
                 for_each_subsys(root, ss)
                         seq_printf(m, "%s%s", count++ ? "," : "", ss->name);
                 if (strlen(root->name))
@@ -3194,7 +3369,7 @@ static int cgroup_open(struct inode *inode, struct file *file)
         return single_open(file, proc_cgroup_show, pid);
  }
  
-struct file_operations proc_cgroup_operations = {
+const struct file_operations proc_cgroup_operations = {
         .open           = cgroup_open,
         .read           = seq_read,
         .llseek         = seq_lseek,
@@ -3210,8 +3385,8 @@ static int proc_cgroupstats_show(struct seq_file *m, void *v)
         mutex_lock(&cgroup_mutex);
         for (i = 0; i < CGROUP_SUBSYS_COUNT; i++) {
                 struct cgroup_subsys *ss = subsys[i];
-               seq_printf(m, "%s\t%lu\t%d\t%d\n",
-                          ss->name, ss->root->subsys_bits,
+               seq_printf(m, "%s\t%d\t%d\t%d\n",
+                          ss->name, ss->root->hierarchy_id,
                            ss->root->number_of_cgroups, !ss->disabled);
         }
         mutex_unlock(&cgroup_mutex);
@@ -3223,7 +3398,7 @@ static int cgroupstats_open(struct inode *inode, struct file *file)
         return single_open(file, proc_cgroupstats_show, NULL);
  }
  
-static struct file_operations proc_cgroupstats_operations = {
+static const struct file_operations proc_cgroupstats_operations = {
         .open = cgroupstats_open,
         .read = seq_read,
         .llseek = seq_lseek,
@@ -3533,8 +3708,10 @@ static void check_for_release(struct cgroup *cgrp)
  void __css_put(struct cgroup_subsys_state *css)
  {
         struct cgroup *cgrp = css->cgroup;
+       int val;
         rcu_read_lock();
-       if (atomic_dec_return(&css->refcnt) == 1) {
+       val = atomic_dec_return(&css->refcnt);
+       if (val == 1) {
                 if (notify_on_release(cgrp)) {
                         set_bit(CGRP_RELEASABLE, &cgrp->flags);
                         check_for_release(cgrp);
@@ -3542,6 +3719,7 @@ void __css_put(struct cgroup_subsys_state *css)
                 cgroup_wakeup_rmdir_waiter(cgrp);
         }
         rcu_read_unlock();
+       WARN_ON_ONCE(val < 1);
  }
  
  /*
@@ -3929,8 +4107,8 @@ static int current_css_set_cg_links_read(struct cgroup *cont,
                         name = c->dentry->d_name.name;
                 else
                         name = "?";
-               seq_printf(seq, "Root %lu group %s\n",
-                          c->root->subsys_bits, name);
+               seq_printf(seq, "Root %d group %s\n",
+                          c->root->hierarchy_id, name);
         }
         rcu_read_unlock();
         read_unlock(&css_set_lock);