oom: don't assume that a coredumping thread will exit soon

[cascardo/linux.git] / mm / memcontrol.c
diff --git a/mm/memcontrol.c b/mm/memcontrol.c

index 85df503..998fb17 100644 (file)
--- a/mm/memcontrol.c
+++ b/mm/memcontrol.c
@@ -296,7 +296,6 @@ struct mem_cgroup {
          * Should the accounting and control be hierarchical, per subtree?
          */
         bool use_hierarchy;
-       unsigned long kmem_account_flags; /* See KMEM_ACCOUNTED_*, below */
  
         bool            oom_lock;
         atomic_t        under_oom;
@@ -366,22 +365,11 @@ struct mem_cgroup {
         /* WARNING: nodeinfo must be the last member here */
  };
  
-/* internal only representation about the status of kmem accounting. */
-enum {
-       KMEM_ACCOUNTED_ACTIVE, /* accounted by this cgroup itself */
-};
-
  #ifdef CONFIG_MEMCG_KMEM
-static inline void memcg_kmem_set_active(struct mem_cgroup *memcg)
-{
-       set_bit(KMEM_ACCOUNTED_ACTIVE, &memcg->kmem_account_flags);
-}
-
  static bool memcg_kmem_is_active(struct mem_cgroup *memcg)
  {
-       return test_bit(KMEM_ACCOUNTED_ACTIVE, &memcg->kmem_account_flags);
+       return memcg->kmemcg_id >= 0;
  }
-
  #endif
  
  /* Stuffs for move charges at task migration. */
@@ -1571,7 +1559,7 @@ static void mem_cgroup_out_of_memory(struct mem_cgroup *memcg, gfp_t gfp_mask,
          * select it.  The goal is to allow it to allocate so that it may
          * quickly exit and free its memory.
          */
-       if (fatal_signal_pending(current) || current->flags & PF_EXITING) {
+       if (fatal_signal_pending(current) || task_will_free_mem(current)) {
                 set_thread_flag(TIF_MEMDIE);
                 return;
         }
@@ -2685,37 +2673,6 @@ static void memcg_unregister_cache(struct kmem_cache *cachep)
         css_put(&memcg->css);
  }
  
-/*
- * During the creation a new cache, we need to disable our accounting mechanism
- * altogether. This is true even if we are not creating, but rather just
- * enqueing new caches to be created.
- *
- * This is because that process will trigger allocations; some visible, like
- * explicit kmallocs to auxiliary data structures, name strings and internal
- * cache structures; some well concealed, like INIT_WORK() that can allocate
- * objects during debug.
- *
- * If any allocation happens during memcg_kmem_get_cache, we will recurse back
- * to it. This may not be a bounded recursion: since the first cache creation
- * failed to complete (waiting on the allocation), we'll just try to create the
- * cache again, failing at the same point.
- *
- * memcg_kmem_get_cache is prepared to abort after seeing a positive count of
- * memcg_kmem_skip_account. So we enclose anything that might allocate memory
- * inside the following two functions.
- */
-static inline void memcg_stop_kmem_account(void)
-{
-       VM_BUG_ON(!current->mm);
-       current->memcg_kmem_skip_account++;
-}
-
-static inline void memcg_resume_kmem_account(void)
-{
-       VM_BUG_ON(!current->mm);
-       current->memcg_kmem_skip_account--;
-}
-
  int __memcg_cleanup_cache_params(struct kmem_cache *s)
  {
         struct kmem_cache *c;
@@ -2810,9 +2767,9 @@ static void memcg_schedule_register_cache(struct mem_cgroup *memcg,
          * this point we can't allow ourselves back into memcg_kmem_get_cache,
          * the safest choice is to do it like this, wrapping the whole function.
          */
-       memcg_stop_kmem_account();
+       current->memcg_kmem_skip_account = 1;
         __memcg_schedule_register_cache(memcg, cachep);
-       memcg_resume_kmem_account();
+       current->memcg_kmem_skip_account = 0;
  }
  
  int __memcg_charge_slab(struct kmem_cache *cachep, gfp_t gfp, int order)
@@ -2847,8 +2804,7 @@ void __memcg_uncharge_slab(struct kmem_cache *cachep, int order)
   * Can't be called in interrupt context or from kernel threads.
   * This function needs to be called with rcu_read_lock() held.
   */
-struct kmem_cache *__memcg_kmem_get_cache(struct kmem_cache *cachep,
-                                         gfp_t gfp)
+struct kmem_cache *__memcg_kmem_get_cache(struct kmem_cache *cachep)
  {
         struct mem_cgroup *memcg;
         struct kmem_cache *memcg_cachep;
@@ -2856,7 +2812,7 @@ struct kmem_cache *__memcg_kmem_get_cache(struct kmem_cache *cachep,
         VM_BUG_ON(!cachep->memcg_params);
         VM_BUG_ON(!cachep->memcg_params->is_root_cache);
  
-       if (!current->mm || current->memcg_kmem_skip_account)
+       if (current->memcg_kmem_skip_account)
                 return cachep;
  
         rcu_read_lock();
@@ -2917,34 +2873,6 @@ __memcg_kmem_newpage_charge(gfp_t gfp, struct mem_cgroup **_memcg, int order)
  
         *_memcg = NULL;
  
-       /*
-        * Disabling accounting is only relevant for some specific memcg
-        * internal allocations. Therefore we would initially not have such
-        * check here, since direct calls to the page allocator that are
-        * accounted to kmemcg (alloc_kmem_pages and friends) only happen
-        * outside memcg core. We are mostly concerned with cache allocations,
-        * and by having this test at memcg_kmem_get_cache, we are already able
-        * to relay the allocation to the root cache and bypass the memcg cache
-        * altogether.
-        *
-        * There is one exception, though: the SLUB allocator does not create
-        * large order caches, but rather service large kmallocs directly from
-        * the page allocator. Therefore, the following sequence when backed by
-        * the SLUB allocator:
-        *
-        *      memcg_stop_kmem_account();
-        *      kmalloc(<large_number>)
-        *      memcg_resume_kmem_account();
-        *
-        * would effectively ignore the fact that we should skip accounting,
-        * since it will drive us directly to this function without passing
-        * through the cache selector memcg_kmem_get_cache. Such large
-        * allocations are extremely rare but can happen, for instance, for the
-        * cache arrays. We bring this test here.
-        */
-       if (!current->mm || current->memcg_kmem_skip_account)
-               return true;
-
         memcg = get_mem_cgroup_from_mm(current->mm);
  
         if (!memcg_kmem_is_active(memcg)) {
@@ -3538,12 +3466,6 @@ static int memcg_activate_kmem(struct mem_cgroup *memcg,
         if (memcg_kmem_is_active(memcg))
                 return 0;
  
-       /*
-        * We are going to allocate memory for data shared by all memory
-        * cgroups so let's stop accounting here.
-        */
-       memcg_stop_kmem_account();
-
         /*
          * For simplicity, we won't allow this to be disabled.  It also can't
          * be changed if the cgroup has children already, or if tasks had
@@ -3570,25 +3492,22 @@ static int memcg_activate_kmem(struct mem_cgroup *memcg,
                 goto out;
         }
  
-       memcg->kmemcg_id = memcg_id;
-       INIT_LIST_HEAD(&memcg->memcg_slab_caches);
-
         /*
-        * We couldn't have accounted to this cgroup, because it hasn't got the
-        * active bit set yet, so this should succeed.
+        * We couldn't have accounted to this cgroup, because it hasn't got
+        * activated yet, so this should succeed.
          */
         err = page_counter_limit(&memcg->kmem, nr_pages);
         VM_BUG_ON(err);
  
         static_key_slow_inc(&memcg_kmem_enabled_key);
         /*
-        * Setting the active bit after enabling static branching will
+        * A memory cgroup is considered kmem-active as soon as it gets
+        * kmemcg_id. Setting the id after enabling static branching will
          * guarantee no one starts accounting before all call sites are
          * patched.
          */
-       memcg_kmem_set_active(memcg);
+       memcg->kmemcg_id = memcg_id;
  out:
-       memcg_resume_kmem_account();
         return err;
  }
  
@@ -4259,7 +4178,6 @@ static int memcg_init_kmem(struct mem_cgroup *memcg, struct cgroup_subsys *ss)
  {
         int ret;
  
-       memcg->kmemcg_id = -1;
         ret = memcg_propagate_kmem(memcg);
         if (ret)
                 return ret;
@@ -4724,17 +4642,6 @@ static void __mem_cgroup_free(struct mem_cgroup *memcg)
  
         free_percpu(memcg->stat);
  
-       /*
-        * We need to make sure that (at least for now), the jump label
-        * destruction code runs outside of the cgroup lock. This is because
-        * get_online_cpus(), which is called from the static_branch update,
-        * can't be called inside the cgroup_lock. cpusets are the ones
-        * enforcing this dependency, so if they ever change, we might as well.
-        *
-        * schedule_work() will guarantee this happens. Be careful if you need
-        * to move this code around, and make sure it is outside
-        * the cgroup_lock.
-        */
         disarm_static_keys(memcg);
         kfree(memcg);
  }
@@ -4804,6 +4711,10 @@ mem_cgroup_css_alloc(struct cgroup_subsys_state *parent_css)
         vmpressure_init(&memcg->vmpressure);
         INIT_LIST_HEAD(&memcg->event_list);
         spin_lock_init(&memcg->event_list_lock);
+#ifdef CONFIG_MEMCG_KMEM
+       memcg->kmemcg_id = -1;
+       INIT_LIST_HEAD(&memcg->memcg_slab_caches);
+#endif
  
         return &memcg->css;