Commit 4e649152 authored by KAMEZAWA Hiroyuki's avatar KAMEZAWA Hiroyuki Committed by Linus Torvalds

memcg: some modification to softlimit under hierarchical memory reclaim.

This patch clean up/fixes for memcg's uncharge soft limit path.

Problems:
  Now, res_counter_charge()/uncharge() handles softlimit information at
  charge/uncharge and softlimit-check is done when event counter per memcg
  goes over limit. Now, event counter per memcg is updated only when
  memory usage is over soft limit. Here, considering hierarchical memcg
  management, ancesotors should be taken care of.

  Now, ancerstors(hierarchy) are handled in charge() but not in uncharge().
  This is not good.

  Prolems:
  1. memcg's event counter incremented only when softlimit hits. That's bad.
     It makes event counter hard to be reused for other purpose.

  2. At uncharge, only the lowest level rescounter is handled. This is bug.
     Because ancesotor's event counter is not incremented, children should
     take care of them.

  3. res_counter_uncharge()'s 3rd argument is NULL in most case.
     ops under res_counter->lock should be small. No "if" sentense is better.

Fixes:
  * Removed soft_limit_xx poitner and checks in charge and uncharge.
    Do-check-only-when-necessary scheme works enough well without them.

  * make event-counter of memcg incremented at every charge/uncharge.
    (per-cpu area will be accessed soon anyway)

  * All ancestors are checked at soft-limit-check. This is necessary because
    ancesotor's event counter may never be modified. Then, they should be
    checked at the same time.
Reviewed-by: default avatarDaisuke Nishimura <nishimura@mxp.nes.nec.co.jp>
Signed-off-by: default avatarKAMEZAWA Hiroyuki <kamezawa.hiroyu@jp.fujitsu.com>
Cc: Paul Menage <menage@google.com>
Cc: Li Zefan <lizf@cn.fujitsu.com>
Cc: Balbir Singh <balbir@in.ibm.com>
Signed-off-by: default avatarAndrew Morton <akpm@linux-foundation.org>
Signed-off-by: default avatarLinus Torvalds <torvalds@linux-foundation.org>
parent 3dece834
...@@ -114,8 +114,7 @@ void res_counter_init(struct res_counter *counter, struct res_counter *parent); ...@@ -114,8 +114,7 @@ void res_counter_init(struct res_counter *counter, struct res_counter *parent);
int __must_check res_counter_charge_locked(struct res_counter *counter, int __must_check res_counter_charge_locked(struct res_counter *counter,
unsigned long val); unsigned long val);
int __must_check res_counter_charge(struct res_counter *counter, int __must_check res_counter_charge(struct res_counter *counter,
unsigned long val, struct res_counter **limit_fail_at, unsigned long val, struct res_counter **limit_fail_at);
struct res_counter **soft_limit_at);
/* /*
* uncharge - tell that some portion of the resource is released * uncharge - tell that some portion of the resource is released
...@@ -128,8 +127,7 @@ int __must_check res_counter_charge(struct res_counter *counter, ...@@ -128,8 +127,7 @@ int __must_check res_counter_charge(struct res_counter *counter,
*/ */
void res_counter_uncharge_locked(struct res_counter *counter, unsigned long val); void res_counter_uncharge_locked(struct res_counter *counter, unsigned long val);
void res_counter_uncharge(struct res_counter *counter, unsigned long val, void res_counter_uncharge(struct res_counter *counter, unsigned long val);
bool *was_soft_limit_excess);
static inline bool res_counter_limit_check_locked(struct res_counter *cnt) static inline bool res_counter_limit_check_locked(struct res_counter *cnt)
{ {
......
...@@ -37,27 +37,17 @@ int res_counter_charge_locked(struct res_counter *counter, unsigned long val) ...@@ -37,27 +37,17 @@ int res_counter_charge_locked(struct res_counter *counter, unsigned long val)
} }
int res_counter_charge(struct res_counter *counter, unsigned long val, int res_counter_charge(struct res_counter *counter, unsigned long val,
struct res_counter **limit_fail_at, struct res_counter **limit_fail_at)
struct res_counter **soft_limit_fail_at)
{ {
int ret; int ret;
unsigned long flags; unsigned long flags;
struct res_counter *c, *u; struct res_counter *c, *u;
*limit_fail_at = NULL; *limit_fail_at = NULL;
if (soft_limit_fail_at)
*soft_limit_fail_at = NULL;
local_irq_save(flags); local_irq_save(flags);
for (c = counter; c != NULL; c = c->parent) { for (c = counter; c != NULL; c = c->parent) {
spin_lock(&c->lock); spin_lock(&c->lock);
ret = res_counter_charge_locked(c, val); ret = res_counter_charge_locked(c, val);
/*
* With soft limits, we return the highest ancestor
* that exceeds its soft limit
*/
if (soft_limit_fail_at &&
!res_counter_soft_limit_check_locked(c))
*soft_limit_fail_at = c;
spin_unlock(&c->lock); spin_unlock(&c->lock);
if (ret < 0) { if (ret < 0) {
*limit_fail_at = c; *limit_fail_at = c;
...@@ -85,8 +75,7 @@ void res_counter_uncharge_locked(struct res_counter *counter, unsigned long val) ...@@ -85,8 +75,7 @@ void res_counter_uncharge_locked(struct res_counter *counter, unsigned long val)
counter->usage -= val; counter->usage -= val;
} }
void res_counter_uncharge(struct res_counter *counter, unsigned long val, void res_counter_uncharge(struct res_counter *counter, unsigned long val)
bool *was_soft_limit_excess)
{ {
unsigned long flags; unsigned long flags;
struct res_counter *c; struct res_counter *c;
...@@ -94,9 +83,6 @@ void res_counter_uncharge(struct res_counter *counter, unsigned long val, ...@@ -94,9 +83,6 @@ void res_counter_uncharge(struct res_counter *counter, unsigned long val,
local_irq_save(flags); local_irq_save(flags);
for (c = counter; c != NULL; c = c->parent) { for (c = counter; c != NULL; c = c->parent) {
spin_lock(&c->lock); spin_lock(&c->lock);
if (was_soft_limit_excess)
*was_soft_limit_excess =
!res_counter_soft_limit_check_locked(c);
res_counter_uncharge_locked(c, val); res_counter_uncharge_locked(c, val);
spin_unlock(&c->lock); spin_unlock(&c->lock);
} }
......
...@@ -352,16 +352,6 @@ __mem_cgroup_remove_exceeded(struct mem_cgroup *mem, ...@@ -352,16 +352,6 @@ __mem_cgroup_remove_exceeded(struct mem_cgroup *mem,
mz->on_tree = false; mz->on_tree = false;
} }
static void
mem_cgroup_insert_exceeded(struct mem_cgroup *mem,
struct mem_cgroup_per_zone *mz,
struct mem_cgroup_tree_per_zone *mctz)
{
spin_lock(&mctz->lock);
__mem_cgroup_insert_exceeded(mem, mz, mctz);
spin_unlock(&mctz->lock);
}
static void static void
mem_cgroup_remove_exceeded(struct mem_cgroup *mem, mem_cgroup_remove_exceeded(struct mem_cgroup *mem,
struct mem_cgroup_per_zone *mz, struct mem_cgroup_per_zone *mz,
...@@ -392,35 +382,41 @@ static bool mem_cgroup_soft_limit_check(struct mem_cgroup *mem) ...@@ -392,35 +382,41 @@ static bool mem_cgroup_soft_limit_check(struct mem_cgroup *mem)
static void mem_cgroup_update_tree(struct mem_cgroup *mem, struct page *page) static void mem_cgroup_update_tree(struct mem_cgroup *mem, struct page *page)
{ {
unsigned long long prev_usage_in_excess, new_usage_in_excess; unsigned long long new_usage_in_excess;
bool updated_tree = false;
struct mem_cgroup_per_zone *mz; struct mem_cgroup_per_zone *mz;
struct mem_cgroup_tree_per_zone *mctz; struct mem_cgroup_tree_per_zone *mctz;
int nid = page_to_nid(page);
mz = mem_cgroup_zoneinfo(mem, page_to_nid(page), page_zonenum(page)); int zid = page_zonenum(page);
mctz = soft_limit_tree_from_page(page); mctz = soft_limit_tree_from_page(page);
/* /*
* We do updates in lazy mode, mem's are removed * Necessary to update all ancestors when hierarchy is used.
* lazily from the per-zone, per-node rb tree * because their event counter is not touched.
*/ */
prev_usage_in_excess = mz->usage_in_excess; for (; mem; mem = parent_mem_cgroup(mem)) {
mz = mem_cgroup_zoneinfo(mem, nid, zid);
new_usage_in_excess = res_counter_soft_limit_excess(&mem->res); new_usage_in_excess =
if (prev_usage_in_excess) { res_counter_soft_limit_excess(&mem->res);
mem_cgroup_remove_exceeded(mem, mz, mctz); /*
updated_tree = true; * We have to update the tree if mz is on RB-tree or
} * mem is over its softlimit.
if (!new_usage_in_excess) */
goto done; if (new_usage_in_excess || mz->on_tree) {
mem_cgroup_insert_exceeded(mem, mz, mctz);
done:
if (updated_tree) {
spin_lock(&mctz->lock); spin_lock(&mctz->lock);
mz->usage_in_excess = new_usage_in_excess; /* if on-tree, remove it */
if (mz->on_tree)
__mem_cgroup_remove_exceeded(mem, mz, mctz);
/*
* if over soft limit, insert again. mz->usage_in_excess
* will be updated properly.
*/
if (new_usage_in_excess)
__mem_cgroup_insert_exceeded(mem, mz, mctz);
else
mz->usage_in_excess = 0;
spin_unlock(&mctz->lock); spin_unlock(&mctz->lock);
} }
}
} }
static void mem_cgroup_remove_from_trees(struct mem_cgroup *mem) static void mem_cgroup_remove_from_trees(struct mem_cgroup *mem)
...@@ -1271,9 +1267,9 @@ static int __mem_cgroup_try_charge(struct mm_struct *mm, ...@@ -1271,9 +1267,9 @@ static int __mem_cgroup_try_charge(struct mm_struct *mm,
gfp_t gfp_mask, struct mem_cgroup **memcg, gfp_t gfp_mask, struct mem_cgroup **memcg,
bool oom, struct page *page) bool oom, struct page *page)
{ {
struct mem_cgroup *mem, *mem_over_limit, *mem_over_soft_limit; struct mem_cgroup *mem, *mem_over_limit;
int nr_retries = MEM_CGROUP_RECLAIM_RETRIES; int nr_retries = MEM_CGROUP_RECLAIM_RETRIES;
struct res_counter *fail_res, *soft_fail_res = NULL; struct res_counter *fail_res;
if (unlikely(test_thread_flag(TIF_MEMDIE))) { if (unlikely(test_thread_flag(TIF_MEMDIE))) {
/* Don't account this! */ /* Don't account this! */
...@@ -1305,17 +1301,16 @@ static int __mem_cgroup_try_charge(struct mm_struct *mm, ...@@ -1305,17 +1301,16 @@ static int __mem_cgroup_try_charge(struct mm_struct *mm,
if (mem_cgroup_is_root(mem)) if (mem_cgroup_is_root(mem))
goto done; goto done;
ret = res_counter_charge(&mem->res, PAGE_SIZE, &fail_res, ret = res_counter_charge(&mem->res, PAGE_SIZE, &fail_res);
&soft_fail_res);
if (likely(!ret)) { if (likely(!ret)) {
if (!do_swap_account) if (!do_swap_account)
break; break;
ret = res_counter_charge(&mem->memsw, PAGE_SIZE, ret = res_counter_charge(&mem->memsw, PAGE_SIZE,
&fail_res, NULL); &fail_res);
if (likely(!ret)) if (likely(!ret))
break; break;
/* mem+swap counter fails */ /* mem+swap counter fails */
res_counter_uncharge(&mem->res, PAGE_SIZE, NULL); res_counter_uncharge(&mem->res, PAGE_SIZE);
flags |= MEM_CGROUP_RECLAIM_NOSWAP; flags |= MEM_CGROUP_RECLAIM_NOSWAP;
mem_over_limit = mem_cgroup_from_res_counter(fail_res, mem_over_limit = mem_cgroup_from_res_counter(fail_res,
memsw); memsw);
...@@ -1354,16 +1349,11 @@ static int __mem_cgroup_try_charge(struct mm_struct *mm, ...@@ -1354,16 +1349,11 @@ static int __mem_cgroup_try_charge(struct mm_struct *mm,
} }
} }
/* /*
* Insert just the ancestor, we should trickle down to the correct * Insert ancestor (and ancestor's ancestors), to softlimit RB-tree.
* cgroup for reclaim, since the other nodes will be below their * if they exceeds softlimit.
* soft limit
*/ */
if (soft_fail_res) { if (mem_cgroup_soft_limit_check(mem))
mem_over_soft_limit = mem_cgroup_update_tree(mem, page);
mem_cgroup_from_res_counter(soft_fail_res, res);
if (mem_cgroup_soft_limit_check(mem_over_soft_limit))
mem_cgroup_update_tree(mem_over_soft_limit, page);
}
done: done:
return 0; return 0;
nomem: nomem:
...@@ -1438,10 +1428,9 @@ static void __mem_cgroup_commit_charge(struct mem_cgroup *mem, ...@@ -1438,10 +1428,9 @@ static void __mem_cgroup_commit_charge(struct mem_cgroup *mem,
if (unlikely(PageCgroupUsed(pc))) { if (unlikely(PageCgroupUsed(pc))) {
unlock_page_cgroup(pc); unlock_page_cgroup(pc);
if (!mem_cgroup_is_root(mem)) { if (!mem_cgroup_is_root(mem)) {
res_counter_uncharge(&mem->res, PAGE_SIZE, NULL); res_counter_uncharge(&mem->res, PAGE_SIZE);
if (do_swap_account) if (do_swap_account)
res_counter_uncharge(&mem->memsw, PAGE_SIZE, res_counter_uncharge(&mem->memsw, PAGE_SIZE);
NULL);
} }
css_put(&mem->css); css_put(&mem->css);
return; return;
...@@ -1520,7 +1509,7 @@ static int mem_cgroup_move_account(struct page_cgroup *pc, ...@@ -1520,7 +1509,7 @@ static int mem_cgroup_move_account(struct page_cgroup *pc,
goto out; goto out;
if (!mem_cgroup_is_root(from)) if (!mem_cgroup_is_root(from))
res_counter_uncharge(&from->res, PAGE_SIZE, NULL); res_counter_uncharge(&from->res, PAGE_SIZE);
mem_cgroup_charge_statistics(from, pc, false); mem_cgroup_charge_statistics(from, pc, false);
page = pc->page; page = pc->page;
...@@ -1540,7 +1529,7 @@ static int mem_cgroup_move_account(struct page_cgroup *pc, ...@@ -1540,7 +1529,7 @@ static int mem_cgroup_move_account(struct page_cgroup *pc,
} }
if (do_swap_account && !mem_cgroup_is_root(from)) if (do_swap_account && !mem_cgroup_is_root(from))
res_counter_uncharge(&from->memsw, PAGE_SIZE, NULL); res_counter_uncharge(&from->memsw, PAGE_SIZE);
css_put(&from->css); css_put(&from->css);
css_get(&to->css); css_get(&to->css);
...@@ -1611,9 +1600,9 @@ static int mem_cgroup_move_parent(struct page_cgroup *pc, ...@@ -1611,9 +1600,9 @@ static int mem_cgroup_move_parent(struct page_cgroup *pc,
css_put(&parent->css); css_put(&parent->css);
/* uncharge if move fails */ /* uncharge if move fails */
if (!mem_cgroup_is_root(parent)) { if (!mem_cgroup_is_root(parent)) {
res_counter_uncharge(&parent->res, PAGE_SIZE, NULL); res_counter_uncharge(&parent->res, PAGE_SIZE);
if (do_swap_account) if (do_swap_account)
res_counter_uncharge(&parent->memsw, PAGE_SIZE, NULL); res_counter_uncharge(&parent->memsw, PAGE_SIZE);
} }
return ret; return ret;
} }
...@@ -1804,8 +1793,7 @@ __mem_cgroup_commit_charge_swapin(struct page *page, struct mem_cgroup *ptr, ...@@ -1804,8 +1793,7 @@ __mem_cgroup_commit_charge_swapin(struct page *page, struct mem_cgroup *ptr,
* calling css_tryget * calling css_tryget
*/ */
if (!mem_cgroup_is_root(memcg)) if (!mem_cgroup_is_root(memcg))
res_counter_uncharge(&memcg->memsw, PAGE_SIZE, res_counter_uncharge(&memcg->memsw, PAGE_SIZE);
NULL);
mem_cgroup_swap_statistics(memcg, false); mem_cgroup_swap_statistics(memcg, false);
mem_cgroup_put(memcg); mem_cgroup_put(memcg);
} }
...@@ -1832,9 +1820,9 @@ void mem_cgroup_cancel_charge_swapin(struct mem_cgroup *mem) ...@@ -1832,9 +1820,9 @@ void mem_cgroup_cancel_charge_swapin(struct mem_cgroup *mem)
if (!mem) if (!mem)
return; return;
if (!mem_cgroup_is_root(mem)) { if (!mem_cgroup_is_root(mem)) {
res_counter_uncharge(&mem->res, PAGE_SIZE, NULL); res_counter_uncharge(&mem->res, PAGE_SIZE);
if (do_swap_account) if (do_swap_account)
res_counter_uncharge(&mem->memsw, PAGE_SIZE, NULL); res_counter_uncharge(&mem->memsw, PAGE_SIZE);
} }
css_put(&mem->css); css_put(&mem->css);
} }
...@@ -1849,7 +1837,6 @@ __mem_cgroup_uncharge_common(struct page *page, enum charge_type ctype) ...@@ -1849,7 +1837,6 @@ __mem_cgroup_uncharge_common(struct page *page, enum charge_type ctype)
struct page_cgroup *pc; struct page_cgroup *pc;
struct mem_cgroup *mem = NULL; struct mem_cgroup *mem = NULL;
struct mem_cgroup_per_zone *mz; struct mem_cgroup_per_zone *mz;
bool soft_limit_excess = false;
if (mem_cgroup_disabled()) if (mem_cgroup_disabled())
return NULL; return NULL;
...@@ -1889,10 +1876,10 @@ __mem_cgroup_uncharge_common(struct page *page, enum charge_type ctype) ...@@ -1889,10 +1876,10 @@ __mem_cgroup_uncharge_common(struct page *page, enum charge_type ctype)
} }
if (!mem_cgroup_is_root(mem)) { if (!mem_cgroup_is_root(mem)) {
res_counter_uncharge(&mem->res, PAGE_SIZE, &soft_limit_excess); res_counter_uncharge(&mem->res, PAGE_SIZE);
if (do_swap_account && if (do_swap_account &&
(ctype != MEM_CGROUP_CHARGE_TYPE_SWAPOUT)) (ctype != MEM_CGROUP_CHARGE_TYPE_SWAPOUT))
res_counter_uncharge(&mem->memsw, PAGE_SIZE, NULL); res_counter_uncharge(&mem->memsw, PAGE_SIZE);
} }
if (ctype == MEM_CGROUP_CHARGE_TYPE_SWAPOUT) if (ctype == MEM_CGROUP_CHARGE_TYPE_SWAPOUT)
mem_cgroup_swap_statistics(mem, true); mem_cgroup_swap_statistics(mem, true);
...@@ -1909,7 +1896,7 @@ __mem_cgroup_uncharge_common(struct page *page, enum charge_type ctype) ...@@ -1909,7 +1896,7 @@ __mem_cgroup_uncharge_common(struct page *page, enum charge_type ctype)
mz = page_cgroup_zoneinfo(pc); mz = page_cgroup_zoneinfo(pc);
unlock_page_cgroup(pc); unlock_page_cgroup(pc);
if (soft_limit_excess && mem_cgroup_soft_limit_check(mem)) if (mem_cgroup_soft_limit_check(mem))
mem_cgroup_update_tree(mem, page); mem_cgroup_update_tree(mem, page);
/* at swapout, this memcg will be accessed to record to swap */ /* at swapout, this memcg will be accessed to record to swap */
if (ctype != MEM_CGROUP_CHARGE_TYPE_SWAPOUT) if (ctype != MEM_CGROUP_CHARGE_TYPE_SWAPOUT)
...@@ -1987,7 +1974,7 @@ void mem_cgroup_uncharge_swap(swp_entry_t ent) ...@@ -1987,7 +1974,7 @@ void mem_cgroup_uncharge_swap(swp_entry_t ent)
* This memcg can be obsolete one. We avoid calling css_tryget * This memcg can be obsolete one. We avoid calling css_tryget
*/ */
if (!mem_cgroup_is_root(memcg)) if (!mem_cgroup_is_root(memcg))
res_counter_uncharge(&memcg->memsw, PAGE_SIZE, NULL); res_counter_uncharge(&memcg->memsw, PAGE_SIZE);
mem_cgroup_swap_statistics(memcg, false); mem_cgroup_swap_statistics(memcg, false);
mem_cgroup_put(memcg); mem_cgroup_put(memcg);
} }
......
Markdown is supported
0%
or
You are about to add 0 people to the discussion. Proceed with caution.
Finish editing this message first!
Please register or to comment