From: Nhat Pham An xswap entry is charged only when it gets physical backing, so a zswap-capable memcg can keep swapping out through xswap even when its physical swap margin is zero. mem_cgroup_get_nr_swap_pages() would otherwise starve anon reclaim with memory.swap.max set to 0. Return PAGE_COUNTER_MAX when an xswap device is active, zswap is on, and the memcg allows zswap. Track active xswap devices in nr_xswap_files, mirroring nr_real_swapfiles. Suggested-by: Johannes Weiner Signed-off-by: Nhat Pham Signed-off-by: Baoquan He --- include/linux/swap.h | 6 ++++++ mm/memcontrol.c | 14 +++++++++++++- mm/swapfile.c | 5 +++++ 3 files changed, 24 insertions(+), 1 deletion(-) diff --git a/include/linux/swap.h b/include/linux/swap.h index 7ceac868a885..540a48f203e4 100644 --- a/include/linux/swap.h +++ b/include/linux/swap.h @@ -391,6 +391,12 @@ void free_pages_and_swap_cache(struct encoded_page **, int); /* linux/mm/swapfile.c */ extern atomic_long_t nr_swap_pages; extern atomic_t nr_real_swapfiles; +extern atomic_t nr_xswap_files; + +static inline bool xswap_enabled(void) +{ + return atomic_read(&nr_xswap_files) > 0; +} extern long total_swap_pages; extern atomic_t nr_rotate_swap; diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 1d35b7ae8d70..7646e66bc5df 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -6035,8 +6035,20 @@ void __mem_cgroup_swap_put(struct mem_cgroup *memcg, unsigned int nr_pages) long mem_cgroup_get_nr_swap_pages(struct mem_cgroup *memcg) { - long nr_swap_pages = get_nr_swap_pages(); + long nr_swap_pages; + /* + * xswap charges physical backing, not allocation, so virtual swap is + * unbounded for a zswap-capable memcg and the swap.max walk below + * would starve anon reclaim. swap.max is still enforced when the + * backing is charged. + */ + if (xswap_enabled() && zswap_is_enabled() && + (mem_cgroup_disabled() || do_memsw_account() || + mem_cgroup_may_zswap(memcg, false))) + return PAGE_COUNTER_MAX; + + nr_swap_pages = get_nr_swap_pages(); if (!mem_cgroup_disabled() && !do_memsw_account()) nr_swap_pages = min(nr_swap_pages, page_counter_margin(&memcg->swap)); diff --git a/mm/swapfile.c b/mm/swapfile.c index f848c6edc6e7..f59dc74f574e 100644 --- a/mm/swapfile.c +++ b/mm/swapfile.c @@ -189,6 +189,7 @@ static DEFINE_SPINLOCK(swap_lock); static unsigned int nr_swapfiles; atomic_long_t nr_swap_pages; atomic_t nr_real_swapfiles; +atomic_t nr_xswap_files; /* * Some modules use swappable objects and may try to swap them out under * memory pressure (via the shrinker). Before doing so, they may wish to @@ -1425,6 +1426,8 @@ static void del_from_avail_list(struct swap_info_struct *si, bool swapoff) /* Count active devices, not merely those on the avail list. */ if (!(si->flags & SWP_XSWAP)) atomic_sub(1, &nr_real_swapfiles); + else + atomic_sub(1, &nr_xswap_files); atomic_long_or(SWAP_USAGE_OFFLIST_BIT, &si->inuse_pages); } else { /* @@ -1486,6 +1489,8 @@ static void add_to_avail_list(struct swap_info_struct *si, bool swapon) plist_add(&si->avail_list, &swap_avail_head); if (swapon && !(si->flags & SWP_XSWAP)) atomic_add(1, &nr_real_swapfiles); + else if (swapon) + atomic_add(1, &nr_xswap_files); skip: spin_unlock(&swap_avail_lock); -- 2.54.0