From: Ridong Chen As commit 359a5e1416ca ("mm: multi-gen LRU: retry folios written back while isolated") mentioned: The page reclaim isolates a batch of folios from the tail of one of the LRU lists and works on those folios one by one. For a suitable swap-backed folio, if the swap device is async, it queues that folio for writeback. After the page reclaim finishes an entire batch, it puts back the folios it queued for writeback to the head of the original LRU list. In the meantime, the page writeback flushes the queued folios also by batches. Its batching logic is independent from that of the page reclaim. For each of the folios it writes back, the page writeback calls folio_rotate_reclaimable() which tries to rotate a folio to the tail. folio_rotate_reclaimable() only works for a folio after the page reclaim has put it back. If an async swap device is fast enough, the page writeback can finish with that folio while the page reclaim is still working on the rest of the batch containing it. In this case, that folio will remain at the head and the page reclaim will not retry it before reaching there". The commit 359a5e1416ca ("mm: multi-gen LRU: retry folios written back while isolated") only fixed the issue for mglru. However, this issue also exists in the traditional active/inactive LRU and was found at [1]. It can be reproduced with below steps: 1. Compile with CONFIG_TRANSPARENT_HUGEPAGE=y 2. Mount memcg v1, and create memcg named test_memcg and set limit_in_bytes=1G, memsw.limit_in_bytes=2G. 3. Create a 1G swap file, and allocate 1.35G anon memory in test_memcg. It was found that: cat memory.usage_in_bytes 1073700864 cat memory.memsw.usage_in_bytes 1413124096 free -h total used free Mem: 1.6Gi 1.2Gi 299Mi Swap: 1.0Gi 678Mi 346Mi As shown above, the test_memcg charged about 324M swap (memsw.usage minus usage), but almost 678M swap memory was used, which means that 350M+ may be wasted because other memcgs can not use these swap memory. This issue should be fixed in the same way as mglru. Therefore, the common logic was extracted to the 'find_folios_written_back' function firstly, which is then reused in the 'shrink_inactive_list' function. Finally, retry reclaiming those folios that may have missed the rotation for traditional LRU. After change, the same test case only wasted about 2M swap. The swap device usage matches what the memcg actually charged. cat memory.usage_in_bytes 1070301184 cat memory.memsw.usage_in_bytes 1412448256 free -h total used free Mem: 1.6Gi 1.2Gi 299Mi Swap: 1.0Gi 327Mi 696Mi [1] https://lore.kernel.org/linux-kernel/20241010081802.290893-1-chenridong@huaweicloud.com/ [2] https://lore.kernel.org/linux-kernel/CAGsJ_4zqL8ZHNRZ44o_CC69kE7DBVXvbZfvmQxMGiFqRxqHQdA@mail.gmail.com/ Signed-off-by: Ridong Chen --- v8: Rebase to the -next tree, adapt to the current reclaim API, and retest. [v7]: https://lore.kernel.org/linux-mm/20250111091504.1363075-1-chenridong@huaweicloud.com/ mm/vmscan.c | 104 ++++++++++++++++++++++++++++++++++++---------------- 1 file changed, 73 insertions(+), 31 deletions(-) diff --git a/mm/vmscan.c b/mm/vmscan.c index d1495a7d469d..6e13e570aab8 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -181,6 +181,10 @@ struct scan_control { struct reclaim_state reclaim_state; }; +static void find_folios_written_back(struct list_head *list, + struct list_head *clean, struct lruvec *lruvec, + int type, bool skip_retry); + #ifdef ARCH_HAS_PREFETCHW static inline void prefetchw_prev_lru_folio(struct folio *folio, struct list_head *base) @@ -2013,14 +2017,16 @@ static unsigned long shrink_inactive_list(unsigned long nr_to_scan, enum lru_list lru) { LIST_HEAD(folio_list); + LIST_HEAD(clean_list); unsigned long nr_scanned; - unsigned int nr_reclaimed = 0; - unsigned long nr_taken; + unsigned int nr_reclaimed, total_reclaimed = 0; + unsigned long nr_taken, isolated; struct reclaim_stat stat; bool file = is_file_lru(lru); enum node_stat_item item; struct pglist_data *pgdat = lruvec_pgdat(lruvec); bool stalled = false; + bool skip_retry = false; while (unlikely(too_many_isolated(pgdat, file, sc))) { if (stalled) @@ -2052,25 +2058,42 @@ static unsigned long shrink_inactive_list(unsigned long nr_to_scan, if (nr_taken == 0) return 0; + isolated = nr_taken; +retry: nr_reclaimed = shrink_folio_list(&folio_list, pgdat, sc, &stat, false, lruvec_memcg(lruvec)); + total_reclaimed += nr_reclaimed; + + /* Retry pass is only meant for clean folios without new isolation */ + if (isolated) + handle_reclaim_writeback(isolated, pgdat, sc, &stat); + trace_mm_vmscan_lru_shrink_inactive(pgdat->node_id, + nr_scanned, nr_reclaimed, &stat, sc->priority, file); + + find_folios_written_back(&folio_list, &clean_list, lruvec, file, skip_retry); move_folios_to_lru(&folio_list); mod_lruvec_state(lruvec, PGDEMOTE_KSWAPD + reclaimer_offset(sc), stat.nr_demoted); - mod_node_page_state(pgdat, NR_ISOLATED_ANON + file, -nr_taken); item = PGSTEAL_KSWAPD + reclaimer_offset(sc); mod_lruvec_state(lruvec, item, nr_reclaimed); mod_lruvec_state(lruvec, PGSTEAL_ANON + file, nr_reclaimed); - if (nr_scanned > nr_reclaimed) + + if (!list_empty(&clean_list)) { + list_splice_init(&clean_list, &folio_list); + skip_retry = true; + /* Retry folios were already isolated and accounted above */ + isolated = 0; + goto retry; + } + + mod_node_page_state(pgdat, NR_ISOLATED_ANON + file, -nr_taken); + if (nr_scanned > total_reclaimed) mod_lruvec_state(lruvec, PGROTATE_ANON + file, - nr_scanned - nr_reclaimed); + nr_scanned - total_reclaimed); - handle_reclaim_writeback(nr_taken, pgdat, sc, &stat); - trace_mm_vmscan_lru_shrink_inactive(pgdat->node_id, - nr_scanned, nr_reclaimed, &stat, sc->priority, file); - return nr_reclaimed; + return total_reclaimed; } /* @@ -5006,8 +5029,6 @@ static int evict_folios(unsigned long nr_to_scan, struct lruvec *lruvec, { LIST_HEAD(list); LIST_HEAD(clean); - struct folio *folio; - struct folio *next; enum node_stat_item item; struct reclaim_stat stat; struct lru_gen_mm_walk *walk; @@ -5046,26 +5067,7 @@ static int evict_folios(unsigned long nr_to_scan, struct lruvec *lruvec, type_scanned, reclaimed, &stat, sc->priority, type ? LRU_INACTIVE_FILE : LRU_INACTIVE_ANON); - list_for_each_entry_safe_reverse(folio, next, &list, lru) { - DEFINE_MIN_SEQ(lruvec); - - /* move_folios_to_lru() culls unevictable folios via folio_putback_lru() */ - if (!folio_evictable(folio)) - continue; - - /* retry folios that may have missed folio_rotate_reclaimable() */ - if (!skip_retry && !folio_test_active(folio) && !folio_mapped(folio) && - !folio_test_dirty(folio) && !folio_test_writeback(folio)) { - list_move(&folio->lru, &clean); - continue; - } - - /* don't add rejected folios to the oldest generation */ - if (lru_gen_folio_seq(lruvec, folio, false) == min_seq[type]) { - folio_set_lru_refs(folio, 0); - folio_set_active(folio); - } - } + find_folios_written_back(&list, &clean, lruvec, type, skip_retry); move_folios_to_lru(&list); @@ -6108,6 +6110,46 @@ static void lru_gen_shrink_node(struct pglist_data *pgdat, struct scan_control * #endif /* CONFIG_LRU_GEN */ +/** + * find_folios_written_back - Find and move the written back folios to a new list. + * @list: folios list + * @clean: the written back folios list + * @lruvec: the lruvec + * @type: LRU type (only used for CONFIG_LRU_GEN) + * @skip_retry: whether skip retry. + */ +static void find_folios_written_back(struct list_head *list, + struct list_head *clean, struct lruvec *lruvec, + int type, bool skip_retry) +{ + struct folio *folio; + struct folio *next; + + list_for_each_entry_safe_reverse(folio, next, list, lru) { +#ifdef CONFIG_LRU_GEN + DEFINE_MIN_SEQ(lruvec); +#endif + /* move_folios_to_lru() culls unevictable folios via folio_putback_lru() */ + if (!folio_evictable(folio)) + continue; + + /* retry folios that may have missed folio_rotate_reclaimable() */ + if (!skip_retry && !folio_test_active(folio) && !folio_mapped(folio) && + !folio_test_dirty(folio) && !folio_test_writeback(folio)) { + list_move(&folio->lru, clean); + continue; + } +#ifdef CONFIG_LRU_GEN + /* don't add rejected folios to the oldest generation */ + if (lruvec->lrugen.enabled && + lru_gen_folio_seq(lruvec, folio, false) == min_seq[type]) { + folio_set_lru_refs(folio, 0); + folio_set_active(folio); + } +#endif + } +} + static void shrink_lruvec(struct lruvec *lruvec, struct scan_control *sc) { unsigned long nr[NR_LRU_LISTS]; -- 2.34.1