From: Kairui Song commit ca707239e8a7 ("mm: update_lru_size warn and reset bad lru_size") introduced a sanity check to catch memcg counter underflow, which was more of a workaround for another bug: lru_zone_size is unsigned, so underflow wraps it around and returns an enormously large number, then the memcg shrinker loops almost forever as the calculated number of folios to shrink is huge. That commit also checked if a zero value matches the empty LRU list, so we have to hold the LRU lock, and handle the positive and negative deltas separately. But later commit b4536f0c829c ("mm, memcg: fix the active list aging for lowmem requests when memcg is enabled") already removed the LRU emptiness check, so handling the deltas separately is no longer needed. And if we just turn it into an atomic long, underflow isn't a big issue either, and can be checked at the reader side, which is called much less frequently than the updater. So let's turn the counter into an atomic long and check at the reader side instead, which has a smaller overhead. The underflow correction is removed: a massive leak of the LRU size counter would indicate that something else has gone very wrong, and one should fix that leaking site instead. Besides, the updater-side sanity check is unlikely to catch the leaking site anyway: if a folio was removed without updating the counter while other folios remain on the LRU, the WARN only triggers much later, from a likely innocent callsite. Reviewed-by: Ridong Chen Signed-off-by: Kairui Song --- include/linux/memcontrol.h | 9 +++++++-- mm/memcontrol.c | 18 +----------------- 2 files changed, 8 insertions(+), 19 deletions(-) diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index 215e2e87f42b..7b89d0cb5f6c 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -113,7 +113,7 @@ struct mem_cgroup_per_node { /* Fields which get updated often at the end. */ struct lruvec lruvec; CACHELINE_PADDING(_pad2_); - unsigned long lru_zone_size[MAX_NR_ZONES][NR_LRU_LISTS]; + atomic_long_t lru_zone_size[MAX_NR_ZONES][NR_LRU_LISTS]; struct mem_cgroup_reclaim_iter iter; /* @@ -902,10 +902,15 @@ static inline unsigned long mem_cgroup_get_zone_lru_size(struct lruvec *lruvec, enum lru_list lru, int zone_idx) { + long val; struct mem_cgroup_per_node *mz; mz = container_of(lruvec, struct mem_cgroup_per_node, lruvec); - return READ_ONCE(mz->lru_zone_size[zone_idx][lru]); + val = atomic_long_read(&mz->lru_zone_size[zone_idx][lru]); + if (WARN_ON_ONCE(val < 0)) + return 0; + + return val; } void __mem_cgroup_handle_over_high(gfp_t gfp_mask); diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 11b85f4b6828..a7572ded56c9 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -1529,28 +1529,12 @@ void mem_cgroup_update_lru_size(struct lruvec *lruvec, enum lru_list lru, int zid, long nr_pages) { struct mem_cgroup_per_node *mz; - unsigned long *lru_size; - long size; if (mem_cgroup_disabled()) return; mz = container_of(lruvec, struct mem_cgroup_per_node, lruvec); - lru_size = &mz->lru_zone_size[zid][lru]; - - if (nr_pages < 0) - *lru_size += nr_pages; - - size = *lru_size; - if (WARN_ONCE(size < 0, - "%s(%p, %d, %ld): lru_size %ld\n", - __func__, lruvec, lru, nr_pages, size)) { - VM_BUG_ON(1); - *lru_size = 0; - } - - if (nr_pages > 0) - *lru_size += nr_pages; + atomic_long_add(nr_pages, &mz->lru_zone_size[zid][lru]); } /** -- 2.55.0 From: Kairui Song Instead of doing bit ops on folio->flags.f, introduce helpers for adjusting a folio's refs and generation info, making the code easier to debug and understand. No functional change is intended: some combined atomic operations are split into two, which only creates harmless transient states. There is no measurable performance impact, and some paths even look slightly better in the generated assembly. Signed-off-by: Kairui Song --- include/linux/mm_inline.h | 76 ++++++++++++++++++++++++++++++++++++++++++----- include/linux/mmzone.h | 1 + mm/folio.c | 19 +++++++----- mm/vmscan.c | 61 ++++++++++++++++++++----------------- 4 files changed, 114 insertions(+), 43 deletions(-) diff --git a/include/linux/mm_inline.h b/include/linux/mm_inline.h index 621c8653d8f7..edfaf2661812 100644 --- a/include/linux/mm_inline.h +++ b/include/linux/mm_inline.h @@ -142,10 +142,42 @@ static inline int lru_tier_from_refs(int refs, bool workingset) return workingset ? MAX_NR_TIERS - 1 : order_base_2(refs); } -static inline int folio_lru_refs(const struct folio *folio) +/** + * lru_gen_from_flags - Return the LRU generation number from folio flags. + * @flags: folio flags + * + * Returns: A number between 0 and (MAX_NR_GENS - 1), inclusive. Returns + * -1 if the flags indicate the folio is off the list (e.g., isolated). + */ +static inline int lru_gen_from_flags(unsigned long flags) +{ + int gen = ((flags & LRU_GEN_MASK) >> LRU_GEN_PGOFF); + + BUILD_BUG_ON(LRU_GEN_MASK & LRU_REFS_MASK); + gen -= 1; + VM_WARN_ON_ONCE(gen != -1 && gen >= MAX_NR_GENS); + return gen; +} + +/** + * lru_gen_set_flags - Set the LRU generation number to specified folio flags. + * @flags: pointer to the folio flags + * @gen: generation number, between 0 and (MAX_NR_GENS - 1), inclusive. + */ +static inline void lru_gen_set_flags(unsigned long *flags, int gen) { - unsigned long flags = READ_ONCE(folio->flags.f); + VM_WARN_ON_ONCE(gen >= MAX_NR_GENS || gen < 0); + + *flags &= ~LRU_GEN_MASK; + *flags |= (gen + 1UL) << LRU_GEN_PGOFF; +} +/** + * lru_refs_from_flags - Return LRU referenced / access count from folio flags. + * @flags: folio flags + */ +static inline int lru_refs_from_flags(unsigned long flags) +{ if (!(flags & BIT(PG_referenced))) return 0; /* @@ -155,11 +187,40 @@ static inline int folio_lru_refs(const struct folio *folio) return ((flags & LRU_REFS_MASK) >> LRU_REFS_PGOFF) + 1; } -static inline int folio_lru_gen(const struct folio *folio) +/** + * lru_refs_set_flags - Set the LRU referenced / access count to specified folio flags. + * @flags: pointer to the folio flags + * @refs: referenced / access count number, between 0 and LRU_REFS_MAX, inclusive. + */ +static inline void lru_refs_set_flags(unsigned long *flags, unsigned int refs) +{ + VM_WARN_ON_ONCE(refs > LRU_REFS_MAX); + BUILD_BUG_ON(LRU_REFS_MAX != (LRU_REFS_MASK >> LRU_REFS_PGOFF) + 1); + + *flags &= ~LRU_REFS_FLAGS; + if (!refs) + return; + *flags |= (BIT(PG_referenced) | ((refs - 1UL) << LRU_REFS_PGOFF)); +} + +static inline int folio_lru_refs(const struct folio *folio) { - unsigned long flags = READ_ONCE(folio->flags.f); + return lru_refs_from_flags(READ_ONCE(*const_folio_flags(folio, 0))); +} + +static inline void folio_set_lru_refs(struct folio *folio, unsigned int refs) +{ + unsigned long new_flags, old_flags = READ_ONCE(*folio_flags(folio, 0)); + + do { + new_flags = old_flags; + lru_refs_set_flags(&new_flags, refs); + } while (!try_cmpxchg(folio_flags(folio, 0), &old_flags, new_flags)); +} - return ((flags & LRU_GEN_MASK) >> LRU_GEN_PGOFF) - 1; +static inline int folio_lru_gen(const struct folio *folio) +{ + return lru_gen_from_flags(READ_ONCE(*const_folio_flags(folio, 0))); } static inline bool lru_gen_is_active(const struct lruvec *lruvec, int gen) @@ -270,7 +331,7 @@ static inline bool lru_gen_add_folio(struct lruvec *lruvec, struct folio *folio, gen = lru_gen_from_seq(seq); flags = (gen + 1UL) << LRU_GEN_PGOFF; /* see the comment on MIN_NR_GENS about PG_active */ - set_mask_bits(&folio->flags.f, LRU_GEN_MASK | BIT(PG_active), flags); + set_mask_bits(folio_flags(folio, 0), LRU_GEN_MASK | BIT(PG_active), flags); lru_gen_update_size(lruvec, folio, -1, gen); /* for folio_rotate_reclaimable() */ @@ -295,7 +356,7 @@ static inline bool lru_gen_del_folio(struct lruvec *lruvec, struct folio *folio, /* for folio_migrate_flags() */ flags = !reclaiming && lru_gen_is_active(lruvec, gen) ? BIT(PG_active) : 0; - flags = set_mask_bits(&folio->flags.f, LRU_GEN_MASK, flags); + flags = set_mask_bits(folio_flags(folio, 0), LRU_GEN_MASK, flags); gen = ((flags & LRU_GEN_MASK) >> LRU_GEN_PGOFF) - 1; lru_gen_update_size(lruvec, folio, gen, -1); @@ -339,7 +400,6 @@ static inline bool lru_gen_del_folio(struct lruvec *lruvec, struct folio *folio, static inline void folio_migrate_refs(struct folio *new, const struct folio *old) { - } #endif /* CONFIG_LRU_GEN */ diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h index 94f9c3ff5416..c9ecf370cd9f 100644 --- a/include/linux/mmzone.h +++ b/include/linux/mmzone.h @@ -497,6 +497,7 @@ enum lruvec_flags { #define LRU_GEN_MASK ((BIT(LRU_GEN_WIDTH) - 1) << LRU_GEN_PGOFF) #define LRU_REFS_MASK ((BIT(LRU_REFS_WIDTH) - 1) << LRU_REFS_PGOFF) +#define LRU_REFS_MAX BIT(LRU_REFS_WIDTH) /* * For folios accessed multiple times through file descriptors, diff --git a/mm/folio.c b/mm/folio.c index c02dcea9c03c..a932059057ac 100644 --- a/mm/folio.c +++ b/mm/folio.c @@ -353,26 +353,28 @@ static void __lru_cache_activate_folio(struct folio *folio) static void lru_gen_inc_refs(struct folio *folio) { - unsigned long new_flags, old_flags = READ_ONCE(folio->flags.f); + unsigned long new_flags, old_flags = READ_ONCE(*folio_flags(folio, 0)); + int refs; if (folio_test_unevictable(folio)) return; /* see the comment on LRU_REFS_FLAGS */ - if (!folio_test_referenced(folio)) { - set_mask_bits(&folio->flags.f, LRU_REFS_MASK, BIT(PG_referenced)); + if (!folio_lru_refs(folio)) { + folio_set_lru_refs(folio, 1); return; } do { - if ((old_flags & LRU_REFS_MASK) == LRU_REFS_MASK) { + new_flags = old_flags; + refs = lru_refs_from_flags(old_flags); + if (refs == LRU_REFS_MAX) { if (!folio_test_workingset(folio)) folio_set_workingset(folio); return; } - - new_flags = old_flags + BIT(LRU_REFS_PGOFF); - } while (!try_cmpxchg(&folio->flags.f, &old_flags, new_flags)); + lru_refs_set_flags(&new_flags, refs + 1); + } while (!try_cmpxchg(folio_flags(folio, 0), &old_flags, new_flags)); } static bool lru_gen_clear_refs(struct folio *folio) @@ -384,7 +386,8 @@ static bool lru_gen_clear_refs(struct folio *folio) if (gen < 0) return true; - set_mask_bits(&folio->flags.f, LRU_REFS_FLAGS | BIT(PG_workingset), 0); + folio_set_lru_refs(folio, 0); + folio_clear_workingset(folio); rcu_read_lock(); seq = READ_ONCE(folio_lruvec(folio)->lrugen.min_seq[type]); diff --git a/mm/vmscan.c b/mm/vmscan.c index 73a81b4a3e16..9ee9f8dc6805 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -843,19 +843,22 @@ static bool lru_gen_set_refs(struct folio *folio, const vma_flags_t *vma_flags) if (!folio_test_referenced(folio) && !folio_test_workingset(folio)) { /* Activate file-backed executable folios after first usage. */ if (is_exec_file_folio(folio, vma_flags)) { - set_mask_bits(&folio->flags.f, LRU_REFS_FLAGS, BIT(PG_workingset)); + folio_set_workingset(folio); + folio_set_lru_refs(folio, 0); return true; } - set_mask_bits(&folio->flags.f, LRU_REFS_MASK, BIT(PG_referenced)); + folio_set_lru_refs(folio, 1); return false; } /* Promote on second access */ - if (folio_lru_refs(folio) > 1) - set_mask_bits(&folio->flags.f, LRU_REFS_FLAGS, BIT(PG_workingset)); - else + if (folio_lru_refs(folio) > 1) { + folio_set_workingset(folio); + folio_set_lru_refs(folio, 0); + } else { folio_mark_accessed(folio); + } return true; } #else @@ -3266,11 +3269,10 @@ static bool positive_ctrl_err(struct ctrl_pos *sp, struct ctrl_pos *pv) ******************************************************************************/ /* promote pages accessed through page tables */ -static int folio_update_gen(struct folio *folio, int gen, const vma_flags_t *vma_flags) +static int folio_update_gen(struct folio *folio, int new_gen, const vma_flags_t *vma_flags) { - unsigned long new_flags, old_flags = READ_ONCE(folio->flags.f); - - VM_WARN_ON_ONCE(gen >= MAX_NR_GENS); + unsigned long new_flags, old_flags = READ_ONCE(*folio_flags(folio, 0)); + int old_gen; /* * See the comment on LRU_REFS_FLAGS, and activate file-backed @@ -3279,20 +3281,24 @@ static int folio_update_gen(struct folio *folio, int gen, const vma_flags_t *vma */ if (!folio_test_referenced(folio) && !folio_test_workingset(folio) && !is_exec_file_folio(folio, vma_flags)) { - set_mask_bits(&folio->flags.f, LRU_REFS_MASK, BIT(PG_referenced)); + folio_set_lru_refs(folio, 1); return -1; } do { + old_gen = lru_gen_from_flags(old_flags); + new_flags = old_flags; + /* lru_gen_del_folio() has isolated this page? */ - if (!(old_flags & LRU_GEN_MASK)) - return -1; + if (old_gen < 0) + break; - new_flags = old_flags & ~(LRU_GEN_MASK | LRU_REFS_FLAGS); - new_flags |= ((gen + 1UL) << LRU_GEN_PGOFF) | BIT(PG_workingset); - } while (!try_cmpxchg(&folio->flags.f, &old_flags, new_flags)); + lru_gen_set_flags(&new_flags, new_gen); + lru_refs_set_flags(&new_flags, 0); + new_flags |= BIT(PG_workingset); + } while (!try_cmpxchg(folio_flags(folio, 0), &old_flags, new_flags)); - return ((old_flags & LRU_GEN_MASK) >> LRU_GEN_PGOFF) - 1; + return old_gen; } /* protect pages accessed multiple times through file descriptors */ @@ -3301,21 +3307,20 @@ static int folio_inc_gen(struct lruvec *lruvec, struct folio *folio) int type = folio_is_file_lru(folio); struct lru_gen_folio *lrugen = &lruvec->lrugen; int new_gen, old_gen = lru_gen_from_seq(lrugen->min_seq[type]); - unsigned long new_flags, old_flags = READ_ONCE(folio->flags.f); - - VM_WARN_ON_ONCE_FOLIO(!(old_flags & LRU_GEN_MASK), folio); + unsigned long new_flags, old_flags = READ_ONCE(*folio_flags(folio, 0)); do { - new_gen = ((old_flags & LRU_GEN_MASK) >> LRU_GEN_PGOFF) - 1; + new_gen = lru_gen_from_flags(old_flags); + /* folio_update_gen() has promoted this page? */ if (new_gen >= 0 && new_gen != old_gen) return new_gen; + new_flags = old_flags; new_gen = (old_gen + 1) % MAX_NR_GENS; - - new_flags = old_flags & ~(LRU_GEN_MASK | LRU_REFS_FLAGS); - new_flags |= (new_gen + 1UL) << LRU_GEN_PGOFF; - } while (!try_cmpxchg(&folio->flags.f, &old_flags, new_flags)); + lru_gen_set_flags(&new_flags, new_gen); + lru_refs_set_flags(&new_flags, 0); + } while (!try_cmpxchg(folio_flags(folio, 0), &old_flags, new_flags)); lru_gen_update_size(lruvec, folio, old_gen, new_gen); @@ -4716,7 +4721,7 @@ static bool isolate_folio(struct lruvec *lruvec, struct folio *folio, struct sca /* see the comment on LRU_REFS_FLAGS */ if (!folio_test_referenced(folio)) - set_mask_bits(&folio->flags.f, LRU_REFS_MASK, 0); + folio_set_lru_refs(folio, 0); success = lru_gen_del_folio(lruvec, folio, true); VM_WARN_ON_ONCE_FOLIO(!success, folio); @@ -4932,8 +4937,10 @@ static int evict_folios(unsigned long nr_to_scan, struct lruvec *lruvec, } /* don't add rejected folios to the oldest generation */ - if (lru_gen_folio_seq(lruvec, folio, false) == min_seq[type]) - set_mask_bits(&folio->flags.f, LRU_REFS_FLAGS, BIT(PG_active)); + if (lru_gen_folio_seq(lruvec, folio, false) == min_seq[type]) { + folio_set_lru_refs(folio, 0); + folio_set_active(folio); + } } move_folios_to_lru(&list); -- 2.55.0 From: Kairui Song folio_migrate_flags() copies PG_referenced separately from the MGLRU refs counter, which folio_migrate_refs() transfers. Yet under MGLRU, PG_referenced and the refs counter bits together describe the referenced status of a folio. Consolidate the two: rename folio_migrate_refs() to folio_migrate_lru_refs() and let it copy the complete referenced status, i.e., the MGLRU refs count including PG_referenced, or just PG_referenced for the active/inactive LRU. Drop the open-coded PG_referenced copy so the referenced status is transferred in one place. No behavior change is intended: under the active/inactive LRU the extra bits are unused, so operating on them is a noop. Transfer the reference state first, before the destination folio is marked uptodate, so a concurrent lockless reader cannot have its reference update overwritten by the copy. Reviewed-by: Baoquan He Signed-off-by: Kairui Song --- include/linux/mm_inline.h | 20 +++++++++++++++----- mm/migrate.c | 6 +++--- 2 files changed, 18 insertions(+), 8 deletions(-) diff --git a/include/linux/mm_inline.h b/include/linux/mm_inline.h index edfaf2661812..7f91a89b5ba3 100644 --- a/include/linux/mm_inline.h +++ b/include/linux/mm_inline.h @@ -365,11 +365,19 @@ static inline bool lru_gen_del_folio(struct lruvec *lruvec, struct folio *folio, return true; } -static inline void folio_migrate_refs(struct folio *new, const struct folio *old) +/** + * folio_migrate_lru_refs - copy the reference state to a new folio + * @new: the destination folio + * @old: the source folio + * + * Transfer the reference state to @new during migration: the MGLRU + * refs count, including PG_referenced, or just PG_referenced for the + * active/inactive LRU. + */ +static inline void folio_migrate_lru_refs(struct folio *new, const struct folio *old) { - unsigned long refs = READ_ONCE(old->flags.f) & LRU_REFS_MASK; - - set_mask_bits(&new->flags.f, LRU_REFS_MASK, refs); + BUILD_BUG_ON(LRU_REFS_MASK & BIT(PG_referenced)); + folio_set_lru_refs(new, folio_lru_refs(old)); } #else /* !CONFIG_LRU_GEN */ @@ -398,8 +406,10 @@ static inline bool lru_gen_del_folio(struct lruvec *lruvec, struct folio *folio, return false; } -static inline void folio_migrate_refs(struct folio *new, const struct folio *old) +static inline void folio_migrate_lru_refs(struct folio *new, const struct folio *old) { + if (folio_test_referenced(old)) + folio_set_referenced(new); } #endif /* CONFIG_LRU_GEN */ diff --git a/mm/migrate.c b/mm/migrate.c index 15b45832bcfa..a369d0c95c38 100644 --- a/mm/migrate.c +++ b/mm/migrate.c @@ -776,8 +776,9 @@ void folio_migrate_flags(struct folio *newfolio, struct folio *folio) { int cpupid; - if (folio_test_referenced(folio)) - folio_set_referenced(newfolio); + /* Copy the reference state, including PG_referenced */ + folio_migrate_lru_refs(newfolio, folio); + if (folio_test_uptodate(folio)) folio_mark_uptodate(newfolio); if (folio_test_clear_active(folio)) { @@ -807,7 +808,6 @@ void folio_migrate_flags(struct folio *newfolio, struct folio *folio) if (folio_test_idle(folio)) folio_set_idle(newfolio); - folio_migrate_refs(newfolio, folio); /* * Copy NUMA information to the new page, to prevent over-eager * future migrations of this same page. -- 2.55.0 From: Kairui Song walk_pte_range(), walk_pmd_range_locked(), and lru_gen_look_around() each read lrugen->max_seq to compute the target generation used by walk_update_folio(), then pass it as a parameter. Move the read into walk_update_folio() itself so the callers no longer need to compute or pass the value. The max_seq read now happens once per folio update rather than once per walk range, so folios always get promoted to the current youngest generation. Reviewed-by: Baoquan He Reviewed-by: Baolin Wang Reviewed-by: Ridong Chen Signed-off-by: Kairui Song --- mm/vmscan.c | 29 ++++++++++++----------------- 1 file changed, 12 insertions(+), 17 deletions(-) diff --git a/mm/vmscan.c b/mm/vmscan.c index 9ee9f8dc6805..52ef4c3705bd 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -3517,13 +3517,15 @@ static bool suitable_to_scan(int total, int young) } static void walk_update_folio(struct lru_gen_mm_walk *walk, struct vm_area_struct *vma, - struct folio *folio, int new_gen, bool dirty) + struct lruvec *lruvec, struct folio *folio, bool dirty) { - int old_gen; + int new_gen, old_gen; if (!folio) return; + new_gen = lru_gen_from_seq(READ_ONCE(lruvec->lrugen.max_seq)); + if (dirty && !folio_test_dirty(folio) && !(folio_test_anon(folio) && folio_test_swapbacked(folio) && !folio_test_swapcache(folio))) @@ -3554,8 +3556,6 @@ static bool walk_pte_range(pmd_t *pmd, unsigned long start, unsigned long end, struct lru_gen_mm_walk *walk = args->private; struct mem_cgroup *memcg = lruvec_memcg(walk->lruvec); struct pglist_data *pgdat = lruvec_pgdat(walk->lruvec); - DEFINE_MAX_SEQ(walk->lruvec); - int gen = lru_gen_from_seq(max_seq); unsigned int nr; pmd_t pmdval; @@ -3606,7 +3606,7 @@ static bool walk_pte_range(pmd_t *pmd, unsigned long start, unsigned long end, continue; if (last != folio) { - walk_update_folio(walk, args->vma, last, gen, dirty); + walk_update_folio(walk, args->vma, walk->lruvec, last, dirty); last = folio; dirty = false; @@ -3619,7 +3619,7 @@ static bool walk_pte_range(pmd_t *pmd, unsigned long start, unsigned long end, walk->mm_stats[MM_LEAF_YOUNG] += nr; } - walk_update_folio(walk, args->vma, last, gen, dirty); + walk_update_folio(walk, args->vma, walk->lruvec, last, dirty); last = NULL; if (i < PTRS_PER_PTE && get_next_vma(PMD_MASK, PAGE_SIZE, args, &start, &end)) @@ -3642,8 +3642,6 @@ static void walk_pmd_range_locked(pud_t *pud, unsigned long addr, struct vm_area struct lru_gen_mm_walk *walk = args->private; struct mem_cgroup *memcg = lruvec_memcg(walk->lruvec); struct pglist_data *pgdat = lruvec_pgdat(walk->lruvec); - DEFINE_MAX_SEQ(walk->lruvec); - int gen = lru_gen_from_seq(max_seq); VM_WARN_ON_ONCE(pud_leaf(*pud)); @@ -3697,7 +3695,7 @@ static void walk_pmd_range_locked(pud_t *pud, unsigned long addr, struct vm_area goto next; if (last != folio) { - walk_update_folio(walk, vma, last, gen, dirty); + walk_update_folio(walk, vma, walk->lruvec, last, dirty); last = folio; dirty = false; @@ -3711,7 +3709,7 @@ static void walk_pmd_range_locked(pud_t *pud, unsigned long addr, struct vm_area i = i > MIN_LRU_BATCH ? 0 : find_next_bit(bitmap, MIN_LRU_BATCH, i) + 1; } while (i <= MIN_LRU_BATCH); - walk_update_folio(walk, vma, last, gen, dirty); + walk_update_folio(walk, vma, walk->lruvec, last, dirty); lazy_mmu_mode_disable(); spin_unlock(ptl); @@ -4275,8 +4273,6 @@ bool lru_gen_look_around(struct page_vma_mapped_walk *pvmw, unsigned int nr) struct pglist_data *pgdat = folio_pgdat(folio); struct lruvec *lruvec; struct lru_gen_mm_state *mm_state; - unsigned long max_seq; - int gen; lockdep_assert_held(pvmw->ptl); VM_WARN_ON_ONCE_FOLIO(folio_test_lru(folio), folio); @@ -4313,8 +4309,6 @@ bool lru_gen_look_around(struct page_vma_mapped_walk *pvmw, unsigned int nr) memcg = get_mem_cgroup_from_folio(folio); lruvec = mem_cgroup_lruvec(memcg, pgdat); - max_seq = READ_ONCE((lruvec)->lrugen.max_seq); - gen = lru_gen_from_seq(max_seq); mm_state = get_mm_state(lruvec); lazy_mmu_mode_enable(); @@ -4346,7 +4340,7 @@ bool lru_gen_look_around(struct page_vma_mapped_walk *pvmw, unsigned int nr) continue; if (last != folio) { - walk_update_folio(walk, vma, last, gen, dirty); + walk_update_folio(walk, vma, lruvec, last, dirty); last = folio; dirty = false; @@ -4358,13 +4352,14 @@ bool lru_gen_look_around(struct page_vma_mapped_walk *pvmw, unsigned int nr) young += nr; } - walk_update_folio(walk, vma, last, gen, dirty); + walk_update_folio(walk, vma, lruvec, last, dirty); lazy_mmu_mode_disable(); /* feedback from rmap walkers to page table walkers */ if (mm_state && suitable_to_scan(i, young)) - update_bloom_filter(mm_state, max_seq, pvmw->pmd); + update_bloom_filter(mm_state, READ_ONCE(lruvec->lrugen.max_seq), + pvmw->pmd); mem_cgroup_put(memcg); -- 2.55.0 From: Kairui Song read_ctrl_pos() encodes the tier range in a single "tier" parameter via "tier % MAX_NR_TIERS" as the start and "min(tier, MAX_NR_TIERS-1)" as the end. This is hard to follow, maintain, or extend. Tier values 0..3 select a single tier, while tier == MAX_NR_TIERS selects the full range. Replace it with explicit (tier_min, tier_max) parameters using a closed [tier_min, tier_max] interval, and add LRU_TIER_MIN and LRU_TIER_MAX for the tier bounds. The call sites now become self-documenting: - get_tier_idx: (LRU_TIER_MIN, LRU_TIER_MIN) for the first tier, (tier, tier) for each subsequent tier - get_type_to_scan: (LRU_TIER_MIN, LRU_TIER_MAX) for the full range No functional change. Reviewed-by: Baolin Wang Reviewed-by: Baoquan He Reviewed-by: Barry Song Signed-off-by: Kairui Song --- include/linux/mmzone.h | 2 ++ mm/vmscan.c | 18 ++++++++++-------- 2 files changed, 12 insertions(+), 8 deletions(-) diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h index c9ecf370cd9f..9b27cf53bdbc 100644 --- a/include/linux/mmzone.h +++ b/include/linux/mmzone.h @@ -492,6 +492,8 @@ enum lruvec_flags { * folio->flags, masked by LRU_REFS_MASK. */ #define MAX_NR_TIERS 4U +#define LRU_TIER_MIN 0U +#define LRU_TIER_MAX (MAX_NR_TIERS - 1) #ifndef __GENERATING_BOUNDS_H diff --git a/mm/vmscan.c b/mm/vmscan.c index 52ef4c3705bd..cdc2e44875da 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -3198,8 +3198,8 @@ struct ctrl_pos { int gain; }; -static void read_ctrl_pos(struct lruvec *lruvec, int type, int tier, int gain, - struct ctrl_pos *pos) +static void read_ctrl_pos(struct lruvec *lruvec, int type, int tier_min, + int tier_max, int gain, struct ctrl_pos *pos) { int i; struct lru_gen_folio *lrugen = &lruvec->lrugen; @@ -3208,7 +3208,7 @@ static void read_ctrl_pos(struct lruvec *lruvec, int type, int tier, int gain, pos->gain = gain; pos->refaulted = pos->total = 0; - for (i = tier % MAX_NR_TIERS; i <= min(tier, MAX_NR_TIERS - 1); i++) { + for (i = tier_min; i <= tier_max; i++) { pos->refaulted += lrugen->avg_refaulted[type][i] + atomic_long_read(&lrugen->refaulted[hist][type][i]); pos->total += lrugen->avg_total[type][i] + @@ -4809,9 +4809,9 @@ static int get_tier_idx(struct lruvec *lruvec, int type) * This value is chosen because any other tier would have at least twice * as many refaults as the first tier. */ - read_ctrl_pos(lruvec, type, 0, 2, &sp); - for (tier = 1; tier < MAX_NR_TIERS; tier++) { - read_ctrl_pos(lruvec, type, tier, 3, &pv); + read_ctrl_pos(lruvec, type, LRU_TIER_MIN, LRU_TIER_MIN, 2, &sp); + for (tier = LRU_TIER_MIN + 1; tier <= LRU_TIER_MAX; tier++) { + read_ctrl_pos(lruvec, type, tier, tier, 3, &pv); if (!positive_ctrl_err(&sp, &pv)) break; } @@ -4832,8 +4832,10 @@ static int get_type_to_scan(struct lruvec *lruvec, int swappiness) * Compare the sum of all tiers of anon with that of file to determine * which type to scan. */ - read_ctrl_pos(lruvec, LRU_GEN_ANON, MAX_NR_TIERS, swappiness, &sp); - read_ctrl_pos(lruvec, LRU_GEN_FILE, MAX_NR_TIERS, MAX_SWAPPINESS - swappiness, &pv); + read_ctrl_pos(lruvec, LRU_GEN_ANON, LRU_TIER_MIN, LRU_TIER_MAX, + swappiness, &sp); + read_ctrl_pos(lruvec, LRU_GEN_FILE, LRU_TIER_MIN, LRU_TIER_MAX, + MAX_SWAPPINESS - swappiness, &pv); return positive_ctrl_err(&sp, &pv); } -- 2.55.0 From: Kairui Song Each generation of MGLRU accounts anon and file folio numbers separately, and the page table walker updates each generation's counters in batch once the walk is done. The walker promotes a folio's generation with a cmpxchg on folio->flags, and update_batch_size() then reads the live flags again to pick the anon/file column to charge. The walk holds neither the lruvec lock nor the folio lock, so the type can flip between the cmpxchg and that read: the lazyfree path clears PG_swapbacked, and reclaim sets it back on a dirty lazyfree folio. The batched delta pair is then recorded in the wrong type column. Nothing reconciles it afterwards, permanently skewing lrugen->nr_pages and the reclaim budgets derived from it. Fix it by capturing the type from the flags snapshot the cmpxchg linearized against: folio_update_gen() returns the type of the state it transitioned from, and update_batch_size() accounts with that. A folio's type only changes while it is off the LRU list, inside a del/add pair under the lruvec lock, with the gen bits cleared in between. The generation and PG_swapbacked sit in the same folio->flags word, so the cmpxchg snapshot captures them together. Let G be the generation that snapshot captured (old_gen) and G' the one it wrote (new_gen); the CAS can land in only three places: - before the del: the folio is anon at G; the batch records anon G -> G', and the del later removes the folio from the anon counters; - between del and add: gen == -1, so folio_update_gen() returns -1 without touching the flags and no batch is recorded; the del/add pair accounts for the move alone; - after the add: the folio is file at the fresh generation the add charged; the batch records file, that gen -> G', matching that charge. Unlike the drift of lazy promotions, which sort_folio() repairs under the lruvec lock, the phantom deltas from before this fix land in a column the folio never occupies again, so nothing ever repairs them. Fixes: 018ee47f1489 ("mm: multi-gen LRU: exploit locality in rmap") Signed-off-by: Kairui Song --- include/linux/mm_inline.h | 7 ++++++- mm/vmscan.c | 13 +++++++------ 2 files changed, 13 insertions(+), 7 deletions(-) diff --git a/include/linux/mm_inline.h b/include/linux/mm_inline.h index 7f91a89b5ba3..8cf82989ec26 100644 --- a/include/linux/mm_inline.h +++ b/include/linux/mm_inline.h @@ -10,6 +10,11 @@ #include #include +static inline int folio_flags_is_file_lru(const unsigned long *flags) +{ + return !test_bit(PG_swapbacked, flags); +} + /** * folio_is_file_lru - Should the folio be on a file LRU or anon LRU? * @folio: The folio to test. @@ -27,7 +32,7 @@ */ static inline int folio_is_file_lru(const struct folio *folio) { - return !folio_test_swapbacked(folio); + return folio_flags_is_file_lru(const_folio_flags(folio, 0)); } static __always_inline void __update_lru_size(struct lruvec *lruvec, diff --git a/mm/vmscan.c b/mm/vmscan.c index cdc2e44875da..6fda948aecf2 100644 --- a/mm/vmscan.c +++ b/mm/vmscan.c @@ -3269,7 +3269,8 @@ static bool positive_ctrl_err(struct ctrl_pos *sp, struct ctrl_pos *pv) ******************************************************************************/ /* promote pages accessed through page tables */ -static int folio_update_gen(struct folio *folio, int new_gen, const vma_flags_t *vma_flags) +static int folio_update_gen(struct folio *folio, int new_gen, int *is_file, + const vma_flags_t *vma_flags) { unsigned long new_flags, old_flags = READ_ONCE(*folio_flags(folio, 0)); int old_gen; @@ -3298,6 +3299,7 @@ static int folio_update_gen(struct folio *folio, int new_gen, const vma_flags_t new_flags |= BIT(PG_workingset); } while (!try_cmpxchg(folio_flags(folio, 0), &old_flags, new_flags)); + *is_file = folio_flags_is_file_lru(&old_flags); return old_gen; } @@ -3328,9 +3330,8 @@ static int folio_inc_gen(struct lruvec *lruvec, struct folio *folio) } static void update_batch_size(struct lru_gen_mm_walk *walk, struct folio *folio, - int old_gen, int new_gen) + int old_gen, int new_gen, int type) { - int type = folio_is_file_lru(folio); int zone = folio_zonenum(folio); int delta = folio_nr_pages(folio); @@ -3519,7 +3520,7 @@ static bool suitable_to_scan(int total, int young) static void walk_update_folio(struct lru_gen_mm_walk *walk, struct vm_area_struct *vma, struct lruvec *lruvec, struct folio *folio, bool dirty) { - int new_gen, old_gen; + int new_gen, old_gen, file; if (!folio) return; @@ -3532,9 +3533,9 @@ static void walk_update_folio(struct lru_gen_mm_walk *walk, struct vm_area_struc folio_mark_dirty(folio); if (walk) { - old_gen = folio_update_gen(folio, new_gen, &vma->flags); + old_gen = folio_update_gen(folio, new_gen, &file, &vma->flags); if (old_gen >= 0 && old_gen != new_gen) - update_batch_size(walk, folio, old_gen, new_gen); + update_batch_size(walk, folio, old_gen, new_gen, file); } else if (lru_gen_set_refs(folio, &vma->flags)) { old_gen = folio_lru_gen(folio); if (old_gen >= 0 && old_gen != new_gen) -- 2.55.0