The root cause of this issue is because multi-tables are updated in non atomic context. To be more specific, the issue could be triggerred as following: swap_alloc_fast swap_cluster_populate() /* Try a sleep allocation */ spin_unlock(&ci->lock); swap_cluster_alloc_table() rcu_assign_pointer(ci->table, table); ci = swap_cluster_lock(si, offset) cluster_is_usable(ci, order) if (!cluster_table_is_alloced(ci)) // ok alloc_swap_scan_cluster() cluster_scan_range() __swap_table_get() /* free table when more table allocation fails */ ci->memcg_table = kzalloc_obj(*ci->memcg_table, gfp); if (!ci->memcg_table) swap_cluster_free_table() rcu_assign_pointer(ci->table, NULL); table = rcu_dereference_check(ci->table, lockdep_is_held(&ci->lock)); atomic_long_read(&table[off]); // NULL dereference Fix the issue by updating allocated tables in atomic context. Fixes: 2fe7a6f5024b8 ("mm/memcg, swap: store cgroup id in cluster table directly") Signed-off-by: Kemeng Shi --- mm/swapfile.c | 21 ++++++++++++++++++++- 1 file changed, 20 insertions(+), 1 deletion(-) diff --git a/mm/swapfile.c b/mm/swapfile.c index 615d90867111..d29062d9c3cd 100644 --- a/mm/swapfile.c +++ b/mm/swapfile.c @@ -490,6 +490,20 @@ static int swap_cluster_alloc_table(struct swap_cluster_info *ci, gfp_t gfp) return 0; } +static void swap_cluster_copy_table(struct swap_cluster_info *d_ci, + struct swap_cluster_info *s_ci) +{ + rcu_assign_pointer(d_ci->table, rcu_access_pointer(s_ci->table)); + +#ifdef CONFIG_MEMCG + d_ci->memcg_table = s_ci->memcg_table; +#endif + +#if !SWAP_TABLE_HAS_ZEROFLAG + d_ci->zero_bitmap = s_ci->zero_bitmap; +#endif +} + /* * Sanity check to ensure nothing leaked, and the specified range is empty. * One special case is that bad slots can't be freed, so check the number of @@ -527,6 +541,7 @@ static struct swap_cluster_info * swap_cluster_populate(struct swap_info_struct *si, struct swap_cluster_info *ci) { + struct swap_cluster_info tmp_ci; int ret; /* @@ -552,7 +567,8 @@ swap_cluster_populate(struct swap_info_struct *si, spin_unlock(&si->global_cluster_lock); local_unlock(&percpu_swap_cluster.lock); - ret = swap_cluster_alloc_table(ci, __GFP_HIGH | __GFP_NOMEMALLOC | + tmp_ci = *ci; + ret = swap_cluster_alloc_table(&tmp_ci, __GFP_HIGH | __GFP_NOMEMALLOC | GFP_KERNEL); /* @@ -573,6 +589,9 @@ swap_cluster_populate(struct swap_info_struct *si, spin_unlock(&ci->lock); return NULL; } + + /* Update tables in atomic context */ + swap_cluster_copy_table(ci, &tmp_ci); return ci; } -- 2.36.1