An xswap entry that zswap evicts had nowhere to go, so zswap_writeback_entry() refused to handle it. Send it to the physical backend instead. Reserve the backend before decompressing the entry, so a failed allocation wastes no work. A slot that was written out no longer has a zswap copy, so swap_read_folio() forwards its read to the physical entry rather than dropping it. Signed-off-by: Baoquan He --- mm/page_io.c | 28 ++++++++++++++++++++-------- mm/zswap.c | 20 +++++++++++++++----- 2 files changed, 35 insertions(+), 13 deletions(-) diff --git a/mm/page_io.c b/mm/page_io.c index 2b1387c87e51..32e426779dce 100644 --- a/mm/page_io.c +++ b/mm/page_io.c @@ -463,6 +463,7 @@ static bool swap_read_folio_zeromap(struct folio *folio) void swap_read_folio(struct swap_io_ctx *ctx, struct folio *folio) { struct swap_info_struct *sis = __swap_entry_to_info(folio->swap); + swp_entry_t entry = folio->swap; bool synchronous = sis->flags & SWP_SYNCHRONOUS_IO; bool workingset = folio_test_workingset(folio); unsigned long pflags; @@ -492,18 +493,29 @@ void swap_read_folio(struct swap_io_ctx *ctx, struct folio *folio) goto finish; if (unlikely(sis->flags & SWP_XSWAP)) { - /* - * An xswap entry only ever lives in zswap, so zswap_load() - * must have found it. Unlock and let the caller retry. - */ - WARN_ON_ONCE(1); - folio_unlock(folio); - goto finish; + struct swap_cluster_info *ci; + swp_entry_t phys = {}; + unsigned long offset = swp_offset(folio->swap); + + /* May have been written back; reuse the physical read path. */ + ci = __swap_offset_to_cluster(sis, offset); + if (ci) { + spin_lock(&ci->lock); + phys = xswap_slot_backend(ci, offset % SWAPFILE_CLUSTER); + spin_unlock(&ci->lock); + } + if (!phys.val) { + /* No backend and no zswap entry: data is gone. */ + folio_unlock(folio); + goto finish; + } + + entry = phys; } /* We have to read from slower devices. Increase zswap protection. */ zswap_folio_swapin(folio); - swap_add_folio(ctx, folio, folio->swap, READ); + swap_add_folio(ctx, folio, entry, READ); finish: if (workingset) { diff --git a/mm/zswap.c b/mm/zswap.c index b4ffd6fb83eb..24904b4c4dce 100644 --- a/mm/zswap.c +++ b/mm/zswap.c @@ -1011,6 +1011,8 @@ static int zswap_writeback_entry(struct zswap_entry *entry, struct mempolicy *mpol; struct swap_info_struct *si; struct swap_io_ctx ctx = {}; + swp_entry_t phys = {}; + bool is_xswap; int ret = 0; /* try to allocate swap cache folio */ @@ -1018,10 +1020,7 @@ static int zswap_writeback_entry(struct zswap_entry *entry, if (IS_ERR_OR_NULL(si)) return -ENOENT; - if (si->flags & SWP_XSWAP) { - put_swap_device(si); - return -EINVAL; - } + is_xswap = !!(si->flags & SWP_XSWAP); mpol = get_task_policy(current); folio = swap_cache_alloc_folio(swpentry, GFP_KERNEL, BIT(0), NULL, mpol, @@ -1053,6 +1052,15 @@ static int zswap_writeback_entry(struct zswap_entry *entry, goto out; } + /* Reserve the destination before dropping the zswap copy. */ + if (is_xswap) { + phys = xswap_backend_alloc(swpentry); + if (!phys.val) { + ret = -ENOMEM; + goto out; + } + } + if (!zswap_decompress(entry, folio)) { ret = -EIO; goto out; @@ -1073,11 +1081,13 @@ static int zswap_writeback_entry(struct zswap_entry *entry, folio_set_reclaim(folio); /* start writeback */ - __swap_writeout(&ctx, folio, folio->swap); + __swap_writeout(&ctx, folio, is_xswap ? phys : folio->swap); swap_write_submit(&ctx); out: if (ret) { + if (is_xswap && phys.val) + xswap_backend_free(swpentry, phys); swap_cache_del_folio(folio); folio_unlock(folio); } -- 2.54.0