An xswap slot can be written out to a physical device. The physical slot then holds its data. That slot has no folio in the swap cache. Record the xswap entry in the physical slot's swap table entry, so the physical side can find it again. A swap table entry can tell the type by its low three bits. Now 0b100 is left for extension. Use it to point at the owner. The xswap type and offset go in the bits above. Two paths hand out or cache a slot. They know only about the other two kinds. Teach them to skip a pointer entry. cluster_scan_range() must not hand it out as free. __swap_cache_add_check() must not put a folio over it. Readahead on a physical device can otherwise walk into it. Signed-off-by: Baoquan He --- mm/swap_state.c | 6 +++++- mm/swap_table.h | 53 +++++++++++++++++++++++++++++++++++++++++++++++++ mm/swapfile.c | 3 +++ 3 files changed, 61 insertions(+), 1 deletion(-) diff --git a/mm/swap_state.c b/mm/swap_state.c index 8bba3e533b28..135d9573a9dc 100644 --- a/mm/swap_state.c +++ b/mm/swap_state.c @@ -179,6 +179,9 @@ static int __swap_cache_add_check(struct swap_cluster_info *ci, return -ENOENT; ci_off = swp_cluster_offset(targ_entry); old_tb = __swap_table_get(ci, ci_off); + /* Readahead of a physical device can hit an xswap-backing slot. */ + if (swp_tb_is_pointer(old_tb)) + return -ENOENT; if (swp_tb_is_folio(old_tb)) return -EEXIST; if (!__swp_tb_get_count(old_tb)) @@ -196,7 +199,8 @@ static int __swap_cache_add_check(struct swap_cluster_info *ci, ci_end = ci_off + nr; do { old_tb = __swap_table_get(ci, ci_off); - if (unlikely(swp_tb_is_folio(old_tb) || + if (unlikely(swp_tb_is_pointer(old_tb) || + swp_tb_is_folio(old_tb) || !__swp_tb_get_count(old_tb) || is_zero != __swap_table_test_zero(ci, ci_off) || (memcg_id && *memcg_id != __swap_cgroup_get(ci, ci_off)))) diff --git a/mm/swap_table.h b/mm/swap_table.h index e6613e62f8d0..524d397a995a 100644 --- a/mm/swap_table.h +++ b/mm/swap_table.h @@ -81,6 +81,59 @@ struct swap_memcg_table { /* Bad slot: ends with 0b1000 and rests of bits are all 1 */ #define SWP_TB_BAD ((~0UL) << 3) +#ifdef CONFIG_XSWAP +/* + * Pointer-tagged swap table entry: the reverse map from a physical slot to + * the xswap entry it backs. Layout: + * + * Pointer: | xswap_type(8) | xswap_offset(53) |100| + * + * The low three bits identify the entry: 0b000 is a free or bad slot, a + * shadow entry ends in 0b01 and a cached folio in 0b10, which leaves + * 0b100 as the only marker still free. + */ +#define SWP_TB_PTR_MARK 0b100UL +#define SWP_TB_PTR_OFF_BITS 53 +#define SWP_TB_PTR_TYPE_BITS 8 +#define SWP_TB_PTR_OFF_SHIFT 3 +#define SWP_TB_PTR_TYPE_SHIFT (SWP_TB_PTR_OFF_SHIFT + \ + SWP_TB_PTR_OFF_BITS) +#define SWP_TB_PTR_OFF_MASK ((1UL << SWP_TB_PTR_OFF_BITS) - 1) +#define SWP_TB_PTR_TYPE_MASK ((1UL << SWP_TB_PTR_TYPE_BITS) - 1) + +static inline bool swp_tb_is_pointer(unsigned long swp_tb) +{ + return (swp_tb & (BIT(3) - 1)) == SWP_TB_PTR_MARK; +} + +static inline unsigned long xswap_entry_to_rmap(swp_entry_t entry) +{ + unsigned long type = swp_type(entry); + unsigned long off = swp_offset(entry); + + VM_WARN_ON_ONCE(type > SWP_TB_PTR_TYPE_MASK); + VM_WARN_ON_ONCE(off > SWP_TB_PTR_OFF_MASK); + return (type << SWP_TB_PTR_TYPE_SHIFT) | + (off << SWP_TB_PTR_OFF_SHIFT) | + SWP_TB_PTR_MARK; +} + +static inline swp_entry_t xswap_rmap_to_entry(unsigned long swp_tb) +{ + unsigned long type = (swp_tb >> SWP_TB_PTR_TYPE_SHIFT) & + SWP_TB_PTR_TYPE_MASK; + unsigned long off = (swp_tb >> SWP_TB_PTR_OFF_SHIFT) & + SWP_TB_PTR_OFF_MASK; + + return swp_entry(type, off); +} +#else /* !CONFIG_XSWAP */ +static inline bool swp_tb_is_pointer(unsigned long swp_tb) +{ + return false; +} +#endif /* CONFIG_XSWAP */ + /* Macro for shadow offset calculation */ #define SWAP_COUNT_SHIFT SWP_TB_FLAGS_BITS diff --git a/mm/swapfile.c b/mm/swapfile.c index 29118082cd7a..ad6703aadd3d 100644 --- a/mm/swapfile.c +++ b/mm/swapfile.c @@ -1032,6 +1032,9 @@ static bool cluster_scan_range(struct swap_info_struct *si, swp_tb = __swap_table_get(ci, ci_off); if (swp_tb_is_null(swp_tb)) continue; + /* Backs an xswap entry: the slot is not allocatable. */ + if (swp_tb_is_pointer(swp_tb)) + return false; if (swp_tb_is_folio(swp_tb) && !__swp_tb_get_count(swp_tb)) { if (!vm_swap_full()) return false; -- 2.54.0