AI Trajectory


Trajectory: (Expand/Collapse All)
Seq Timestamp Type Name Duration
0/0 2026/06/27 07:39 flow patch-triage
0m Args:
null

Results:
{
  "EnableConfigs": [
    "TRANSPARENT_HUGEPAGE"
  ],
  "FocusSymbols": [
    "madvise_hugepage_policy",
    "reserved_thp_charge",
    "reserved_thp_uncharge",
    "reserved_thp_pageblocks",
    "check_reserved_thp_alignment",
    "vma_alloc_anon_folio_pmd"
  ],
  "Reasoning": "The patch introduces a new MADV_RESERVED_THP madvise behavior and VM_RESERVED_THP VMA flag, modifying core memory management logic including page fault handling, mremap, and VMA splitting. This is a functional change to the kernel's memory management subsystem that should be fuzzed.",
  "WorthFuzzing": true
}

1/1 2026/06/27 07:39 action read-patch-diff
0m Args:
null

Results:
{
  "PatchDiff": "commit 88c2815565df6c04ad94ebfa480054c81836fe78\nAuthor: syz-cluster \u003ctriage@syzkaller.com\u003e\nDate:   Sat Jun 27 07:39:54 2026 +0000\n\n    syz-cluster: applied patch under review\n\ndiff --git a/arch/alpha/include/uapi/asm/mman.h b/arch/alpha/include/uapi/asm/mman.h\nindex 1e700468a68580..672a2fc3438614 100644\n--- a/arch/alpha/include/uapi/asm/mman.h\n+++ b/arch/alpha/include/uapi/asm/mman.h\n@@ -78,6 +78,8 @@\n \n #define MADV_COLLAPSE\t25\t\t/* Synchronous hugepage collapse */\n \n+#define MADV_RESERVED_THP 26\t\t/* Use reserved transparent hugepages */\n+\n #define MADV_GUARD_INSTALL 102\t\t/* fatal signal on access to range */\n #define MADV_GUARD_REMOVE 103\t\t/* unguard range */\n \ndiff --git a/arch/mips/include/uapi/asm/mman.h b/arch/mips/include/uapi/asm/mman.h\nindex b700dae28c482d..a94bf74dee21ca 100644\n--- a/arch/mips/include/uapi/asm/mman.h\n+++ b/arch/mips/include/uapi/asm/mman.h\n@@ -105,6 +105,8 @@\n \n #define MADV_COLLAPSE\t25\t\t/* Synchronous hugepage collapse */\n \n+#define MADV_RESERVED_THP 26\t\t/* Use reserved transparent hugepages */\n+\n #define MADV_GUARD_INSTALL 102\t\t/* fatal signal on access to range */\n #define MADV_GUARD_REMOVE 103\t\t/* unguard range */\n \ndiff --git a/arch/parisc/include/uapi/asm/mman.h b/arch/parisc/include/uapi/asm/mman.h\nindex b6a709506987e0..fe2fddefb6c5d9 100644\n--- a/arch/parisc/include/uapi/asm/mman.h\n+++ b/arch/parisc/include/uapi/asm/mman.h\n@@ -72,6 +72,8 @@\n \n #define MADV_COLLAPSE\t25\t\t/* Synchronous hugepage collapse */\n \n+#define MADV_RESERVED_THP 26\t\t/* Use reserved transparent hugepages */\n+\n #define MADV_HWPOISON     100\t\t/* poison a page for testing */\n #define MADV_SOFT_OFFLINE 101\t\t/* soft offline page for testing */\n \ndiff --git a/arch/xtensa/include/uapi/asm/mman.h b/arch/xtensa/include/uapi/asm/mman.h\nindex 99d4ccee7f6e8a..bb603530ba799f 100644\n--- a/arch/xtensa/include/uapi/asm/mman.h\n+++ b/arch/xtensa/include/uapi/asm/mman.h\n@@ -113,6 +113,8 @@\n \n #define MADV_COLLAPSE\t25\t\t/* Synchronous hugepage collapse */\n \n+#define MADV_RESERVED_THP 26\t\t/* Use reserved transparent hugepages */\n+\n #define MADV_GUARD_INSTALL 102\t\t/* fatal signal on access to range */\n #define MADV_GUARD_REMOVE 103\t\t/* unguard range */\n \ndiff --git a/fs/proc/task_mmu.c b/fs/proc/task_mmu.c\nindex d32408f7cd5ed2..65c4b2a61aeac7 100644\n--- a/fs/proc/task_mmu.c\n+++ b/fs/proc/task_mmu.c\n@@ -1245,6 +1245,9 @@ static void show_smap_vma_flags(struct seq_file *m, struct vm_area_struct *vma)\n #endif\n #ifdef CONFIG_64BIT\n \t\t[ilog2(VM_SEALED)] = \"sl\",\n+#endif\n+#ifdef CONFIG_TRANSPARENT_HUGEPAGE\n+\t\t[ilog2(VM_RESERVED_THP)] = \"rt\",\n #endif\n \t};\n \tsize_t i;\ndiff --git a/include/linux/gfp.h b/include/linux/gfp.h\nindex cdf95a9f0b87c1..2d05929fd8c72d 100644\n--- a/include/linux/gfp.h\n+++ b/include/linux/gfp.h\n@@ -30,6 +30,9 @@ static inline int gfp_migratetype(const gfp_t gfp_flags)\n \tBUILD_BUG_ON(((___GFP_MOVABLE | ___GFP_RECLAIMABLE) \u003e\u003e\n \t\t      GFP_MOVABLE_SHIFT) != MIGRATE_HIGHATOMIC);\n \n+\tif (unlikely(gfp_flags \u0026 __GFP_RESERVED_THP))\n+\t\treturn MIGRATE_RESERVED_THP;\n+\n \tif (unlikely(page_group_by_mobility_disabled))\n \t\treturn MIGRATE_UNMOVABLE;\n \ndiff --git a/include/linux/gfp_types.h b/include/linux/gfp_types.h\nindex 54ca0c88bab6e8..1f82a9491d357f 100644\n--- a/include/linux/gfp_types.h\n+++ b/include/linux/gfp_types.h\n@@ -33,7 +33,7 @@ enum {\n \t___GFP_IO_BIT,\n \t___GFP_FS_BIT,\n \t___GFP_ZERO_BIT,\n-\t___GFP_UNUSED_BIT,\t/* 0x200u unused */\n+\t___GFP_RESERVED_THP_BIT,\n \t___GFP_DIRECT_RECLAIM_BIT,\n \t___GFP_KSWAPD_RECLAIM_BIT,\n \t___GFP_WRITE_BIT,\n@@ -69,7 +69,7 @@ enum {\n #define ___GFP_IO\t\tBIT(___GFP_IO_BIT)\n #define ___GFP_FS\t\tBIT(___GFP_FS_BIT)\n #define ___GFP_ZERO\t\tBIT(___GFP_ZERO_BIT)\n-/* 0x200u unused */\n+#define ___GFP_RESERVED_THP\tBIT(___GFP_RESERVED_THP_BIT)\n #define ___GFP_DIRECT_RECLAIM\tBIT(___GFP_DIRECT_RECLAIM_BIT)\n #define ___GFP_KSWAPD_RECLAIM\tBIT(___GFP_KSWAPD_RECLAIM_BIT)\n #define ___GFP_WRITE\t\tBIT(___GFP_WRITE_BIT)\n@@ -141,6 +141,9 @@ enum {\n  * %__GFP_NO_OBJ_EXT causes slab allocation to have no object extension.\n  * mark_obj_codetag_empty() should be called upon freeing for objects allocated\n  * with this flag to indicate that their NULL tags are expected and normal.\n+ *\n+ * %__GFP_RESERVED_THP is an internal flag for reserved THP faults. It restricts\n+ *the allocation to %MIGRATE_RESERVED_THP pageblocks.\n  */\n #define __GFP_RECLAIMABLE ((__force gfp_t)___GFP_RECLAIMABLE)\n #define __GFP_WRITE\t((__force gfp_t)___GFP_WRITE)\n@@ -148,6 +151,7 @@ enum {\n #define __GFP_THISNODE\t((__force gfp_t)___GFP_THISNODE)\n #define __GFP_ACCOUNT\t((__force gfp_t)___GFP_ACCOUNT)\n #define __GFP_NO_OBJ_EXT   ((__force gfp_t)___GFP_NO_OBJ_EXT)\n+#define __GFP_RESERVED_THP ((__force gfp_t)___GFP_RESERVED_THP)\n \n /**\n  * DOC: Watermark modifiers\ndiff --git a/include/linux/huge_mm.h b/include/linux/huge_mm.h\nindex ad20f7f8c17944..4fe9651cd86b5e 100644\n--- a/include/linux/huge_mm.h\n+++ b/include/linux/huge_mm.h\n@@ -330,6 +330,8 @@ unsigned long thp_vma_allowable_orders(struct vm_area_struct *vma,\n \n \t\tif (vm_flags \u0026 VM_HUGEPAGE)\n \t\t\tmask |= READ_ONCE(huge_anon_orders_madvise);\n+\t\tif (vm_flags \u0026 VM_RESERVED_THP)\n+\t\t\tmask |= BIT(PMD_ORDER);\n \t\tif (hugepage_global_always() ||\n \t\t    ((vm_flags \u0026 VM_HUGEPAGE) \u0026\u0026 hugepage_global_enabled()))\n \t\t\tmask |= READ_ONCE(huge_anon_orders_inherit);\n@@ -371,7 +373,7 @@ static inline bool vma_thp_disabled(struct vm_area_struct *vma,\n \t * Are THPs disabled only for VMAs where we didn't get an explicit\n \t * advise to use them?\n \t */\n-\tif (vm_flags \u0026 VM_HUGEPAGE)\n+\tif (vm_flags \u0026 (VM_HUGEPAGE | VM_RESERVED_THP))\n \t\treturn false;\n \t/*\n \t * Forcing a collapse (e.g., madv_collapse), is a clear advice to\ndiff --git a/include/linux/mm.h b/include/linux/mm.h\nindex 485df9c2dbddb3..278cf4bfd4ec5b 100644\n--- a/include/linux/mm.h\n+++ b/include/linux/mm.h\n@@ -353,6 +353,7 @@ enum {\n #endif\n \tDECLARE_VMA_BIT(UFFD_MINOR, 41),\n \tDECLARE_VMA_BIT(SEALED, 42),\n+\tDECLARE_VMA_BIT(RESERVED_THP, 43),\n \t/* Flags that reuse flags above. */\n \tDECLARE_VMA_BIT_ALIAS(PKEY_BIT0, HIGH_ARCH_0),\n \tDECLARE_VMA_BIT_ALIAS(PKEY_BIT1, HIGH_ARCH_1),\n@@ -526,6 +527,12 @@ enum {\n #define VMA_DROPPABLE\t\tEMPTY_VMA_FLAGS\n #endif\n \n+#ifdef CONFIG_TRANSPARENT_HUGEPAGE\n+#define VM_RESERVED_THP\t\tINIT_VM_FLAG(RESERVED_THP)\n+#else\n+#define VM_RESERVED_THP\t\tVM_NONE\n+#endif\n+\n /* Bits set in the VMA until the stack is in its final location */\n #define VM_STACK_INCOMPLETE_SETUP (VM_RAND_READ | VM_SEQ_READ | VM_STACK_EARLY)\n \ndiff --git a/include/linux/mmzone.h b/include/linux/mmzone.h\nindex ca27121871475c..4418d0c9accdc8 100644\n--- a/include/linux/mmzone.h\n+++ b/include/linux/mmzone.h\n@@ -133,10 +133,9 @@ enum migratetype {\n \t * __free_pageblock_cma() function.\n \t */\n \tMIGRATE_CMA,\n-\t__MIGRATE_TYPE_END = MIGRATE_CMA,\n-#else\n-\t__MIGRATE_TYPE_END = MIGRATE_HIGHATOMIC,\n #endif\n+\tMIGRATE_RESERVED_THP,\n+\t__MIGRATE_TYPE_END = MIGRATE_RESERVED_THP,\n #ifdef CONFIG_MEMORY_ISOLATION\n \tMIGRATE_ISOLATE,\t/* can't allocate from here */\n #endif\n@@ -161,6 +160,9 @@ extern const char * const migratetype_names[MIGRATE_TYPES];\n #  define is_migrate_cma_folio(folio, pfn) false\n #endif\n \n+#define is_migrate_reserved_thp(migratetype) \\\n+\tunlikely((migratetype) == MIGRATE_RESERVED_THP)\n+\n static inline bool is_migrate_movable(int mt)\n {\n \treturn is_migrate_cma(mt) || mt == MIGRATE_MOVABLE;\n@@ -975,6 +977,9 @@ struct zone {\n \tunsigned long nr_reserved_highatomic;\n \tunsigned long nr_free_highatomic;\n \n+\tunsigned long nr_reserved_thp;\n+\tunsigned long nr_free_reserved_thp;\n+\n \t/*\n \t * We don't know if the memory that we're going to allocate will be\n \t * freeable or/and it will be released eventually, so to avoid totally\ndiff --git a/include/trace/events/mmflags.h b/include/trace/events/mmflags.h\nindex a6e5a44c9b4299..3db40ebd7060bd 100644\n--- a/include/trace/events/mmflags.h\n+++ b/include/trace/events/mmflags.h\n@@ -24,6 +24,7 @@\n \tTRACE_GFP_EM(IO)\t\t\t\\\n \tTRACE_GFP_EM(FS)\t\t\t\\\n \tTRACE_GFP_EM(ZERO)\t\t\t\\\n+\tTRACE_GFP_EM(RESERVED_THP)\t\t\\\n \tTRACE_GFP_EM(DIRECT_RECLAIM)\t\t\\\n \tTRACE_GFP_EM(KSWAPD_RECLAIM)\t\t\\\n \tTRACE_GFP_EM(WRITE)\t\t\t\\\n@@ -72,8 +73,7 @@\n \n TRACE_GFP_FLAGS\n \n-/* Just in case these are ever used */\n-TRACE_DEFINE_ENUM(___GFP_UNUSED_BIT);\n+/* Just in case this is ever used */\n TRACE_DEFINE_ENUM(___GFP_LAST_BIT);\n \n #define gfpflag_string(flag) {(__force unsigned long)flag, #flag}\ndiff --git a/include/uapi/asm-generic/mman-common.h b/include/uapi/asm-generic/mman-common.h\nindex ef1c27fa3c570f..b3d1448935ead1 100644\n--- a/include/uapi/asm-generic/mman-common.h\n+++ b/include/uapi/asm-generic/mman-common.h\n@@ -79,6 +79,8 @@\n \n #define MADV_COLLAPSE\t25\t\t/* Synchronous hugepage collapse */\n \n+#define MADV_RESERVED_THP 26\t\t/* Use reserved transparent hugepages */\n+\n #define MADV_GUARD_INSTALL 102\t\t/* fatal signal on access to range */\n #define MADV_GUARD_REMOVE 103\t\t/* unguard range */\n \ndiff --git a/mm/Makefile b/mm/Makefile\nindex eff9f9e7e061c1..fd74a7392e3467 100644\n--- a/mm/Makefile\n+++ b/mm/Makefile\n@@ -98,7 +98,7 @@ obj-$(CONFIG_MEMTEST)\t\t+= memtest.o\n obj-$(CONFIG_MIGRATION) += migrate.o\n obj-$(CONFIG_NUMA) += memory-tiers.o\n obj-$(CONFIG_DEVICE_MIGRATION) += migrate_device.o\n-obj-$(CONFIG_TRANSPARENT_HUGEPAGE) += huge_memory.o khugepaged.o\n+obj-$(CONFIG_TRANSPARENT_HUGEPAGE) += huge_memory.o khugepaged.o reserved_thp.o\n obj-$(CONFIG_PAGE_COUNTER) += page_counter.o\n obj-$(CONFIG_LIVEUPDATE_MEMFD) += memfd_luo.o\n obj-$(CONFIG_MEMCG_V1) += memcontrol-v1.o\ndiff --git a/mm/huge_memory.c b/mm/huge_memory.c\nindex 2bccb0a53a0a60..66d85a2fa855f3 100644\n--- a/mm/huge_memory.c\n+++ b/mm/huge_memory.c\n@@ -1267,6 +1267,9 @@ static struct folio *vma_alloc_anon_folio_pmd(struct vm_area_struct *vma,\n \tconst int order = HPAGE_PMD_ORDER;\n \tstruct folio *folio;\n \n+\tif (vma-\u003evm_flags \u0026 VM_RESERVED_THP)\n+\t\tgfp |= __GFP_RESERVED_THP;\n+\n \tfolio = vma_alloc_folio(gfp, order, vma, addr \u0026 HPAGE_PMD_MASK);\n \n \tif (unlikely(!folio)) {\n@@ -1344,8 +1347,11 @@ static vm_fault_t __do_huge_pmd_anonymous_page(struct vm_fault *vmf)\n \tvm_fault_t ret = 0;\n \n \tfolio = vma_alloc_anon_folio_pmd(vma, vmf-\u003eaddress);\n-\tif (unlikely(!folio))\n+\tif (unlikely(!folio)) {\n+\t\tif (vma-\u003evm_flags \u0026 VM_RESERVED_THP)\n+\t\t\treturn VM_FAULT_OOM;\n \t\treturn VM_FAULT_FALLBACK;\n+\t}\n \n \tpgtable = pte_alloc_one(vma-\u003evm_mm);\n \tif (unlikely(!pgtable)) {\n@@ -1480,15 +1486,17 @@ vm_fault_t do_huge_pmd_anonymous_page(struct vm_fault *vmf)\n \tvm_fault_t ret;\n \n \tif (!thp_vma_suitable_order(vma, haddr, PMD_ORDER))\n-\t\treturn VM_FAULT_FALLBACK;\n+\t\treturn (vma-\u003evm_flags \u0026 VM_RESERVED_THP) ? VM_FAULT_OOM :\n+\t\t\t\t\t\t\t   VM_FAULT_FALLBACK;\n \tret = vmf_anon_prepare(vmf);\n \tif (ret)\n \t\treturn ret;\n \tkhugepaged_enter_vma(vma, vma-\u003evm_flags);\n \n-\tif (!(vmf-\u003eflags \u0026 FAULT_FLAG_WRITE) \u0026\u0026\n-\t\t\t!mm_forbids_zeropage(vma-\u003evm_mm) \u0026\u0026\n-\t\t\ttransparent_hugepage_use_zero_page()) {\n+\tif (!(vma-\u003evm_flags \u0026 VM_RESERVED_THP) \u0026\u0026\n+\t    !(vmf-\u003eflags \u0026 FAULT_FLAG_WRITE) \u0026\u0026\n+\t    !mm_forbids_zeropage(vma-\u003evm_mm) \u0026\u0026\n+\t    transparent_hugepage_use_zero_page()) {\n \t\tpgtable_t pgtable;\n \t\tstruct folio *zero_folio;\n \t\tvm_fault_t ret;\ndiff --git a/mm/internal.h b/mm/internal.h\nindex 181e79f1d6a207..4b2a13d3537721 100644\n--- a/mm/internal.h\n+++ b/mm/internal.h\n@@ -1477,6 +1477,7 @@ unsigned int reclaim_clean_pages_from_list(struct zone *zone,\n #define ALLOC_HIGHATOMIC\t0x200 /* Allows access to MIGRATE_HIGHATOMIC */\n #define ALLOC_TRYLOCK\t\t0x400 /* Only use spin_trylock in allocation path */\n #define ALLOC_KSWAPD\t\t0x800 /* allow waking of kswapd, __GFP_KSWAPD_RECLAIM set */\n+#define ALLOC_RESERVED_THP\t0x1000 /* Allows access to reserved THP pageblocks */\n \n /* Flags that allow allocations below the min watermark. */\n #define ALLOC_RESERVES (ALLOC_NON_BLOCK|ALLOC_MIN_RESERVE|ALLOC_HIGHATOMIC|ALLOC_OOM)\n@@ -1951,4 +1952,9 @@ static inline int get_sysctl_max_map_count(void)\n bool may_expand_vm(struct mm_struct *mm, const vma_flags_t *vma_flags,\n \t\t   unsigned long npages);\n \n+unsigned long reserved_thp_pageblocks(unsigned long nr_hpages);\n+unsigned long reserved_thp_hpage_nr(unsigned long start, unsigned long end);\n+int reserved_thp_charge(unsigned long nr_hpages);\n+void reserved_thp_uncharge(unsigned long nr_hpages);\n+\n #endif\t/* __MM_INTERNAL_H */\ndiff --git a/mm/khugepaged.c b/mm/khugepaged.c\nindex 617bca76db49b6..80293e8c1e4e76 100644\n--- a/mm/khugepaged.c\n+++ b/mm/khugepaged.c\n@@ -451,6 +451,7 @@ int hugepage_madvise(struct vm_area_struct *vma,\n \tswitch (advice) {\n \tcase MADV_HUGEPAGE:\n \t\t*vm_flags \u0026= ~VM_NOHUGEPAGE;\n+\t\t*vm_flags \u0026= ~VM_RESERVED_THP;\n \t\t*vm_flags |= VM_HUGEPAGE;\n \t\t/*\n \t\t * If the vma become good for khugepaged to scan,\n@@ -461,6 +462,7 @@ int hugepage_madvise(struct vm_area_struct *vma,\n \t\tbreak;\n \tcase MADV_NOHUGEPAGE:\n \t\t*vm_flags \u0026= ~VM_HUGEPAGE;\n+\t\t*vm_flags \u0026= ~VM_RESERVED_THP;\n \t\t*vm_flags |= VM_NOHUGEPAGE;\n \t\t/*\n \t\t * Setting VM_NOHUGEPAGE will prevent khugepaged from scanning\n@@ -468,6 +470,12 @@ int hugepage_madvise(struct vm_area_struct *vma,\n \t\t * it got registered before VM_NOHUGEPAGE was set.\n \t\t */\n \t\tbreak;\n+\tcase MADV_RESERVED_THP:\n+\t\t*vm_flags \u0026= ~(VM_HUGEPAGE | VM_NOHUGEPAGE);\n+\t\t*vm_flags |= VM_RESERVED_THP;\n+\t\tbreak;\n+\tdefault:\n+\t\treturn -EINVAL;\n \t}\n \n \treturn 0;\ndiff --git a/mm/madvise.c b/mm/madvise.c\nindex cd9bb077072ccb..dd91105db68c75 100644\n--- a/mm/madvise.c\n+++ b/mm/madvise.c\n@@ -13,6 +13,7 @@\n #include \u003clinux/page-isolation.h\u003e\n #include \u003clinux/page_idle.h\u003e\n #include \u003clinux/userfaultfd_k.h\u003e\n+#include \u003clinux/huge_mm.h\u003e\n #include \u003clinux/hugetlb.h\u003e\n #include \u003clinux/falloc.h\u003e\n #include \u003clinux/fadvise.h\u003e\n@@ -1331,6 +1332,65 @@ static bool can_madvise_modify(struct madvise_behavior *madv_behavior)\n }\n #endif\n \n+static bool reserved_thp_madvise_aligned(struct vm_area_struct *vma,\n+\t\t\t\t\t struct madvise_behavior_range *range)\n+{\n+\tif (!(vma-\u003evm_flags \u0026 VM_RESERVED_THP))\n+\t\treturn true;\n+\n+\treturn IS_ALIGNED(range-\u003estart, HPAGE_PMD_SIZE) \u0026\u0026\n+\t       IS_ALIGNED(range-\u003eend, HPAGE_PMD_SIZE);\n+}\n+\n+static int madvise_hugepage_policy(struct madvise_behavior *madv_behavior,\n+\t\t\t\t   vm_flags_t *new_flags,\n+\t\t\t\t   unsigned long *reserved_hpages,\n+\t\t\t\t   bool *charge_reserved_thp,\n+\t\t\t\t   bool *uncharge_reserved_thp)\n+{\n+\tstruct vm_area_struct *vma = madv_behavior-\u003evma;\n+\tstruct madvise_behavior_range *range = \u0026madv_behavior-\u003erange;\n+\tunsigned long hpages;\n+\tint behavior = madv_behavior-\u003ebehavior;\n+\tint error;\n+\n+\tswitch (behavior) {\n+\tcase MADV_HUGEPAGE:\n+\tcase MADV_NOHUGEPAGE:\n+\t\terror = hugepage_madvise(vma, new_flags, behavior);\n+\t\tif (error)\n+\t\t\treturn error;\n+\t\t*uncharge_reserved_thp = (vma-\u003evm_flags \u0026 VM_RESERVED_THP) \u0026\u0026\n+\t\t\t\t\t !(*new_flags \u0026 VM_RESERVED_THP);\n+\t\treturn 0;\n+\tcase MADV_RESERVED_THP:\n+\t\tif (!IS_ENABLED(CONFIG_64BIT))\n+\t\t\treturn -EINVAL;\n+\t\tif (!vma_is_anonymous(vma) || (*new_flags \u0026 VM_SHARED) ||\n+\t\t    (*new_flags \u0026 VM_SPECIAL))\n+\t\t\treturn -EINVAL;\n+\t\tif (!IS_ALIGNED(range-\u003estart, HPAGE_PMD_SIZE) ||\n+\t\t    !IS_ALIGNED(range-\u003eend, HPAGE_PMD_SIZE))\n+\t\t\treturn -EINVAL;\n+\n+\t\terror = hugepage_madvise(vma, new_flags, behavior);\n+\t\tif (error)\n+\t\t\treturn error;\n+\n+\t\tif (!(vma-\u003evm_flags \u0026 VM_RESERVED_THP)) {\n+\t\t\thpages = reserved_thp_hpage_nr(range-\u003estart, range-\u003eend);\n+\t\t\terror = reserved_thp_charge(hpages);\n+\t\t\tif (error)\n+\t\t\t\treturn error;\n+\t\t\t*reserved_hpages = hpages;\n+\t\t\t*charge_reserved_thp = true;\n+\t\t}\n+\t\treturn 0;\n+\tdefault:\n+\t\treturn -EINVAL;\n+\t}\n+}\n+\n /*\n  * Apply an madvise behavior to a region of a vma.  madvise_update_vma\n  * will handle splitting a vm area into separate areas, each area with its own\n@@ -1342,6 +1402,9 @@ static int madvise_vma_behavior(struct madvise_behavior *madv_behavior)\n \tstruct vm_area_struct *vma = madv_behavior-\u003evma;\n \tvm_flags_t new_flags = vma-\u003evm_flags;\n \tstruct madvise_behavior_range *range = \u0026madv_behavior-\u003erange;\n+\tunsigned long reserved_hpages = 0;\n+\tbool charge_reserved_thp = false;\n+\tbool uncharge_reserved_thp = false;\n \tint error;\n \n \tif (unlikely(!can_madvise_modify(madv_behavior)))\n@@ -1353,14 +1416,22 @@ static int madvise_vma_behavior(struct madvise_behavior *madv_behavior)\n \tcase MADV_WILLNEED:\n \t\treturn madvise_willneed(madv_behavior);\n \tcase MADV_COLD:\n+\t\tif (!reserved_thp_madvise_aligned(vma, range))\n+\t\t\treturn -EINVAL;\n \t\treturn madvise_cold(madv_behavior);\n \tcase MADV_PAGEOUT:\n+\t\tif (!reserved_thp_madvise_aligned(vma, range))\n+\t\t\treturn -EINVAL;\n \t\treturn madvise_pageout(madv_behavior);\n \tcase MADV_FREE:\n \tcase MADV_DONTNEED:\n \tcase MADV_DONTNEED_LOCKED:\n+\t\tif (!reserved_thp_madvise_aligned(vma, range))\n+\t\t\treturn -EINVAL;\n \t\treturn madvise_dontneed_free(madv_behavior);\n \tcase MADV_COLLAPSE:\n+\t\tif (vma-\u003evm_flags \u0026 VM_RESERVED_THP)\n+\t\t\treturn -EINVAL;\n \t\treturn madvise_collapse(vma, range-\u003estart, range-\u003eend,\n \t\t\t\u0026madv_behavior-\u003elock_dropped);\n \tcase MADV_GUARD_INSTALL:\n@@ -1416,7 +1487,11 @@ static int madvise_vma_behavior(struct madvise_behavior *madv_behavior)\n \t\tbreak;\n \tcase MADV_HUGEPAGE:\n \tcase MADV_NOHUGEPAGE:\n-\t\terror = hugepage_madvise(vma, \u0026new_flags, behavior);\n+\tcase MADV_RESERVED_THP:\n+\t\terror = madvise_hugepage_policy(madv_behavior, \u0026new_flags,\n+\t\t\t\t\t\t\u0026reserved_hpages,\n+\t\t\t\t\t\t\u0026charge_reserved_thp,\n+\t\t\t\t\t\t\u0026uncharge_reserved_thp);\n \t\tif (error)\n \t\t\tgoto out;\n \t\tbreak;\n@@ -1431,6 +1506,11 @@ static int madvise_vma_behavior(struct madvise_behavior *madv_behavior)\n \tVM_WARN_ON_ONCE(madv_behavior-\u003elock_mode != MADVISE_MMAP_WRITE_LOCK);\n \n \terror = madvise_update_vma(new_flags, madv_behavior);\n+\tif (error \u0026\u0026 charge_reserved_thp)\n+\t\treserved_thp_uncharge(reserved_hpages);\n+\telse if (!error \u0026\u0026 uncharge_reserved_thp)\n+\t\treserved_thp_uncharge(reserved_thp_hpage_nr(range-\u003estart,\n+\t\t\t\t\t\t\t    range-\u003eend));\n out:\n \t/*\n \t * madvise() returns EAGAIN if kernel resources, such as\n@@ -1541,6 +1621,7 @@ madvise_behavior_valid(int behavior)\n \tcase MADV_HUGEPAGE:\n \tcase MADV_NOHUGEPAGE:\n \tcase MADV_COLLAPSE:\n+\tcase MADV_RESERVED_THP:\n #endif\n \tcase MADV_DONTDUMP:\n \tcase MADV_DODUMP:\ndiff --git a/mm/memory.c b/mm/memory.c\nindex ff338c2abe9231..225fc1ae22386d 100644\n--- a/mm/memory.c\n+++ b/mm/memory.c\n@@ -5297,6 +5297,9 @@ static vm_fault_t do_anonymous_page(struct vm_fault *vmf)\n \tif (vma-\u003evm_flags \u0026 VM_SHARED)\n \t\treturn VM_FAULT_SIGBUS;\n \n+\tif (unlikely(vma-\u003evm_flags \u0026 VM_RESERVED_THP))\n+\t\treturn VM_FAULT_OOM;\n+\n \t/*\n \t * Use pte_alloc() instead of pte_alloc_map(), so that OOM can\n \t * be distinguished from a transient failure of pte_offset_map().\ndiff --git a/mm/mmap.c b/mm/mmap.c\nindex 2311ae7c2ff45c..4818b14ec0ff6c 100644\n--- a/mm/mmap.c\n+++ b/mm/mmap.c\n@@ -1251,6 +1251,7 @@ unsigned long tear_down_vmas(struct mm_struct *mm, struct vma_iterator *vmi,\n \t\tstruct vm_area_struct *vma, unsigned long end)\n {\n \tunsigned long nr_accounted = 0;\n+\tunsigned long nr_reserved_thp = 0;\n \tint count = 0;\n \n \tmmap_assert_write_locked(mm);\n@@ -1258,6 +1259,10 @@ unsigned long tear_down_vmas(struct mm_struct *mm, struct vma_iterator *vmi,\n \tdo {\n \t\tif (vma-\u003evm_flags \u0026 VM_ACCOUNT)\n \t\t\tnr_accounted += vma_pages(vma);\n+\t\tif (vma-\u003evm_flags \u0026 VM_RESERVED_THP)\n+\t\t\tnr_reserved_thp +=\n+\t\t\t\treserved_thp_hpage_nr(vma-\u003evm_start,\n+\t\t\t\t\t\t      vma-\u003evm_end);\n \t\tvma_mark_detached(vma);\n \t\tremove_vma(vma);\n \t\tcount++;\n@@ -1266,6 +1271,7 @@ unsigned long tear_down_vmas(struct mm_struct *mm, struct vma_iterator *vmi,\n \t} while (vma \u0026\u0026 vma-\u003evm_end \u003c= end);\n \n \tVM_WARN_ON_ONCE(count != mm-\u003emap_count);\n+\treserved_thp_uncharge(nr_reserved_thp);\n \treturn nr_accounted;\n }\n \n@@ -1733,6 +1739,7 @@ __latent_entropy int dup_mmap(struct mm_struct *mm, struct mm_struct *oldmm)\n \tstruct vm_area_struct *mpnt, *tmp;\n \tint retval;\n \tunsigned long charge = 0;\n+\tunsigned long reserved_charge = 0;\n \tLIST_HEAD(uf);\n \tVMA_ITERATOR(vmi, mm, 0);\n \n@@ -1775,6 +1782,7 @@ __latent_entropy int dup_mmap(struct mm_struct *mm, struct mm_struct *oldmm)\n \t\t\tcontinue;\n \t\t}\n \t\tcharge = 0;\n+\t\treserved_charge = 0;\n \t\tif (mpnt-\u003evm_flags \u0026 VM_ACCOUNT) {\n \t\t\tunsigned long len = vma_pages(mpnt);\n \n@@ -1782,6 +1790,15 @@ __latent_entropy int dup_mmap(struct mm_struct *mm, struct mm_struct *oldmm)\n \t\t\t\tgoto fail_nomem;\n \t\t\tcharge = len;\n \t\t}\n+\t\tif (mpnt-\u003evm_flags \u0026 VM_RESERVED_THP) {\n+\t\t\tunsigned long len;\n+\n+\t\t\tlen = reserved_thp_hpage_nr(mpnt-\u003evm_start,\n+\t\t\t\t\t\t    mpnt-\u003evm_end);\n+\t\t\tif (reserved_thp_charge(len))\n+\t\t\t\tgoto fail_nomem;\n+\t\t\treserved_charge = len;\n+\t\t}\n \n \t\ttmp = vm_area_dup(mpnt);\n \t\tif (!tmp)\n@@ -1916,6 +1933,7 @@ __latent_entropy int dup_mmap(struct mm_struct *mm, struct mm_struct *oldmm)\n \tvm_area_free(tmp);\n fail_nomem:\n \tretval = -ENOMEM;\n+\treserved_thp_uncharge(reserved_charge);\n \tvm_unacct_memory(charge);\n \tgoto loop_out;\n }\ndiff --git a/mm/mremap.c b/mm/mremap.c\nindex e9c8b1d05832be..ae37e0b3ce788e 100644\n--- a/mm/mremap.c\n+++ b/mm/mremap.c\n@@ -24,6 +24,7 @@\n #include \u003clinux/mmu_notifier.h\u003e\n #include \u003clinux/uaccess.h\u003e\n #include \u003clinux/userfaultfd_k.h\u003e\n+#include \u003clinux/huge_mm.h\u003e\n #include \u003clinux/mempolicy.h\u003e\n #include \u003clinux/pgalloc.h\u003e\n \n@@ -69,6 +70,7 @@ struct vma_remap_struct {\n \tenum mremap_type remap_type;\t/* expand, shrink, etc. */\n \tbool mmap_locked;\t\t/* Is mm currently write-locked? */\n \tunsigned long charged;\t\t/* If VM_ACCOUNT, # pages to account. */\n+\tunsigned long reserved_thp_charged; /* If VM_RESERVED_THP, # hpages. */\n \tbool vmi_needs_invalidate;\t/* Is the VMA iterator invalidated? */\n };\n \n@@ -962,6 +964,9 @@ static unsigned long vrm_set_new_addr(struct vma_remap_struct *vrm)\n \t\t\t\tmap_flags);\n \tif (IS_ERR_VALUE(res))\n \t\treturn res;\n+\tif ((vma-\u003evm_flags \u0026 VM_RESERVED_THP) \u0026\u0026\n+\t    !IS_ALIGNED(res, HPAGE_PMD_SIZE))\n+\t\treturn -ENOMEM;\n \n \tvrm-\u003enew_addr = res;\n \treturn 0;\n@@ -977,24 +982,44 @@ static bool vrm_calc_charge(struct vma_remap_struct *vrm)\n {\n \tunsigned long charged;\n \n-\tif (!(vrm-\u003evma-\u003evm_flags \u0026 VM_ACCOUNT))\n-\t\treturn true;\n+\tvrm-\u003echarged = 0;\n+\tvrm-\u003ereserved_thp_charged = 0;\n \n-\t/*\n-\t * If we don't unmap the old mapping, then we account the entirety of\n-\t * the length of the new one. Otherwise it's just the delta in size.\n-\t */\n-\tif (vrm-\u003eflags \u0026 MREMAP_DONTUNMAP)\n-\t\tcharged = vrm-\u003enew_len \u003e\u003e PAGE_SHIFT;\n-\telse\n-\t\tcharged = vrm-\u003edelta \u003e\u003e PAGE_SHIFT;\n+\tif (vrm-\u003evma-\u003evm_flags \u0026 VM_ACCOUNT) {\n+\t\t/*\n+\t\t * If we don't unmap the old mapping, then we account the\n+\t\t * entirety of the length of the new one. Otherwise it's just\n+\t\t * the delta in size.\n+\t\t */\n+\t\tif (vrm-\u003eflags \u0026 MREMAP_DONTUNMAP)\n+\t\t\tcharged = vrm-\u003enew_len \u003e\u003e PAGE_SHIFT;\n+\t\telse\n+\t\t\tcharged = vrm-\u003edelta \u003e\u003e PAGE_SHIFT;\n \n \n-\t/* This accounts 'charged' pages of memory. */\n-\tif (security_vm_enough_memory_mm(current-\u003emm, charged))\n-\t\treturn false;\n+\t\t/* This accounts 'charged' pages of memory. */\n+\t\tif (security_vm_enough_memory_mm(current-\u003emm, charged))\n+\t\t\treturn false;\n \n-\tvrm-\u003echarged = charged;\n+\t\tvrm-\u003echarged = charged;\n+\t}\n+\n+\tif (vrm-\u003evma-\u003evm_flags \u0026 VM_RESERVED_THP) {\n+\t\tunsigned long hpages;\n+\n+\t\tif (vrm-\u003eflags \u0026 MREMAP_DONTUNMAP)\n+\t\t\thpages = reserved_thp_hpage_nr(0, vrm-\u003enew_len);\n+\t\telse\n+\t\t\thpages = reserved_thp_hpage_nr(0, vrm-\u003edelta);\n+\n+\t\tif (reserved_thp_charge(hpages)) {\n+\t\t\tvm_unacct_memory(vrm-\u003echarged);\n+\t\t\tvrm-\u003echarged = 0;\n+\t\t\treturn false;\n+\t\t}\n+\n+\t\tvrm-\u003ereserved_thp_charged = hpages;\n+\t}\n \treturn true;\n }\n \n@@ -1004,11 +1029,10 @@ static bool vrm_calc_charge(struct vma_remap_struct *vrm)\n  */\n static void vrm_uncharge(struct vma_remap_struct *vrm)\n {\n-\tif (!(vrm-\u003evma-\u003evm_flags \u0026 VM_ACCOUNT))\n-\t\treturn;\n-\n \tvm_unacct_memory(vrm-\u003echarged);\n \tvrm-\u003echarged = 0;\n+\treserved_thp_uncharge(vrm-\u003ereserved_thp_charged);\n+\tvrm-\u003ereserved_thp_charged = 0;\n }\n \n /*\n@@ -1157,8 +1181,8 @@ static void unmap_source_vma(struct vma_remap_struct *vrm)\n \tstruct vm_area_struct *vma = vrm-\u003evma;\n \tVMA_ITERATOR(vmi, mm, addr);\n \tint err;\n-\tunsigned long vm_start;\n-\tunsigned long vm_end;\n+\tunsigned long vm_start = 0;\n+\tunsigned long vm_end = 0;\n \t/*\n \t * It might seem odd that we check for MREMAP_DONTUNMAP here, given this\n \t * function implies that we unmap the original VMA, which seems\n@@ -1170,6 +1194,8 @@ static void unmap_source_vma(struct vma_remap_struct *vrm)\n \t */\n \tbool accountable_move = (vma-\u003evm_flags \u0026 VM_ACCOUNT) \u0026\u0026\n \t\t!(vrm-\u003eflags \u0026 MREMAP_DONTUNMAP);\n+\tbool reserved_thp_move = (vma-\u003evm_flags \u0026 VM_RESERVED_THP) \u0026\u0026\n+\t\t!(vrm-\u003eflags \u0026 MREMAP_DONTUNMAP);\n \n \t/*\n \t * So we perform a trick here to prevent incorrect accounting. Any merge\n@@ -1192,6 +1218,13 @@ static void unmap_source_vma(struct vma_remap_struct *vrm)\n \t\tvm_start = vma-\u003evm_start;\n \t\tvm_end = vma-\u003evm_end;\n \t}\n+\tif (reserved_thp_move) {\n+\t\tvm_flags_clear(vma, VM_RESERVED_THP);\n+\t\tif (!accountable_move) {\n+\t\t\tvm_start = vma-\u003evm_start;\n+\t\t\tvm_end = vma-\u003evm_end;\n+\t\t}\n+\t}\n \n \terr = do_vmi_munmap(\u0026vmi, mm, addr, len, vrm-\u003euf_unmap, /* unlock= */false);\n \tvrm-\u003evma = NULL; /* Invalidated. */\n@@ -1227,19 +1260,27 @@ static void unmap_source_vma(struct vma_remap_struct *vrm)\n \t *\n \t * do_vmi_munmap() will have restored the VMI back to addr.\n \t */\n-\tif (accountable_move) {\n+\tif (accountable_move || reserved_thp_move) {\n \t\tunsigned long end = addr + len;\n-\n-\t\tif (vm_start \u003c addr) {\n-\t\t\tstruct vm_area_struct *prev = vma_prev(\u0026vmi);\n-\n-\t\t\tvm_flags_set(prev, VM_ACCOUNT); /* Acquires VMA lock. */\n+\t\tstruct vm_area_struct *prev = NULL;\n+\t\tstruct vm_area_struct *next = NULL;\n+\n+\t\tif (vm_start \u003c addr)\n+\t\t\tprev = vma_prev(\u0026vmi);\n+\t\tif (vm_end \u003e end)\n+\t\t\tnext = vma_next(\u0026vmi);\n+\n+\t\tif (accountable_move) {\n+\t\t\tif (prev)\n+\t\t\t\tvm_flags_set(prev, VM_ACCOUNT); /* Acquires VMA lock. */\n+\t\t\tif (next)\n+\t\t\t\tvm_flags_set(next, VM_ACCOUNT); /* Acquires VMA lock. */\n \t\t}\n-\n-\t\tif (vm_end \u003e end) {\n-\t\t\tstruct vm_area_struct *next = vma_next(\u0026vmi);\n-\n-\t\t\tvm_flags_set(next, VM_ACCOUNT); /* Acquires VMA lock. */\n+\t\tif (reserved_thp_move) {\n+\t\t\tif (prev)\n+\t\t\t\tvm_flags_set(prev, VM_RESERVED_THP);\n+\t\t\tif (next)\n+\t\t\t\tvm_flags_set(next, VM_RESERVED_THP);\n \t\t}\n \t}\n }\n@@ -1309,7 +1350,6 @@ static int copy_vma_and_data(struct vma_remap_struct *vrm,\n \t*new_vma_ptr = new_vma;\n \treturn err;\n }\n-\n /*\n  * Perform final tasks for MADV_DONTUNMAP operation, clearing mlock() flag on\n  * remaining VMA by convention (it cannot be mlock()'d any longer, as pages in\n@@ -1576,6 +1616,23 @@ static bool align_hugetlb(struct vma_remap_struct *vrm)\n \treturn true;\n }\n \n+static bool check_reserved_thp_alignment(struct vma_remap_struct *vrm)\n+{\n+\tif (!(vrm-\u003evma-\u003evm_flags \u0026 VM_RESERVED_THP))\n+\t\treturn true;\n+\n+\tif (!IS_ALIGNED(vrm-\u003eaddr, HPAGE_PMD_SIZE) ||\n+\t    !IS_ALIGNED(vrm-\u003eold_len, HPAGE_PMD_SIZE) ||\n+\t    !IS_ALIGNED(vrm-\u003enew_len, HPAGE_PMD_SIZE))\n+\t\treturn false;\n+\n+\tif ((vrm-\u003eremap_type == MREMAP_EXPAND || vrm_implies_new_addr(vrm)) \u0026\u0026\n+\t    !IS_ALIGNED(vrm-\u003enew_addr, HPAGE_PMD_SIZE))\n+\t\treturn false;\n+\n+\treturn true;\n+}\n+\n /*\n  * We are mremap()'ing without specifying a fixed address to move to, but are\n  * requesting that the VMA's size be increased.\n@@ -1745,6 +1802,8 @@ static int check_prep_vma(struct vma_remap_struct *vrm)\n \t/* For convenience, we set new_addr even if VMA won't move. */\n \tif (!vrm_implies_new_addr(vrm))\n \t\tvrm-\u003enew_addr = addr;\n+\tif (!check_reserved_thp_alignment(vrm))\n+\t\treturn -EINVAL;\n \n \t/* Below only meaningful if we expand or move a VMA. */\n \tif (!vrm_will_map_new(vrm))\ndiff --git a/mm/page_alloc.c b/mm/page_alloc.c\nindex ee902a468c2f5b..660e501bf676b6 100644\n--- a/mm/page_alloc.c\n+++ b/mm/page_alloc.c\n@@ -263,6 +263,7 @@ const char * const migratetype_names[MIGRATE_TYPES] = {\n #ifdef CONFIG_CMA\n \t\"CMA\",\n #endif\n+\t\"ReserveTHP\",\n #ifdef CONFIG_MEMORY_ISOLATION\n \t\"Isolate\",\n #endif\n@@ -784,6 +785,9 @@ static inline void account_freepages(struct zone *zone, int nr_pages,\n \telse if (migratetype == MIGRATE_HIGHATOMIC)\n \t\tWRITE_ONCE(zone-\u003enr_free_highatomic,\n \t\t\t   zone-\u003enr_free_highatomic + nr_pages);\n+\telse if (migratetype == MIGRATE_RESERVED_THP)\n+\t\tWRITE_ONCE(zone-\u003enr_free_reserved_thp,\n+\t\t\t   zone-\u003enr_free_reserved_thp + nr_pages);\n }\n \n /* Used for pages not on another list */\n@@ -2456,6 +2460,9 @@ __rmqueue(struct zone *zone, unsigned int order, int migratetype,\n {\n \tstruct page *page;\n \n+\tif (alloc_flags \u0026 ALLOC_RESERVED_THP)\n+\t\treturn __rmqueue_smallest(zone, order, MIGRATE_RESERVED_THP);\n+\n \tif (IS_ENABLED(CONFIG_CMA)) {\n \t\t/*\n \t\t * Balance movable allocations between regular and CMA areas by\n@@ -2960,7 +2967,8 @@ static void __free_frozen_pages(struct page *page, unsigned int order,\n \tzone = page_zone(page);\n \tmigratetype = get_pfnblock_migratetype(page, pfn);\n \tif (unlikely(migratetype \u003e= MIGRATE_PCPTYPES)) {\n-\t\tif (unlikely(is_migrate_isolate(migratetype))) {\n+\t\tif (unlikely(is_migrate_reserved_thp(migratetype) ||\n+\t\t\t     is_migrate_isolate(migratetype))) {\n \t\t\tfree_one_page(zone, page, pfn, order, fpi_flags);\n \t\t\treturn;\n \t\t}\n@@ -3038,6 +3046,7 @@ void free_unref_folios(struct folio_batch *folios)\n \n \t\t/* Different zone requires a different pcp lock */\n \t\tif (zone != locked_zone ||\n+\t\t    is_migrate_reserved_thp(migratetype) ||\n \t\t    is_migrate_isolate(migratetype)) {\n \t\t\tif (pcp) {\n \t\t\t\tpcp_spin_unlock(pcp);\n@@ -3045,6 +3054,12 @@ void free_unref_folios(struct folio_batch *folios)\n \t\t\t\tpcp = NULL;\n \t\t\t}\n \n+\t\t\tif (is_migrate_reserved_thp(migratetype)) {\n+\t\t\t\tfree_one_page(zone, \u0026folio-\u003epage, pfn,\n+\t\t\t\t\t      order, FPI_NONE);\n+\t\t\t\tcontinue;\n+\t\t\t}\n+\n \t\t\t/*\n \t\t\t * Free isolated pages directly to the\n \t\t\t * allocator, see comment in free_frozen_pages.\n@@ -3235,7 +3250,8 @@ struct page *rmqueue_buddy(struct zone *preferred_zone, struct zone *zone,\n \t\t\t * reserves as failing now is worse than failing a\n \t\t\t * high-order atomic allocation in the future.\n \t\t\t */\n-\t\t\tif (!page \u0026\u0026 (alloc_flags \u0026 (ALLOC_OOM|ALLOC_NON_BLOCK)))\n+\t\t\tif (!page \u0026\u0026 !(alloc_flags \u0026 ALLOC_RESERVED_THP) \u0026\u0026\n+\t\t\t    (alloc_flags \u0026 (ALLOC_OOM|ALLOC_NON_BLOCK)))\n \t\t\t\tpage = __rmqueue_smallest(zone, order, MIGRATE_HIGHATOMIC);\n \n \t\t\tif (!page) {\n@@ -3405,7 +3421,8 @@ struct page *rmqueue(struct zone *preferred_zone,\n {\n \tstruct page *page;\n \n-\tif (likely(pcp_allowed_order(order))) {\n+\tif (likely(pcp_allowed_order(order)) \u0026\u0026\n+\t    !(alloc_flags \u0026 ALLOC_RESERVED_THP)) {\n \t\tpage = rmqueue_pcplist(preferred_zone, zone, order,\n \t\t\t\t       migratetype, alloc_flags);\n \t\tif (likely(page))\n@@ -3556,6 +3573,35 @@ static bool unreserve_highatomic_pageblock(const struct alloc_context *ac,\n \treturn false;\n }\n \n+unsigned long reserved_thp_pageblocks(unsigned long nr_hpages)\n+{\n+\tunsigned int order = max_t(unsigned int, HPAGE_PMD_ORDER,\n+\t\t\t\t   pageblock_order);\n+\tunsigned long hpages_per_block = 1UL \u003c\u003c (order - HPAGE_PMD_ORDER);\n+\tunsigned long reserved = 0;\n+\tgfp_t gfp = (GFP_HIGHUSER | __GFP_COMP | __GFP_NOMEMALLOC |\n+\t\t     __GFP_NOWARN | __GFP_NORETRY);\n+\n+\twhile (reserved \u003c nr_hpages) {\n+\t\tstruct page *page;\n+\t\tstruct zone *zone;\n+\t\tunsigned long flags;\n+\n+\t\tpage = alloc_pages(gfp, order);\n+\t\tif (!page)\n+\t\t\tbreak;\n+\n+\t\tzone = page_zone(page);\n+\t\tspin_lock_irqsave(\u0026zone-\u003elock, flags);\n+\t\tchange_pageblock_range(page, order, MIGRATE_RESERVED_THP);\n+\t\tzone-\u003enr_reserved_thp += 1UL \u003c\u003c order;\n+\t\tspin_unlock_irqrestore(\u0026zone-\u003elock, flags);\n+\t\t__free_pages(page, order);\n+\t\treserved += hpages_per_block;\n+\t}\n+\treturn reserved;\n+}\n+\n static inline long __zone_watermark_unusable_free(struct zone *z,\n \t\t\t\tunsigned int order, unsigned int alloc_flags)\n {\n@@ -3568,6 +3614,9 @@ static inline long __zone_watermark_unusable_free(struct zone *z,\n \tif (likely(!(alloc_flags \u0026 ALLOC_RESERVES)))\n \t\tunusable_free += READ_ONCE(z-\u003enr_free_highatomic);\n \n+\tif (!(alloc_flags \u0026 ALLOC_RESERVED_THP))\n+\t\tunusable_free += READ_ONCE(z-\u003enr_free_reserved_thp);\n+\n #ifdef CONFIG_CMA\n \t/* If allocation can't use CMA areas don't use free CMA pages */\n \tif (!(alloc_flags \u0026 ALLOC_CMA))\n@@ -3642,6 +3691,12 @@ bool __zone_watermark_ok(struct zone *z, unsigned int order, unsigned long mark,\n \t\tif (!area-\u003enr_free)\n \t\t\tcontinue;\n \n+\t\tif (alloc_flags \u0026 ALLOC_RESERVED_THP) {\n+\t\t\tif (!free_area_empty(area, MIGRATE_RESERVED_THP))\n+\t\t\t\treturn true;\n+\t\t\tcontinue;\n+\t\t}\n+\n \t\tfor (mt = 0; mt \u003c MIGRATE_PCPTYPES; mt++) {\n \t\t\tif (!free_area_empty(area, mt))\n \t\t\t\treturn true;\n@@ -3876,6 +3931,9 @@ get_page_from_freelist(gfp_t gfp_mask, unsigned int order, int alloc_flags,\n \n \t\tcond_accept_memory(zone, order, alloc_flags);\n \n+\t\tif (alloc_flags \u0026 ALLOC_RESERVED_THP)\n+\t\t\tgoto try_this_zone;\n+\n \t\t/*\n \t\t * Detect whether the number of free pages is below high\n \t\t * watermark.  If so, we will decrease pcp-\u003ehigh and free\n@@ -5033,6 +5091,15 @@ static inline bool prepare_alloc_pages(gfp_t gfp_mask, unsigned int order,\n \tac-\u003enodemask = nodemask;\n \tac-\u003emigratetype = gfp_migratetype(gfp_mask);\n \n+\tif (gfp_mask \u0026 __GFP_RESERVED_THP) {\n+\t\tif (!IS_ENABLED(CONFIG_TRANSPARENT_HUGEPAGE) ||\n+\t\t    WARN_ON_ONCE_GFP(order != HPAGE_PMD_ORDER, gfp_mask))\n+\t\t\treturn false;\n+\n+\t\tac-\u003emigratetype = MIGRATE_RESERVED_THP;\n+\t\t*alloc_flags |= ALLOC_RESERVED_THP;\n+\t}\n+\n \tif (cpusets_enabled()) {\n \t\t*alloc_gfp |= __GFP_HARDWALL;\n \t\t/*\ndiff --git a/mm/reserved_thp.c b/mm/reserved_thp.c\nnew file mode 100644\nindex 00000000000000..931c539c15a709\n--- /dev/null\n+++ b/mm/reserved_thp.c\n@@ -0,0 +1,133 @@\n+// SPDX-License-Identifier: GPL-2.0\n+\n+#include \u003clinux/mm.h\u003e\n+#include \"internal.h\"\n+\n+static DEFINE_SPINLOCK(reserved_thp_lock);\n+\n+static unsigned long reserved_thp_cmdline_size __initdata = HPAGE_PMD_SIZE;\n+static bool reserved_thp_cmdline_size_valid __initdata = true;\n+static unsigned long reserved_thp_requested __initdata;\n+static unsigned long reserved_thp_total;\n+static unsigned long reserved_thp_used;\n+\n+static int __init setup_reserved_thp_size(char *str)\n+{\n+\tunsigned long size;\n+\tsize = memparse(str, NULL);\n+\tif (size != HPAGE_PMD_SIZE) {\n+\t\tpr_warn(\"unsupported thp_reserved_size=%s, only %lu is supported\\n\",\n+\t\t\tstr, HPAGE_PMD_SIZE);\n+\t\treserved_thp_cmdline_size_valid = false;\n+\t\treturn -EINVAL;\n+\t}\n+\treserved_thp_cmdline_size = size;\n+\treserved_thp_cmdline_size_valid = true;\n+\treturn 0;\n+}\n+early_param(\"thp_reserved_size\", setup_reserved_thp_size);\n+static int __init setup_reserved_thp_nr(char *str)\n+{\n+\tint count;\n+\tif (sscanf(str, \"%lu%n\", \u0026reserved_thp_requested, \u0026count) != 1 ||\n+\t    str[count]) {\n+\t\tpr_warn(\"invalid thp_reserved_nr=%s\\n\", str);\n+\t\treserved_thp_requested = 0;\n+\t\treturn -EINVAL;\n+\t}\n+\treturn 0;\n+}\n+early_param(\"thp_reserved_nr\", setup_reserved_thp_nr);\n+\n+unsigned long reserved_thp_hpage_nr(unsigned long start, unsigned long end)\n+{\n+\treturn (end - start) \u003e\u003e HPAGE_PMD_SHIFT;\n+}\n+\n+int reserved_thp_charge(unsigned long nr_hpages)\n+{\n+\tint ret = 0;\n+\n+\tif (!nr_hpages)\n+\t\treturn 0;\n+\n+\tspin_lock(\u0026reserved_thp_lock);\n+\tif (nr_hpages \u003e reserved_thp_total - reserved_thp_used)\n+\t\tret = -ENOMEM;\n+\telse\n+\t\treserved_thp_used += nr_hpages;\n+\tspin_unlock(\u0026reserved_thp_lock);\n+\n+\treturn ret;\n+}\n+\n+void reserved_thp_uncharge(unsigned long nr_hpages)\n+{\n+\tif (!nr_hpages)\n+\t\treturn;\n+\n+\tspin_lock(\u0026reserved_thp_lock);\n+\tif (WARN_ON_ONCE(nr_hpages \u003e reserved_thp_used))\n+\t\treserved_thp_used = 0;\n+\telse\n+\t\treserved_thp_used -= nr_hpages;\n+\tspin_unlock(\u0026reserved_thp_lock);\n+}\n+\n+static ssize_t total_hpages_show(struct kobject *kobj,\n+\t\t\t\t struct kobj_attribute *attr, char *buf)\n+{\n+\treturn sysfs_emit(buf, \"%lu\\n\", READ_ONCE(reserved_thp_total));\n+}\n+static ssize_t free_hpages_show(struct kobject *kobj,\n+\t\t\t\tstruct kobj_attribute *attr, char *buf)\n+{\n+\tunsigned long free_hpages;\n+\n+\tspin_lock(\u0026reserved_thp_lock);\n+\tfree_hpages = reserved_thp_total - reserved_thp_used;\n+\tspin_unlock(\u0026reserved_thp_lock);\n+\n+\treturn sysfs_emit(buf, \"%lu\\n\", free_hpages);\n+}\n+static ssize_t used_hpages_show(struct kobject *kobj,\n+\t\t\t\tstruct kobj_attribute *attr, char *buf)\n+{\n+\treturn sysfs_emit(buf, \"%lu\\n\", READ_ONCE(reserved_thp_used));\n+}\n+\n+static struct kobj_attribute total_hpages_attr = __ATTR_RO(total_hpages);\n+static struct kobj_attribute free_hpages_attr = __ATTR_RO(free_hpages);\n+static struct kobj_attribute used_hpages_attr = __ATTR_RO(used_hpages);\n+\n+static struct attribute *reserved_thp_attrs[] = {\n+\t\u0026total_hpages_attr.attr,\n+\t\u0026free_hpages_attr.attr,\n+\t\u0026used_hpages_attr.attr,\n+\tNULL,\n+};\n+\n+static const struct attribute_group reserved_thp_attr_group = {\n+\t.attrs = reserved_thp_attrs,\n+};\n+\n+static int __init reserved_thp_init(void)\n+{\n+\tstruct kobject *kobj;\n+\tint ret;\n+\n+\tif (reserved_thp_requested \u0026\u0026 reserved_thp_cmdline_size_valid) {\n+\t\treserved_thp_total = reserved_thp_pageblocks(reserved_thp_requested);\n+\t\tpr_info(\"reserved %lu/%lu PMD THP pageblocks (%lu bytes each)\\n\",\n+\t\t\treserved_thp_total, reserved_thp_requested,\n+\t\t\treserved_thp_cmdline_size);\n+\t}\n+\tkobj = kobject_create_and_add(\"reserved_thp\", mm_kobj);\n+\tif (!kobj)\n+\t\treturn -ENOMEM;\n+\tret = sysfs_create_group(kobj, \u0026reserved_thp_attr_group);\n+\tif (ret)\n+\t\tkobject_put(kobj);\n+\treturn ret;\n+}\n+subsys_initcall(reserved_thp_init);\n\\ No newline at end of file\ndiff --git a/mm/show_mem.c b/mm/show_mem.c\nindex 43aca5a2ac990a..e9381afca4acae 100644\n--- a/mm/show_mem.c\n+++ b/mm/show_mem.c\n@@ -142,6 +142,7 @@ static void show_migration_types(unsigned char type)\n #ifdef CONFIG_CMA\n \t\t[MIGRATE_CMA]\t\t= 'C',\n #endif\n+\t\t[MIGRATE_RESERVED_THP]\t= 'T',\n #ifdef CONFIG_MEMORY_ISOLATION\n \t\t[MIGRATE_ISOLATE]\t= 'I',\n #endif\n@@ -308,6 +309,8 @@ static void show_free_areas(unsigned int filter, nodemask_t *nodemask, int max_z\n \t\t\t\" high:%lukB\"\n \t\t\t\" reserved_highatomic:%luKB\"\n \t\t\t\" free_highatomic:%luKB\"\n+\t\t\t\" reserved_thp:%luKB\"\n+\t\t\t\" free_reserved_thp:%luKB\"\n \t\t\t\" active_anon:%lukB\"\n \t\t\t\" inactive_anon:%lukB\"\n \t\t\t\" active_file:%lukB\"\n@@ -331,6 +334,8 @@ static void show_free_areas(unsigned int filter, nodemask_t *nodemask, int max_z\n \t\t\tK(high_wmark_pages(zone)),\n \t\t\tK(zone-\u003enr_reserved_highatomic),\n \t\t\tK(zone-\u003enr_free_highatomic),\n+\t\t\tK(zone-\u003enr_reserved_thp),\n+\t\t\tK(zone-\u003enr_free_reserved_thp),\n \t\t\tK(zone_page_state(zone, NR_ZONE_ACTIVE_ANON)),\n \t\t\tK(zone_page_state(zone, NR_ZONE_INACTIVE_ANON)),\n \t\t\tK(zone_page_state(zone, NR_ZONE_ACTIVE_FILE)),\ndiff --git a/mm/vma.c b/mm/vma.c\nindex 9eea2850818a85..8c4cd7c97a984c 100644\n--- a/mm/vma.c\n+++ b/mm/vma.c\n@@ -7,6 +7,8 @@\n #include \"vma_internal.h\"\n #include \"vma.h\"\n \n+#include \u003clinux/huge_mm.h\u003e\n+\n struct mmap_state {\n \tstruct mm_struct *mm;\n \tstruct vma_iterator *vmi;\n@@ -507,6 +509,10 @@ __split_vma(struct vma_iterator *vmi, struct vm_area_struct *vma,\n \tWARN_ON(vma-\u003evm_start \u003e= addr);\n \tWARN_ON(vma-\u003evm_end \u003c= addr);\n \n+\tif ((vma-\u003evm_flags \u0026 VM_RESERVED_THP) \u0026\u0026\n+\t    !IS_ALIGNED(addr, HPAGE_PMD_SIZE))\n+\t\treturn -EINVAL;\n+\n \tif (vma-\u003evm_ops \u0026\u0026 vma-\u003evm_ops-\u003emay_split) {\n \t\terr = vma-\u003evm_ops-\u003emay_split(vma, addr);\n \t\tif (err)\n@@ -1361,6 +1367,7 @@ static void vms_complete_munmap_vmas(struct vma_munmap_struct *vms,\n \t\tremove_vma(vma);\n \n \tvm_unacct_memory(vms-\u003enr_accounted);\n+\treserved_thp_uncharge(vms-\u003enr_reserved_thp);\n \tvalidate_mm(mm);\n \tif (vms-\u003eunlock)\n \t\tmmap_read_unlock(mm);\n@@ -1423,6 +1430,11 @@ static int vms_gather_munmap_vmas(struct vma_munmap_struct *vms,\n \t\t\terror = -EPERM;\n \t\t\tgoto start_split_failed;\n \t\t}\n+\t\tif ((vms-\u003evma-\u003evm_flags \u0026 VM_RESERVED_THP) \u0026\u0026\n+\t\t    !IS_ALIGNED(vms-\u003estart, HPAGE_PMD_SIZE)) {\n+\t\t\terror = -EINVAL;\n+\t\t\tgoto start_split_failed;\n+\t\t}\n \n \t\terror = __split_vma(vms-\u003evmi, vms-\u003evma, vms-\u003estart, 1);\n \t\tif (error)\n@@ -1445,6 +1457,11 @@ static int vms_gather_munmap_vmas(struct vma_munmap_struct *vms,\n \t\t}\n \t\t/* Does it split the end? */\n \t\tif (next-\u003evm_end \u003e vms-\u003eend) {\n+\t\t\tif ((next-\u003evm_flags \u0026 VM_RESERVED_THP) \u0026\u0026\n+\t\t\t    !IS_ALIGNED(vms-\u003eend, HPAGE_PMD_SIZE)) {\n+\t\t\t\terror = -EINVAL;\n+\t\t\t\tgoto end_split_failed;\n+\t\t\t}\n \t\t\terror = __split_vma(vms-\u003evmi, next, vms-\u003eend, 0);\n \t\t\tif (error)\n \t\t\t\tgoto end_split_failed;\n@@ -1465,6 +1482,11 @@ static int vms_gather_munmap_vmas(struct vma_munmap_struct *vms,\n \t\tif (vma_test(next, VMA_ACCOUNT_BIT))\n \t\t\tvms-\u003enr_accounted += nrpages;\n \n+\t\tif (next-\u003evm_flags \u0026 VM_RESERVED_THP)\n+\t\t\tvms-\u003enr_reserved_thp +=\n+\t\t\t\treserved_thp_hpage_nr(next-\u003evm_start,\n+\t\t\t\t\t\t      next-\u003evm_end);\n+\n \t\tif (is_exec_mapping(next-\u003evm_flags))\n \t\t\tvms-\u003eexec_vm += nrpages;\n \t\telse if (is_stack_mapping(next-\u003evm_flags))\n@@ -1560,6 +1582,7 @@ static void init_vma_munmap(struct vma_munmap_struct *vms,\n \tvms-\u003euf = uf;\n \tvms-\u003evma_count = 0;\n \tvms-\u003enr_pages = vms-\u003elocked_vm = vms-\u003enr_accounted = 0;\n+\tvms-\u003enr_reserved_thp = 0;\n \tvms-\u003eexec_vm = vms-\u003estack_vm = vms-\u003edata_vm = 0;\n \tvms-\u003eunmap_start = FIRST_USER_ADDRESS;\n \tvms-\u003eunmap_end = USER_PGTABLES_CEILING;\ndiff --git a/mm/vma.h b/mm/vma.h\nindex 8e4b61a7304c68..68e44adee5c89e 100644\n--- a/mm/vma.h\n+++ b/mm/vma.h\n@@ -48,6 +48,7 @@ struct vma_munmap_struct {\n \tunsigned long nr_pages;         /* Number of pages being removed */\n \tunsigned long locked_vm;        /* Number of locked pages */\n \tunsigned long nr_accounted;     /* Number of VM_ACCOUNT pages */\n+\tunsigned long nr_reserved_thp;  /* Number of reserved PMD THP slots */\n \tunsigned long exec_vm;\n \tunsigned long stack_vm;\n \tunsigned long data_vm;\ndiff --git a/tools/include/linux/gfp_types.h b/tools/include/linux/gfp_types.h\nindex 6c75df30a281d1..53a1d22fcf957e 100644\n--- a/tools/include/linux/gfp_types.h\n+++ b/tools/include/linux/gfp_types.h\n@@ -33,7 +33,7 @@ enum {\n \t___GFP_IO_BIT,\n \t___GFP_FS_BIT,\n \t___GFP_ZERO_BIT,\n-\t___GFP_UNUSED_BIT,\t/* 0x200u unused */\n+\t___GFP_RESERVED_THP_BIT,\n \t___GFP_DIRECT_RECLAIM_BIT,\n \t___GFP_KSWAPD_RECLAIM_BIT,\n \t___GFP_WRITE_BIT,\n@@ -69,7 +69,7 @@ enum {\n #define ___GFP_IO\t\tBIT(___GFP_IO_BIT)\n #define ___GFP_FS\t\tBIT(___GFP_FS_BIT)\n #define ___GFP_ZERO\t\tBIT(___GFP_ZERO_BIT)\n-/* 0x200u unused */\n+#define ___GFP_RESERVED_THP\tBIT(___GFP_RESERVED_THP_BIT)\n #define ___GFP_DIRECT_RECLAIM\tBIT(___GFP_DIRECT_RECLAIM_BIT)\n #define ___GFP_KSWAPD_RECLAIM\tBIT(___GFP_KSWAPD_RECLAIM_BIT)\n #define ___GFP_WRITE\t\tBIT(___GFP_WRITE_BIT)\ndiff --git a/tools/perf/builtin-kmem.c b/tools/perf/builtin-kmem.c\nindex e1b2f5bc1ba8d8..45732aaf1a525e 100644\n--- a/tools/perf/builtin-kmem.c\n+++ b/tools/perf/builtin-kmem.c\n@@ -672,6 +672,7 @@ static const struct {\n \t{ \"__GFP_NORETRY\",\t\t\"NR\" },\n \t{ \"__GFP_COMP\",\t\t\t\"C\" },\n \t{ \"__GFP_ZERO\",\t\t\t\"Z\" },\n+\t{ \"__GFP_RESERVED_THP\",\t\t\"RTHP\" },\n \t{ \"__GFP_NOMEMALLOC\",\t\t\"NMA\" },\n \t{ \"__GFP_MEMALLOC\",\t\t\"MA\" },\n \t{ \"__GFP_HARDWALL\",\t\t\"HW\" },\ndiff --git a/tools/testing/vma/include/dup.h b/tools/testing/vma/include/dup.h\nindex cf73bcd9bb9d52..022b7a56b6a9f3 100644\n--- a/tools/testing/vma/include/dup.h\n+++ b/tools/testing/vma/include/dup.h\n@@ -160,6 +160,7 @@ enum {\n #endif\n \tDECLARE_VMA_BIT(UFFD_MINOR, 41),\n \tDECLARE_VMA_BIT(SEALED, 42),\n+\tDECLARE_VMA_BIT(RESERVED_THP, 43),\n \t/* Flags that reuse flags above. */\n \tDECLARE_VMA_BIT_ALIAS(PKEY_BIT0, HIGH_ARCH_0),\n \tDECLARE_VMA_BIT_ALIAS(PKEY_BIT1, HIGH_ARCH_1),\n"
}

2/1 2026/06/27 07:39 agent patch-evaluator
0m Model:
gemini-3.1-pro-preview

Args:
null

Results:
{
  "EnableConfigs": [
    "TRANSPARENT_HUGEPAGE"
  ],
  "FocusSymbols": [
    "madvise_hugepage_policy",
    "reserved_thp_charge",
    "reserved_thp_uncharge",
    "reserved_thp_pageblocks",
    "check_reserved_thp_alignment",
    "vma_alloc_anon_folio_pmd"
  ],
  "Reasoning": "The patch introduces a new MADV_RESERVED_THP madvise behavior and VM_RESERVED_THP VMA flag, modifying core memory management logic including page fault handling, mremap, and VMA splitting. This is a functional change to the kernel's memory management subsystem that should be fuzzed.",
  "WorthFuzzing": true
}

Instruction:
You are an expert Linux kernel maintainer.
Your job is to review a provided patch series and determine
if it makes functional changes to the kernel that should be fuzzed.

IMPORTANT: The changes have ALREADY been applied and committed as the HEAD commit in
your workspace. Do NOT rely on your internal knowledge of the kernel. You must actively
use your code access tools to examine the actual source code and confirm any assumptions.

Return WorthFuzzing=false if the patch only contains:
- Modifications to Documentation/, Kconfig files, or code comments.
- Purely decorative changes, such as logging (e.g., pr_err, printk) or tracepoints.
- Changes to numeric constants or macros that do not functionally alter execution flow.
- Code paths that are impossible to reach in virtualized environments like GCE or QEMU,
even when utilizing software-emulated hardware (e.g., usb gadget, mac80211_hwsim).

If it modifies reachable core kernel logic, drivers, or architectures, use your code search
tools to verify the code can be executed, then return WorthFuzzing=true.

When returning WorthFuzzing=true, you MUST ALSO:
1. Extract any specific kernel functions that should be heavily fuzzed into FocusSymbols.
   Avoid listing generic hot-path functions to prevent skewed test distributions.
2. Identify any specific CONFIG_ options required to properly test this new/modified feature.
   Go and look into the Kconfig files and check for ifdefs around the code, do not make assumptions.
   Do not list too generic configs (we already have them enabled). Only list those that
   specifically cover the modified code. List them in the EnableConfigs output array,
   and DO NOT add a 'CONFIG_' prefix (e.g., return "NET_IPV4" instead of "CONFIG_NET_IPV4").
Prefer calling several tools at the same time to save round-trips.


Use set-results tool to provide results of the analysis.
It must be called exactly once before the final reply.
Ignore results of this tool.

Prompt:
For your convenience, here is the diff of the changes:
commit 88c2815565df6c04ad94ebfa480054c81836fe78
Author: syz-cluster <triage@syzkaller.com>
Date:   Sat Jun 27 07:39:54 2026 +0000

    syz-cluster: applied patch under review

diff --git a/arch/alpha/include/uapi/asm/mman.h b/arch/alpha/include/uapi/asm/mman.h
index 1e700468a68580..672a2fc3438614 100644
--- a/arch/alpha/include/uapi/asm/mman.h
+++ b/arch/alpha/include/uapi/asm/mman.h
@@ -78,6 +78,8 @@
 
 #define MADV_COLLAPSE	25		/* Synchronous hugepage collapse */
 
+#define MADV_RESERVED_THP 26		/* Use reserved transparent hugepages */
+
 #define MADV_GUARD_INSTALL 102		/* fatal signal on access to range */
 #define MADV_GUARD_REMOVE 103		/* unguard range */
 
diff --git a/arch/mips/include/uapi/asm/mman.h b/arch/mips/include/uapi/asm/mman.h
index b700dae28c482d..a94bf74dee21ca 100644
--- a/arch/mips/include/uapi/asm/mman.h
+++ b/arch/mips/include/uapi/asm/mman.h
@@ -105,6 +105,8 @@
 
 #define MADV_COLLAPSE	25		/* Synchronous hugepage collapse */
 
+#define MADV_RESERVED_THP 26		/* Use reserved transparent hugepages */
+
 #define MADV_GUARD_INSTALL 102		/* fatal signal on access to range */
 #define MADV_GUARD_REMOVE 103		/* unguard range */
 
diff --git a/arch/parisc/include/uapi/asm/mman.h b/arch/parisc/include/uapi/asm/mman.h
index b6a709506987e0..fe2fddefb6c5d9 100644
--- a/arch/parisc/include/uapi/asm/mman.h
+++ b/arch/parisc/include/uapi/asm/mman.h
@@ -72,6 +72,8 @@
 
 #define MADV_COLLAPSE	25		/* Synchronous hugepage collapse */
 
+#define MADV_RESERVED_THP 26		/* Use reserved transparent hugepages */
+
 #define MADV_HWPOISON     100		/* poison a page for testing */
 #define MADV_SOFT_OFFLINE 101		/* soft offline page for testing */
 
diff --git a/arch/xtensa/include/uapi/asm/mman.h b/arch/xtensa/include/uapi/asm/mman.h
index 99d4ccee7f6e8a..bb603530ba799f 100644
--- a/arch/xtensa/include/uapi/asm/mman.h
+++ b/arch/xtensa/include/uapi/asm/mman.h
@@ -113,6 +113,8 @@
 
 #define MADV_COLLAPSE	25		/* Synchronous hugepage collapse */
 
+#define MADV_RESERVED_THP 26		/* Use reserved transparent hugepages */
+
 #define MADV_GUARD_INSTALL 102		/* fatal signal on access to range */
 #define MADV_GUARD_REMOVE 103		/* unguard range */
 
diff --git a/fs/proc/task_mmu.c b/fs/proc/task_mmu.c
index d32408f7cd5ed2..65c4b2a61aeac7 100644
--- a/fs/proc/task_mmu.c
+++ b/fs/proc/task_mmu.c
@@ -1245,6 +1245,9 @@ static void show_smap_vma_flags(struct seq_file *m, struct vm_area_struct *vma)
 #endif
 #ifdef CONFIG_64BIT
 		[ilog2(VM_SEALED)] = "sl",
+#endif
+#ifdef CONFIG_TRANSPARENT_HUGEPAGE
+		[ilog2(VM_RESERVED_THP)] = "rt",
 #endif
 	};
 	size_t i;
diff --git a/include/linux/gfp.h b/include/linux/gfp.h
index cdf95a9f0b87c1..2d05929fd8c72d 100644
--- a/include/linux/gfp.h
+++ b/include/linux/gfp.h
@@ -30,6 +30,9 @@ static inline int gfp_migratetype(const gfp_t gfp_flags)
 	BUILD_BUG_ON(((___GFP_MOVABLE | ___GFP_RECLAIMABLE) >>
 		      GFP_MOVABLE_SHIFT) != MIGRATE_HIGHATOMIC);
 
+	if (unlikely(gfp_flags & __GFP_RESERVED_THP))
+		return MIGRATE_RESERVED_THP;
+
 	if (unlikely(page_group_by_mobility_disabled))
 		return MIGRATE_UNMOVABLE;
 
diff --git a/include/linux/gfp_types.h b/include/linux/gfp_types.h
index 54ca0c88bab6e8..1f82a9491d357f 100644
--- a/include/linux/gfp_types.h
+++ b/include/linux/gfp_types.h
@@ -33,7 +33,7 @@ enum {
 	___GFP_IO_BIT,
 	___GFP_FS_BIT,
 	___GFP_ZERO_BIT,
-	___GFP_UNUSED_BIT,	/* 0x200u unused */
+	___GFP_RESERVED_THP_BIT,
 	___GFP_DIRECT_RECLAIM_BIT,
 	___GFP_KSWAPD_RECLAIM_BIT,
 	___GFP_WRITE_BIT,
@@ -69,7 +69,7 @@ enum {
 #define ___GFP_IO		BIT(___GFP_IO_BIT)
 #define ___GFP_FS		BIT(___GFP_FS_BIT)
 #define ___GFP_ZERO		BIT(___GFP_ZERO_BIT)
-/* 0x200u unused */
+#define ___GFP_RESERVED_THP	BIT(___GFP_RESERVED_THP_BIT)
 #define ___GFP_DIRECT_RECLAIM	BIT(___GFP_DIRECT_RECLAIM_BIT)
 #define ___GFP_KSWAPD_RECLAIM	BIT(___GFP_KSWAPD_RECLAIM_BIT)
 #define ___GFP_WRITE		BIT(___GFP_WRITE_BIT)
@@ -141,6 +141,9 @@ enum {
  * %__GFP_NO_OBJ_EXT causes slab allocation to have no object extension.
  * mark_obj_codetag_empty() should be called upon freeing for objects allocated
  * with this flag to indicate that their NULL tags are expected and normal.
+ *
+ * %__GFP_RESERVED_THP is an internal flag for reserved THP faults. It restricts
+ *the allocation to %MIGRATE_RESERVED_THP pageblocks.
  */
 #define __GFP_RECLAIMABLE ((__force gfp_t)___GFP_RECLAIMABLE)
 #define __GFP_WRITE	((__force gfp_t)___GFP_WRITE)
@@ -148,6 +151,7 @@ enum {
 #define __GFP_THISNODE	((__force gfp_t)___GFP_THISNODE)
 #define __GFP_ACCOUNT	((__force gfp_t)___GFP_ACCOUNT)
 #define __GFP_NO_OBJ_EXT   ((__force gfp_t)___GFP_NO_OBJ_EXT)
+#define __GFP_RESERVED_THP ((__force gfp_t)___GFP_RESERVED_THP)
 
 /**
  * DOC: Watermark modifiers
diff --git a/include/linux/huge_mm.h b/include/linux/huge_mm.h
index ad20f7f8c17944..4fe9651cd86b5e 100644
--- a/include/linux/huge_mm.h
+++ b/include/linux/huge_mm.h
@@ -330,6 +330,8 @@ unsigned long thp_vma_allowable_orders(struct vm_area_struct *vma,
 
 		if (vm_flags & VM_HUGEPAGE)
 			mask |= READ_ONCE(huge_anon_orders_madvise);
+		if (vm_flags & VM_RESERVED_THP)
+			mask |= BIT(PMD_ORDER);
 		if (hugepage_global_always() ||
 		    ((vm_flags & VM_HUGEPAGE) && hugepage_global_enabled()))
 			mask |= READ_ONCE(huge_anon_orders_inherit);
@@ -371,7 +373,7 @@ static inline bool vma_thp_disabled(struct vm_area_struct *vma,
 	 * Are THPs disabled only for VMAs where we didn't get an explicit
 	 * advise to use them?
 	 */
-	if (vm_flags & VM_HUGEPAGE)
+	if (vm_flags & (VM_HUGEPAGE | VM_RESERVED_THP))
 		return false;
 	/*
 	 * Forcing a collapse (e.g., madv_collapse), is a clear advice to
diff --git a/include/linux/mm.h b/include/linux/mm.h
index 485df9c2dbddb3..278cf4bfd4ec5b 100644
--- a/include/linux/mm.h
+++ b/include/linux/mm.h
@@ -353,6 +353,7 @@ enum {
 #endif
 	DECLARE_VMA_BIT(UFFD_MINOR, 41),
 	DECLARE_VMA_BIT(SEALED, 42),
+	DECLARE_VMA_BIT(RESERVED_THP, 43),
 	/* Flags that reuse flags above. */
 	DECLARE_VMA_BIT_ALIAS(PKEY_BIT0, HIGH_ARCH_0),
 	DECLARE_VMA_BIT_ALIAS(PKEY_BIT1, HIGH_ARCH_1),
@@ -526,6 +527,12 @@ enum {
 #define VMA_DROPPABLE		EMPTY_VMA_FLAGS
 #endif
 
+#ifdef CONFIG_TRANSPARENT_HUGEPAGE
+#define VM_RESERVED_THP		INIT_VM_FLAG(RESERVED_THP)
+#else
+#define VM_RESERVED_THP		VM_NONE
+#endif
+
 /* Bits set in the VMA until the stack is in its final location */
 #define VM_STACK_INCOMPLETE_SETUP (VM_RAND_READ | VM_SEQ_READ | VM_STACK_EARLY)
 
diff --git a/include/linux/mmzone.h b/include/linux/mmzone.h
index ca27121871475c..4418d0c9accdc8 100644
--- a/include/linux/mmzone.h
+++ b/include/linux/mmzone.h
@@ -133,10 +133,9 @@ enum migratetype {
 	 * __free_pageblock_cma() function.
 	 */
 	MIGRATE_CMA,
-	__MIGRATE_TYPE_END = MIGRATE_CMA,
-#else
-	__MIGRATE_TYPE_END = MIGRATE_HIGHATOMIC,
 #endif
+	MIGRATE_RESERVED_THP,
+	__MIGRATE_TYPE_END = MIGRATE_RESERVED_THP,
 #ifdef CONFIG_MEMORY_ISOLATION
 	MIGRATE_ISOLATE,	/* can't allocate from here */
 #endif
@@ -161,6 +160,9 @@ extern const char * const migratetype_names[MIGRATE_TYPES];
 #  define is_migrate_cma_folio(folio, pfn) false
 #endif
 
+#define is_migrate_reserved_thp(migratetype) \
+	unlikely((migratetype) == MIGRATE_RESERVED_THP)
+
 static inline bool is_migrate_movable(int mt)
 {
 	return is_migrate_cma(mt) || mt == MIGRATE_MOVABLE;
@@ -975,6 +977,9 @@ struct zone {
 	unsigned long nr_reserved_highatomic;
 	unsigned long nr_free_highatomic;
 
+	unsigned long nr_reserved_thp;
+	unsigned long nr_free_reserved_thp;
+
 	/*
 	 * We don't know if the memory that we're going to allocate will be
 	 * freeable or/and it will be released eventually, so to avoid totally
diff --git a/include/trace/events/mmflags.h b/include/trace/events/mmflags.h
index a6e5a44c9b4299..3db40ebd7060bd 100644
--- a/include/trace/events/mmflags.h
+++ b/include/trace/events/mmflags.h
@@ -24,6 +24,7 @@
 	TRACE_GFP_EM(IO)			\
 	TRACE_GFP_EM(FS)			\
 	TRACE_GFP_EM(ZERO)			\
+	TRACE_GFP_EM(RESERVED_THP)		\
 	TRACE_GFP_EM(DIRECT_RECLAIM)		\
 	TRACE_GFP_EM(KSWAPD_RECLAIM)		\
 	TRACE_GFP_EM(WRITE)			\
@@ -72,8 +73,7 @@
 
 TRACE_GFP_FLAGS
 
-/* Just in case these are ever used */
-TRACE_DEFINE_ENUM(___GFP_UNUSED_BIT);
+/* Just in case this is ever used */
 TRACE_DEFINE_ENUM(___GFP_LAST_BIT);
 
 #define gfpflag_string(flag) {(__force unsigned long)flag, #flag}
diff --git a/include/uapi/asm-generic/mman-common.h b/include/uapi/asm-generic/mman-common.h
index ef1c27fa3c570f..b3d1448935ead1 100644
--- a/include/uapi/asm-generic/mman-common.h
+++ b/include/uapi/asm-generic/mman-common.h
@@ -79,6 +79,8 @@
 
 #define MADV_COLLAPSE	25		/* Synchronous hugepage collapse */
 
+#define MADV_RESERVED_THP 26		/* Use reserved transparent hugepages */
+
 #define MADV_GUARD_INSTALL 102		/* fatal signal on access to range */
 #define MADV_GUARD_REMOVE 103		/* unguard range */
 
diff --git a/mm/Makefile b/mm/Makefile
index eff9f9e7e061c1..fd74a7392e3467 100644
--- a/mm/Makefile
+++ b/mm/Makefile
@@ -98,7 +98,7 @@ obj-$(CONFIG_MEMTEST)		+= memtest.o
 obj-$(CONFIG_MIGRATION) += migrate.o
 obj-$(CONFIG_NUMA) += memory-tiers.o
 obj-$(CONFIG_DEVICE_MIGRATION) += migrate_device.o
-obj-$(CONFIG_TRANSPARENT_HUGEPAGE) += huge_memory.o khugepaged.o
+obj-$(CONFIG_TRANSPARENT_HUGEPAGE) += huge_memory.o khugepaged.o reserved_thp.o
 obj-$(CONFIG_PAGE_COUNTER) += page_counter.o
 obj-$(CONFIG_LIVEUPDATE_MEMFD) += memfd_luo.o
 obj-$(CONFIG_MEMCG_V1) += memcontrol-v1.o
diff --git a/mm/huge_memory.c b/mm/huge_memory.c
index 2bccb0a53a0a60..66d85a2fa855f3 100644
--- a/mm/huge_memory.c
+++ b/mm/huge_memory.c
@@ -1267,6 +1267,9 @@ static struct folio *vma_alloc_anon_folio_pmd(struct vm_area_struct *vma,
 	const int order = HPAGE_PMD_ORDER;
 	struct folio *folio;
 
+	if (vma->vm_flags & VM_RESERVED_THP)
+		gfp |= __GFP_RESERVED_THP;
+
 	folio = vma_alloc_folio(gfp, order, vma, addr & HPAGE_PMD_MASK);
 
 	if (unlikely(!folio)) {
@@ -1344,8 +1347,11 @@ static vm_fault_t __do_huge_pmd_anonymous_page(struct vm_fault *vmf)
 	vm_fault_t ret = 0;
 
 	folio = vma_alloc_anon_folio_pmd(vma, vmf->address);
-	if (unlikely(!folio))
+	if (unlikely(!folio)) {
+		if (vma->vm_flags & VM_RESERVED_THP)
+			return VM_FAULT_OOM;
 		return VM_FAULT_FALLBACK;
+	}
 
 	pgtable = pte_alloc_one(vma->vm_mm);
 	if (unlikely(!pgtable)) {
@@ -1480,15 +1486,17 @@ vm_fault_t do_huge_pmd_anonymous_page(struct vm_fault *vmf)
 	vm_fault_t ret;
 
 	if (!thp_vma_suitable_order(vma, haddr, PMD_ORDER))
-		return VM_FAULT_FALLBACK;
+		return (vma->vm_flags & VM_RESERVED_THP) ? VM_FAULT_OOM :
+							   VM_FAULT_FALLBACK;
 	ret = vmf_anon_prepare(vmf);
 	if (ret)
 		return ret;
 	khugepaged_enter_vma(vma, vma->vm_flags);
 
-	if (!(vmf->flags & FAULT_FLAG_WRITE) &&
-			!mm_forbids_zeropage(vma->vm_mm) &&
-			transparent_hugepage_use_zero_page()) {
+	if (!(vma->vm_flags & VM_RESERVED_THP) &&
+	    !(vmf->flags & FAULT_FLAG_WRITE) &&
+	    !mm_forbids_zeropage(vma->vm_mm) &&
+	    transparent_hugepage_use_zero_page()) {
 		pgtable_t pgtable;
 		struct folio *zero_folio;
 		vm_fault_t ret;
diff --git a/mm/internal.h b/mm/internal.h
index 181e79f1d6a207..4b2a13d3537721 100644
--- a/mm/internal.h
+++ b/mm/internal.h
@@ -1477,6 +1477,7 @@ unsigned int reclaim_clean_pages_from_list(struct zone *zone,
 #define ALLOC_HIGHATOMIC	0x200 /* Allows access to MIGRATE_HIGHATOMIC */
 #define ALLOC_TRYLOCK		0x400 /* Only use spin_trylock in allocation path */
 #define ALLOC_KSWAPD		0x800 /* allow waking of kswapd, __GFP_KSWAPD_RECLAIM set */
+#define ALLOC_RESERVED_THP	0x1000 /* Allows access to reserved THP pageblocks */
 
 /* Flags that allow allocations below the min watermark. */
 #define ALLOC_RESERVES (ALLOC_NON_BLOCK|ALLOC_MIN_RESERVE|ALLOC_HIGHATOMIC|ALLOC_OOM)
@@ -1951,4 +1952,9 @@ static inline int get_sysctl_max_map_count(void)
 bool may_expand_vm(struct mm_struct *mm, const vma_flags_t *vma_flags,
 		   unsigned long npages);
 
+unsigned long reserved_thp_pageblocks(unsigned long nr_hpages);
+unsigned long reserved_thp_hpage_nr(unsigned long start, unsigned long end);
+int reserved_thp_charge(unsigned long nr_hpages);
+void reserved_thp_uncharge(unsigned long nr_hpages);
+
 #endif	/* __MM_INTERNAL_H */
diff --git a/mm/khugepaged.c b/mm/khugepaged.c
index 617bca76db49b6..80293e8c1e4e76 100644
--- a/mm/khugepaged.c
+++ b/mm/khugepaged.c
@@ -451,6 +451,7 @@ int hugepage_madvise(struct vm_area_struct *vma,
 	switch (advice) {
 	case MADV_HUGEPAGE:
 		*vm_flags &= ~VM_NOHUGEPAGE;
+		*vm_flags &= ~VM_RESERVED_THP;
 		*vm_flags |= VM_HUGEPAGE;
 		/*
 		 * If the vma become good for khugepaged to scan,
@@ -461,6 +462,7 @@ int hugepage_madvise(struct vm_area_struct *vma,
 		break;
 	case MADV_NOHUGEPAGE:
 		*vm_flags &= ~VM_HUGEPAGE;
+		*vm_flags &= ~VM_RESERVED_THP;
 		*vm_flags |= VM_NOHUGEPAGE;
 		/*
 		 * Setting VM_NOHUGEPAGE will prevent khugepaged from scanning
@@ -468,6 +470,12 @@ int hugepage_madvise(struct vm_area_struct *vma,
 		 * it got registered before VM_NOHUGEPAGE was set.
 		 */
 		break;
+	case MADV_RESERVED_THP:
+		*vm_flags &= ~(VM_HUGEPAGE | VM_NOHUGEPAGE);
+		*vm_flags |= VM_RESERVED_THP;
+		break;
+	default:
+		return -EINVAL;
 	}
 
 	return 0;
diff --git a/mm/madvise.c b/mm/madvise.c
index cd9bb077072ccb..dd91105db68c75 100644
--- a/mm/madvise.c
+++ b/mm/madvise.c
@@ -13,6 +13,7 @@
 #include <linux/page-isolation.h>
 #include <linux/page_idle.h>
 #include <linux/userfaultfd_k.h>
+#include <linux/huge_mm.h>
 #include <linux/hugetlb.h>
 #include <linux/falloc.h>
 #include <linux/fadvise.h>
@@ -1331,6 +1332,65 @@ static bool can_madvise_modify(struct madvise_behavior *madv_behavior)
 }
 #endif
 
+static bool reserved_thp_madvise_aligned(struct vm_area_struct *vma,
+					 struct madvise_behavior_range *range)
+{
+	if (!(vma->vm_flags & VM_RESERVED_THP))
+		return true;
+
+	return IS_ALIGNED(range->start, HPAGE_PMD_SIZE) &&
+	       IS_ALIGNED(range->end, HPAGE_PMD_SIZE);
+}
+
+static int madvise_hugepage_policy(struct madvise_behavior *madv_behavior,
+				   vm_flags_t *new_flags,
+				   unsigned long *reserved_hpages,
+				   bool *charge_reserved_thp,
+				   bool *uncharge_reserved_thp)
+{
+	struct vm_area_struct *vma = madv_behavior->vma;
+	struct madvise_behavior_range *range = &madv_behavior->range;
+	unsigned long hpages;
+	int behavior = madv_behavior->behavior;
+	int error;
+
+	switch (behavior) {
+	case MADV_HUGEPAGE:
+	case MADV_NOHUGEPAGE:
+		error = hugepage_madvise(vma, new_flags, behavior);
+		if (error)
+			return error;
+		*uncharge_reserved_thp = (vma->vm_flags & VM_RESERVED_THP) &&
+					 !(*new_flags & VM_RESERVED_THP);
+		return 0;
+	case MADV_RESERVED_THP:
+		if (!IS_ENABLED(CONFIG_64BIT))
+			return -EINVAL;
+		if (!vma_is_anonymous(vma) || (*new_flags & VM_SHARED) ||
+		    (*new_flags & VM_SPECIAL))
+			return -EINVAL;
+		if (!IS_ALIGNED(range->start, HPAGE_PMD_SIZE) ||
+		    !IS_ALIGNED(range->end, HPAGE_PMD_SIZE))
+			return -EINVAL;
+
+		error = hugepage_madvise(vma, new_flags, behavior);
+		if (error)
+			return error;
+
+		if (!(vma->vm_flags & VM_RESERVED_THP)) {
+			hpages = reserved_thp_hpage_nr(range->start, range->end);
+			error = reserved_thp_charge(hpages);
+			if (error)
+				return error;
+			*reserved_hpages = hpages;
+			*charge_reserved_thp = true;
+		}
+		return 0;
+	default:
+		return -EINVAL;
+	}
+}
+
 /*
  * Apply an madvise behavior to a region of a vma.  madvise_update_vma
  * will handle splitting a vm area into separate areas, each area with its own
@@ -1342,6 +1402,9 @@ static int madvise_vma_behavior(struct madvise_behavior *madv_behavior)
 	struct vm_area_struct *vma = madv_behavior->vma;
 	vm_flags_t new_flags = vma->vm_flags;
 	struct madvise_behavior_range *range = &madv_behavior->range;
+	unsigned long reserved_hpages = 0;
+	bool charge_reserved_thp = false;
+	bool uncharge_reserved_thp = false;
 	int error;
 
 	if (unlikely(!can_madvise_modify(madv_behavior)))
@@ -1353,14 +1416,22 @@ static int madvise_vma_behavior(struct madvise_behavior *madv_behavior)
 	case MADV_WILLNEED:
 		return madvise_willneed(madv_behavior);
 	case MADV_COLD:
+		if (!reserved_thp_madvise_aligned(vma, range))
+			return -EINVAL;
 		return madvise_cold(madv_behavior);
 	case MADV_PAGEOUT:
+		if (!reserved_thp_madvise_aligned(vma, range))
+			return -EINVAL;
 		return madvise_pageout(madv_behavior);
 	case MADV_FREE:
 	case MADV_DONTNEED:
 	case MADV_DONTNEED_LOCKED:
+		if (!reserved_thp_madvise_aligned(vma, range))
+			return -EINVAL;
 		return madvise_dontneed_free(madv_behavior);
 	case MADV_COLLAPSE:
+		if (vma->vm_flags & VM_RESERVED_THP)
+			return -EINVAL;
 		return madvise_collapse(vma, range->start, range->end,
 			&madv_behavior->lock_dropped);
 	case MADV_GUARD_INSTALL:
@@ -1416,7 +1487,11 @@ static int madvise_vma_behavior(struct madvise_behavior *madv_behavior)
 		break;
 	case MADV_HUGEPAGE:
 	case MADV_NOHUGEPAGE:
-		error = hugepage_madvise(vma, &new_flags, behavior);
+	case MADV_RESERVED_THP:
+		error = madvise_hugepage_policy(madv_behavior, &new_flags,
+						&reserved_hpages,
+						&charge_reserved_thp,
+						&uncharge_reserved_thp);
 		if (error)
 			goto out;
 		break;
@@ -1431,6 +1506,11 @@ static int madvise_vma_behavior(struct madvise_behavior *madv_behavior)
 	VM_WARN_ON_ONCE(madv_behavior->lock_mode != MADVISE_MMAP_WRITE_LOCK);
 
 	error = madvise_update_vma(new_flags, madv_behavior);
+	if (error && charge_reserved_thp)
+		reserved_thp_uncharge(reserved_hpages);
+	else if (!error && uncharge_reserved_thp)
+		reserved_thp_uncharge(reserved_thp_hpage_nr(range->start,
+							    range->end));
 out:
 	/*
 	 * madvise() returns EAGAIN if kernel resources, such as
@@ -1541,6 +1621,7 @@ madvise_behavior_valid(int behavior)
 	case MADV_HUGEPAGE:
 	case MADV_NOHUGEPAGE:
 	case MADV_COLLAPSE:
+	case MADV_RESERVED_THP:
 #endif
 	case MADV_DONTDUMP:
 	case MADV_DODUMP:
diff --git a/mm/memory.c b/mm/memory.c
index ff338c2abe9231..225fc1ae22386d 100644
--- a/mm/memory.c
+++ b/mm/memory.c
@@ -5297,6 +5297,9 @@ static vm_fault_t do_anonymous_page(struct vm_fault *vmf)
 	if (vma->vm_flags & VM_SHARED)
 		return VM_FAULT_SIGBUS;
 
+	if (unlikely(vma->vm_flags & VM_RESERVED_THP))
+		return VM_FAULT_OOM;
+
 	/*
 	 * Use pte_alloc() instead of pte_alloc_map(), so that OOM can
 	 * be distinguished from a transient failure of pte_offset_map().
diff --git a/mm/mmap.c b/mm/mmap.c
index 2311ae7c2ff45c..4818b14ec0ff6c 100644
--- a/mm/mmap.c
+++ b/mm/mmap.c
@@ -1251,6 +1251,7 @@ unsigned long tear_down_vmas(struct mm_struct *mm, struct vma_iterator *vmi,
 		struct vm_area_struct *vma, unsigned long end)
 {
 	unsigned long nr_accounted = 0;
+	unsigned long nr_reserved_thp = 0;
 	int count = 0;
 
 	mmap_assert_write_locked(mm);
@@ -1258,6 +1259,10 @@ unsigned long tear_down_vmas(struct mm_struct *mm, struct vma_iterator *vmi,
 	do {
 		if (vma->vm_flags & VM_ACCOUNT)
 			nr_accounted += vma_pages(vma);
+		if (vma->vm_flags & VM_RESERVED_THP)
+			nr_reserved_thp +=
+				reserved_thp_hpage_nr(vma->vm_start,
+						      vma->vm_end);
 		vma_mark_detached(vma);
 		remove_vma(vma);
 		count++;
@@ -1266,6 +1271,7 @@ unsigned long tear_down_vmas(struct mm_struct *mm, struct vma_iterator *vmi,
 	} while (vma && vma->vm_end <= end);
 
 	VM_WARN_ON_ONCE(count != mm->map_count);
+	reserved_thp_uncharge(nr_reserved_thp);
 	return nr_accounted;
 }
 
@@ -1733,6 +1739,7 @@ __latent_entropy int dup_mmap(struct mm_struct *mm, struct mm_struct *oldmm)
 	struct vm_area_struct *mpnt, *tmp;
 	int retval;
 	unsigned long charge = 0;
+	unsigned long reserved_charge = 0;
 	LIST_HEAD(uf);
 	VMA_ITERATOR(vmi, mm, 0);
 
@@ -1775,6 +1782,7 @@ __latent_entropy int dup_mmap(struct mm_struct *mm, struct mm_struct *oldmm)
 			continue;
 		}
 		charge = 0;
+		reserved_charge = 0;
 		if (mpnt->vm_flags & VM_ACCOUNT) {
 			unsigned long len = vma_pages(mpnt);
 
@@ -1782,6 +1790,15 @@ __latent_entropy int dup_mmap(struct mm_struct *mm, struct mm_struct *oldmm)
 				goto fail_nomem;
 			charge = len;
 		}
+		if (mpnt->vm_flags & VM_RESERVED_THP) {
+			unsigned long len;
+
+			len = reserved_thp_hpage_nr(mpnt->vm_start,
+						    mpnt->vm_end);
+			if (reserved_thp_charge(len))
+				goto fail_nomem;
+			reserved_charge = len;
+		}
 
 		tmp = vm_area_dup(mpnt);
 		if (!tmp)
@@ -1916,6 +1933,7 @@ __latent_entropy int dup_mmap(struct mm_struct *mm, struct mm_struct *oldmm)
 	vm_area_free(tmp);
 fail_nomem:
 	retval = -ENOMEM;
+	reserved_thp_uncharge(reserved_charge);
 	vm_unacct_memory(charge);
 	goto loop_out;
 }
diff --git a/mm/mremap.c b/mm/mremap.c
index e9c8b1d05832be..ae37e0b3ce788e 100644
--- a/mm/mremap.c
+++ b/mm/mremap.c
@@ -24,6 +24,7 @@
 #include <linux/mmu_notifier.h>
 #include <linux/uaccess.h>
 #include <linux/userfaultfd_k.h>
+#include <linux/huge_mm.h>
 #include <linux/mempolicy.h>
 #include <linux/pgalloc.h>
 
@@ -69,6 +70,7 @@ struct vma_remap_struct {
 	enum mremap_type remap_type;	/* expand, shrink, etc. */
 	bool mmap_locked;		/* Is mm currently write-locked? */
 	unsigned long charged;		/* If VM_ACCOUNT, # pages to account. */
+	unsigned long reserved_thp_charged; /* If VM_RESERVED_THP, # hpages. */
 	bool vmi_needs_invalidate;	/* Is the VMA iterator invalidated? */
 };
 
@@ -962,6 +964,9 @@ static unsigned long vrm_set_new_addr(struct vma_remap_struct *vrm)
 				map_flags);
 	if (IS_ERR_VALUE(res))
 		return res;
+	if ((vma->vm_flags & VM_RESERVED_THP) &&
+	    !IS_ALIGNED(res, HPAGE_PMD_SIZE))
+		return -ENOMEM;
 
 	vrm->new_addr = res;
 	return 0;
@@ -977,24 +982,44 @@ static bool vrm_calc_charge(struct vma_remap_struct *vrm)
 {
 	unsigned long charged;
 
-	if (!(vrm->vma->vm_flags & VM_ACCOUNT))
-		return true;
+	vrm->charged = 0;
+	vrm->reserved_thp_charged = 0;
 
-	/*
-	 * If we don't unmap the old mapping, then we account the entirety of
-	 * the length of the new one. Otherwise it's just the delta in size.
-	 */
-	if (vrm->flags & MREMAP_DONTUNMAP)
-		charged = vrm->new_len >> PAGE_SHIFT;
-	else
-		charged = vrm->delta >> PAGE_SHIFT;
+	if (vrm->vma->vm_flags & VM_ACCOUNT) {
+		/*
+		 * If we don't unmap the old mapping, then we account the
+		 * entirety of the length of the new one. Otherwise it's just
+		 * the delta in size.
+		 */
+		if (vrm->flags & MREMAP_DONTUNMAP)
+			charged = vrm->new_len >> PAGE_SHIFT;
+		else
+			charged = vrm->delta >> PAGE_SHIFT;
 
 
-	/* This accounts 'charged' pages of memory. */
-	if (security_vm_enough_memory_mm(current->mm, charged))
-		return false;
+		/* This accounts 'charged' pages of memory. */
+		if (security_vm_enough_memory_mm(current->mm, charged))
+			return false;
 
-	vrm->charged = charged;
+		vrm->charged = charged;
+	}
+
+	if (vrm->vma->vm_flags & VM_RESERVED_THP) {
+		unsigned long hpages;
+
+		if (vrm->flags & MREMAP_DONTUNMAP)
+			hpages = reserved_thp_hpage_nr(0, vrm->new_len);
+		else
+			hpages = reserved_thp_hpage_nr(0, vrm->delta);
+
+		if (reserved_thp_charge(hpages)) {
+			vm_unacct_memory(vrm->charged);
+			vrm->charged = 0;
+			return false;
+		}
+
+		vrm->reserved_thp_charged = hpages;
+	}
 	return true;
 }
 
@@ -1004,11 +1029,10 @@ static bool vrm_calc_charge(struct vma_remap_struct *vrm)
  */
 static void vrm_uncharge(struct vma_remap_struct *vrm)
 {
-	if (!(vrm->vma->vm_flags & VM_ACCOUNT))
-		return;
-
 	vm_unacct_memory(vrm->charged);
 	vrm->charged = 0;
+	reserved_thp_uncharge(vrm->reserved_thp_charged);
+	vrm->reserved_thp_charged = 0;
 }
 
 /*
@@ -1157,8 +1181,8 @@ static void unmap_source_vma(struct vma_remap_struct *vrm)
 	struct vm_area_struct *vma = vrm->vma;
 	VMA_ITERATOR(vmi, mm, addr);
 	int err;
-	unsigned long vm_start;
-	unsigned long vm_end;
+	unsigned long vm_start = 0;
+	unsigned long vm_end = 0;
 	/*
 	 * It might seem odd that we check for MREMAP_DONTUNMAP here, given this
 	 * function implies that we unmap the original VMA, which seems
@@ -1170,6 +1194,8 @@ static void unmap_source_vma(struct vma_remap_struct *vrm)
 	 */
 	bool accountable_move = (vma->vm_flags & VM_ACCOUNT) &&
 		!(vrm->flags & MREMAP_DONTUNMAP);
+	bool reserved_thp_move = (vma->vm_flags & VM_RESERVED_THP) &&
+		!(vrm->flags & MREMAP_DONTUNMAP);
 
 	/*
 	 * So we perform a trick here to prevent incorrect accounting. Any merge
@@ -1192,6 +1218,13 @@ static void unmap_source_vma(struct vma_remap_struct *vrm)
 		vm_start = vma->vm_start;
 		vm_end = vma->vm_end;
 	}
+	if (reserved_thp_move) {
+		vm_flags_clear(vma, VM_RESERVED_THP);
+		if (!accountable_move) {
+			vm_start = vma->vm_start;
+			vm_end = vma->vm_end;
+		}
+	}
 
 	err = do_vmi_munmap(&vmi, mm, addr, len, vrm->uf_unmap, /* unlock= */false);
 	vrm->vma = NULL; /* Invalidated. */
@@ -1227,19 +1260,27 @@ static void unmap_source_vma(struct vma_remap_struct *vrm)
 	 *
 	 * do_vmi_munmap() will have restored the VMI back to addr.
 	 */
-	if (accountable_move) {
+	if (accountable_move || reserved_thp_move) {
 		unsigned long end = addr + len;
-
-		if (vm_start < addr) {
-			struct vm_area_struct *prev = vma_prev(&vmi);
-
-			vm_flags_set(prev, VM_ACCOUNT); /* Acquires VMA lock. */
+		struct vm_area_struct *prev = NULL;
+		struct vm_area_struct *next = NULL;
+
+		if (vm_start < addr)
+			prev = vma_prev(&vmi);
+		if (vm_end > end)
+			next = vma_next(&vmi);
+
+		if (accountable_move) {
+			if (prev)
+				vm_flags_set(prev, VM_ACCOUNT); /* Acquires VMA lock. */
+			if (next)
+				vm_flags_set(next, VM_ACCOUNT); /* Acquires VMA lock. */
 		}
-
-		if (vm_end > end) {
-			struct vm_area_struct *next = vma_next(&vmi);
-
-			vm_flags_set(next, VM_ACCOUNT); /* Acquires VMA lock. */
+		if (reserved_thp_move) {
+			if (prev)
+				vm_flags_set(prev, VM_RESERVED_THP);
+			if (next)
+				vm_flags_set(next, VM_RESERVED_THP);
 		}
 	}
 }
@@ -1309,7 +1350,6 @@ static int copy_vma_and_data(struct vma_remap_struct *vrm,
 	*new_vma_ptr = new_vma;
 	return err;
 }
-
 /*
  * Perform final tasks for MADV_DONTUNMAP operation, clearing mlock() flag on
  * remaining VMA by convention (it cannot be mlock()'d any longer, as pages in
@@ -1576,6 +1616,23 @@ static bool align_hugetlb(struct vma_remap_struct *vrm)
 	return true;
 }
 
+static bool check_reserved_thp_alignment(struct vma_remap_struct *vrm)
+{
+	if (!(vrm->vma->vm_flags & VM_RESERVED_THP))
+		return true;
+
+	if (!IS_ALIGNED(vrm->addr, HPAGE_PMD_SIZE) ||
+	    !IS_ALIGNED(vrm->old_len, HPAGE_PMD_SIZE) ||
+	    !IS_ALIGNED(vrm->new_len, HPAGE_PMD_SIZE))
+		return false;
+
+	if ((vrm->remap_type == MREMAP_EXPAND || vrm_implies_new_addr(vrm)) &&
+	    !IS_ALIGNED(vrm->new_addr, HPAGE_PMD_SIZE))
+		return false;
+
+	return true;
+}
+
 /*
  * We are mremap()'ing without specifying a fixed address to move to, but are
  * requesting that the VMA's size be increased.
@@ -1745,6 +1802,8 @@ static int check_prep_vma(struct vma_remap_struct *vrm)
 	/* For convenience, we set new_addr even if VMA won't move. */
 	if (!vrm_implies_new_addr(vrm))
 		vrm->new_addr = addr;
+	if (!check_reserved_thp_alignment(vrm))
+		return -EINVAL;
 
 	/* Below only meaningful if we expand or move a VMA. */
 	if (!vrm_will_map_new(vrm))
diff --git a/mm/page_alloc.c b/mm/page_alloc.c
index ee902a468c2f5b..660e501bf676b6 100644
--- a/mm/page_alloc.c
+++ b/mm/page_alloc.c
@@ -263,6 +263,7 @@ const char * const migratetype_names[MIGRATE_TYPES] = {
 #ifdef CONFIG_CMA
 	"CMA",
 #endif
+	"ReserveTHP",
 #ifdef CONFIG_MEMORY_ISOLATION
 	"Isolate",
 #endif
@@ -784,6 +785,9 @@ static inline void account_freepages(struct zone *zone, int nr_pages,
 	else if (migratetype == MIGRATE_HIGHATOMIC)
 		WRITE_ONCE(zone->nr_free_highatomic,
 			   zone->nr_free_highatomic + nr_pages);
+	else if (migratetype == MIGRATE_RESERVED_THP)
+		WRITE_ONCE(zone->nr_free_reserved_thp,
+			   zone->nr_free_reserved_thp + nr_pages);
 }
 
 /* Used for pages not on another list */
@@ -2456,6 +2460,9 @@ __rmqueue(struct zone *zone, unsigned int order, int migratetype,
 {
 	struct page *page;
 
+	if (alloc_flags & ALLOC_RESERVED_THP)
+		return __rmqueue_smallest(zone, order, MIGRATE_RESERVED_THP);
+
 	if (IS_ENABLED(CONFIG_CMA)) {
 		/*
 		 * Balance movable allocations between regular and CMA areas by
@@ -2960,7 +2967,8 @@ static void __free_frozen_pages(struct page *page, unsigned int order,
 	zone = page_zone(page);
 	migratetype = get_pfnblock_migratetype(page, pfn);
 	if (unlikely(migratetype >= MIGRATE_PCPTYPES)) {
-		if (unlikely(is_migrate_isolate(migratetype))) {
+		if (unlikely(is_migrate_reserved_thp(migratetype) ||
+			     is_migrate_isolate(migratetype))) {
 			free_one_page(zone, page, pfn, order, fpi_flags);
 			return;
 		}
@@ -3038,6 +3046,7 @@ void free_unref_folios(struct folio_batch *folios)
 
 		/* Different zone requires a different pcp lock */
 		if (zone != locked_zone ||
+		    is_migrate_reserved_thp(migratetype) ||
 		    is_migrate_isolate(migratetype)) {
 			if (pcp) {
 				pcp_spin_unlock(pcp);
@@ -3045,6 +3054,12 @@ void free_unref_folios(struct folio_batch *folios)
 				pcp = NULL;
 			}
 
+			if (is_migrate_reserved_thp(migratetype)) {
+				free_one_page(zone, &folio->page, pfn,
+					      order, FPI_NONE);
+				continue;
+			}
+
 			/*
 			 * Free isolated pages directly to the
 			 * allocator, see comment in free_frozen_pages.
@@ -3235,7 +3250,8 @@ struct page *rmqueue_buddy(struct zone *preferred_zone, struct zone *zone,
 			 * reserves as failing now is worse than failing a
 			 * high-order atomic allocation in the future.
 			 */
-			if (!page && (alloc_flags & (ALLOC_OOM|ALLOC_NON_BLOCK)))
+			if (!page && !(alloc_flags & ALLOC_RESERVED_THP) &&
+			    (alloc_flags & (ALLOC_OOM|ALLOC_NON_BLOCK)))
 				page = __rmqueue_smallest(zone, order, MIGRATE_HIGHATOMIC);
 
 			if (!page) {
@@ -3405,7 +3421,8 @@ struct page *rmqueue(struct zone *preferred_zone,
 {
 	struct page *page;
 
-	if (likely(pcp_allowed_order(order))) {
+	if (likely(pcp_allowed_order(order)) &&
+	    !(alloc_flags & ALLOC_RESERVED_THP)) {
 		page = rmqueue_pcplist(preferred_zone, zone, order,
 				       migratetype, alloc_flags);
 		if (likely(page))
@@ -3556,6 +3573,35 @@ static bool unreserve_highatomic_pageblock(const struct alloc_context *ac,
 	return false;
 }
 
+unsigned long reserved_thp_pageblocks(unsigned long nr_hpages)
+{
+	unsigned int order = max_t(unsigned int, HPAGE_PMD_ORDER,
+				   pageblock_order);
+	unsigned long hpages_per_block = 1UL << (order - HPAGE_PMD_ORDER);
+	unsigned long reserved = 0;
+	gfp_t gfp = (GFP_HIGHUSER | __GFP_COMP | __GFP_NOMEMALLOC |
+		     __GFP_NOWARN | __GFP_NORETRY);
+
+	while (reserved < nr_hpages) {
+		struct page *page;
+		struct zone *zone;
+		unsigned long flags;
+
+		page = alloc_pages(gfp, order);
+		if (!page)
+			break;
+
+		zone = page_zone(page);
+		spin_lock_irqsave(&zone->lock, flags);
+		change_pageblock_range(page, order, MIGRATE_RESERVED_THP);
+		zone->nr_reserved_thp += 1UL << order;
+		spin_unlock_irqrestore(&zone->lock, flags);
+		__free_pages(page, order);
+		reserved += hpages_per_block;
+	}
+	return reserved;
+}
+
 static inline long __zone_watermark_unusable_free(struct zone *z,
 				unsigned int order, unsigned int alloc_flags)
 {
@@ -3568,6 +3614,9 @@ static inline long __zone_watermark_unusable_free(struct zone *z,
 	if (likely(!(alloc_flags & ALLOC_RESERVES)))
 		unusable_free += READ_ONCE(z->nr_free_highatomic);
 
+	if (!(alloc_flags & ALLOC_RESERVED_THP))
+		unusable_free += READ_ONCE(z->nr_free_reserved_thp);
+
 #ifdef CONFIG_CMA
 	/* If allocation can't use CMA areas don't use free CMA pages */
 	if (!(alloc_flags & ALLOC_CMA))
@@ -3642,6 +3691,12 @@ bool __zone_watermark_ok(struct zone *z, unsigned int order, unsigned long mark,
 		if (!area->nr_free)
 			continue;
 
+		if (alloc_flags & ALLOC_RESERVED_THP) {
+			if (!free_area_empty(area, MIGRATE_RESERVED_THP))
+				return true;
+			continue;
+		}
+
 		for (mt = 0; mt < MIGRATE_PCPTYPES; mt++) {
 			if (!free_area_empty(area, mt))
 				return true;
@@ -3876,6 +3931,9 @@ get_page_from_freelist(gfp_t gfp_mask, unsigned int order, int alloc_flags,
 
 		cond_accept_memory(zone, order, alloc_flags);
 
+		if (alloc_flags & ALLOC_RESERVED_THP)
+			goto try_this_zone;
+
 		/*
 		 * Detect whether the number of free pages is below high
 		 * watermark.  If so, we will decrease pcp->high and free
@@ -5033,6 +5091,15 @@ static inline bool prepare_alloc_pages(gfp_t gfp_mask, unsigned int order,
 	ac->nodemask = nodemask;
 	ac->migratetype = gfp_migratetype(gfp_mask);
 
+	if (gfp_mask & __GFP_RESERVED_THP) {
+		if (!IS_ENABLED(CONFIG_TRANSPARENT_HUGEPAGE) ||
+		    WARN_ON_ONCE_GFP(order != HPAGE_PMD_ORDER, gfp_mask))
+			return false;
+
+		ac->migratetype = MIGRATE_RESERVED_THP;
+		*alloc_flags |= ALLOC_RESERVED_THP;
+	}
+
 	if (cpusets_enabled()) {
 		*alloc_gfp |= __GFP_HARDWALL;
 		/*
diff --git a/mm/reserved_thp.c b/mm/reserved_thp.c
new file mode 100644
index 00000000000000..931c539c15a709
--- /dev/null
+++ b/mm/reserved_thp.c
@@ -0,0 +1,133 @@
+// SPDX-License-Identifier: GPL-2.0
+
+#include <linux/mm.h>
+#include "internal.h"
+
+static DEFINE_SPINLOCK(reserved_thp_lock);
+
+static unsigned long reserved_thp_cmdline_size __initdata = HPAGE_PMD_SIZE;
+static bool reserved_thp_cmdline_size_valid __initdata = true;
+static unsigned long reserved_thp_requested __initdata;
+static unsigned long reserved_thp_total;
+static unsigned long reserved_thp_used;
+
+static int __init setup_reserved_thp_size(char *str)
+{
+	unsigned long size;
+	size = memparse(str, NULL);
+	if (size != HPAGE_PMD_SIZE) {
+		pr_warn("unsupported thp_reserved_size=%s, only %lu is supported\n",
+			str, HPAGE_PMD_SIZE);
+		reserved_thp_cmdline_size_valid = false;
+		return -EINVAL;
+	}
+	reserved_thp_cmdline_size = size;
+	reserved_thp_cmdline_size_valid = true;
+	return 0;
+}
+early_param("thp_reserved_size", setup_reserved_thp_size);
+static int __init setup_reserved_thp_nr(char *str)
+{
+	int count;
+	if (sscanf(str, "%lu%n", &reserved_thp_requested, &count) != 1 ||
+	    str[count]) {
+		pr_warn("invalid thp_reserved_nr=%s\n", str);
+		reserved_thp_requested = 0;
+		return -EINVAL;
+	}
+	return 0;
+}
+early_param("thp_reserved_nr", setup_reserved_thp_nr);
+
+unsigned long reserved_thp_hpage_nr(unsigned long start, unsigned long end)
+{
+	return (end - start) >> HPAGE_PMD_SHIFT;
+}
+
+int reserved_thp_charge(unsigned long nr_hpages)
+{
+	int ret = 0;
+
+	if (!nr_hpages)
+		return 0;
+
+	spin_lock(&reserved_thp_lock);
+	if (nr_hpages > reserved_thp_total - reserved_thp_used)
+		ret = -ENOMEM;
+	else
+		reserved_thp_used += nr_hpages;
+	spin_unlock(&reserved_thp_lock);
+
+	return ret;
+}
+
+void reserved_thp_uncharge(unsigned long nr_hpages)
+{
+	if (!nr_hpages)
+		return;
+
+	spin_lock(&reserved_thp_lock);
+	if (WARN_ON_ONCE(nr_hpages > reserved_thp_used))
+		reserved_thp_used = 0;
+	else
+		reserved_thp_used -= nr_hpages;
+	spin_unlock(&reserved_thp_lock);
+}
+
+static ssize_t total_hpages_show(struct kobject *kobj,
+				 struct kobj_attribute *attr, char *buf)
+{
+	return sysfs_emit(buf, "%lu\n", READ_ONCE(reserved_thp_total));
+}
+static ssize_t free_hpages_show(struct kobject *kobj,
+				struct kobj_attribute *attr, char *buf)
+{
+	unsigned long free_hpages;
+
+	spin_lock(&reserved_thp_lock);
+	free_hpages = reserved_thp_total - reserved_thp_used;
+	spin_unlock(&reserved_thp_lock);
+
+	return sysfs_emit(buf, "%lu\n", free_hpages);
+}
+static ssize_t used_hpages_show(struct kobject *kobj,
+				struct kobj_attribute *attr, char *buf)
+{
+	return sysfs_emit(buf, "%lu\n", READ_ONCE(reserved_thp_used));
+}
+
+static struct kobj_attribute total_hpages_attr = __ATTR_RO(total_hpages);
+static struct kobj_attribute free_hpages_attr = __ATTR_RO(free_hpages);
+static struct kobj_attribute used_hpages_attr = __ATTR_RO(used_hpages);
+
+static struct attribute *reserved_thp_attrs[] = {
+	&total_hpages_attr.attr,
+	&free_hpages_attr.attr,
+	&used_hpages_attr.attr,
+	NULL,
+};
+
+static const struct attribute_group reserved_thp_attr_group = {
+	.attrs = reserved_thp_attrs,
+};
+
+static int __init reserved_thp_init(void)
+{
+	struct kobject *kobj;
+	int ret;
+
+	if (reserved_thp_requested && reserved_thp_cmdline_size_valid) {
+		reserved_thp_total = reserved_thp_pageblocks(reserved_thp_requested);
+		pr_info("reserved %lu/%lu PMD THP pageblocks (%lu bytes each)\n",
+			reserved_thp_total, reserved_thp_requested,
+			reserved_thp_cmdline_size);
+	}
+	kobj = kobject_create_and_add("reserved_thp", mm_kobj);
+	if (!kobj)
+		return -ENOMEM;
+	ret = sysfs_create_group(kobj, &reserved_thp_attr_group);
+	if (ret)
+		kobject_put(kobj);
+	return ret;
+}
+subsys_initcall(reserved_thp_init);
\ No newline at end of file
diff --git a/mm/show_mem.c b/mm/show_mem.c
index 43aca5a2ac990a..e9381afca4acae 100644
--- a/mm/show_mem.c
+++ b/mm/show_mem.c
@@ -142,6 +142,7 @@ static void show_migration_types(unsigned char type)
 #ifdef CONFIG_CMA
 		[MIGRATE_CMA]		= 'C',
 #endif
+		[MIGRATE_RESERVED_THP]	= 'T',
 #ifdef CONFIG_MEMORY_ISOLATION
 		[MIGRATE_ISOLATE]	= 'I',
 #endif
@@ -308,6 +309,8 @@ static void show_free_areas(unsigned int filter, nodemask_t *nodemask, int max_z
 			" high:%lukB"
 			" reserved_highatomic:%luKB"
 			" free_highatomic:%luKB"
+			" reserved_thp:%luKB"
+			" free_reserved_thp:%luKB"
 			" active_anon:%lukB"
 			" inactive_anon:%lukB"
 			" active_file:%lukB"
@@ -331,6 +334,8 @@ static void show_free_areas(unsigned int filter, nodemask_t *nodemask, int max_z
 			K(high_wmark_pages(zone)),
 			K(zone->nr_reserved_highatomic),
 			K(zone->nr_free_highatomic),
+			K(zone->nr_reserved_thp),
+			K(zone->nr_free_reserved_thp),
 			K(zone_page_state(zone, NR_ZONE_ACTIVE_ANON)),
 			K(zone_page_state(zone, NR_ZONE_INACTIVE_ANON)),
 			K(zone_page_state(zone, NR_ZONE_ACTIVE_FILE)),
diff --git a/mm/vma.c b/mm/vma.c
index 9eea2850818a85..8c4cd7c97a984c 100644
--- a/mm/vma.c
+++ b/mm/vma.c
@@ -7,6 +7,8 @@
 #include "vma_internal.h"
 #include "vma.h"
 
+#include <linux/huge_mm.h>
+
 struct mmap_state {
 	struct mm_struct *mm;
 	struct vma_iterator *vmi;
@@ -507,6 +509,10 @@ __split_vma(struct vma_iterator *vmi, struct vm_area_struct *vma,
 	WARN_ON(vma->vm_start >= addr);
 	WARN_ON(vma->vm_end <= addr);
 
+	if ((vma->vm_flags & VM_RESERVED_THP) &&
+	    !IS_ALIGNED(addr, HPAGE_PMD_SIZE))
+		return -EINVAL;
+
 	if (vma->vm_ops && vma->vm_ops->may_split) {
 		err = vma->vm_ops->may_split(vma, addr);
 		if (err)
@@ -1361,6 +1367,7 @@ static void vms_complete_munmap_vmas(struct vma_munmap_struct *vms,
 		remove_vma(vma);
 
 	vm_unacct_memory(vms->nr_accounted);
+	reserved_thp_uncharge(vms->nr_reserved_thp);
 	validate_mm(mm);
 	if (vms->unlock)
 		mmap_read_unlock(mm);
@@ -1423,6 +1430,11 @@ static int vms_gather_munmap_vmas(struct vma_munmap_struct *vms,
 			error = -EPERM;
 			goto start_split_failed;
 		}
+		if ((vms->vma->vm_flags & VM_RESERVED_THP) &&
+		    !IS_ALIGNED(vms->start, HPAGE_PMD_SIZE)) {
+			error = -EINVAL;
+			goto start_split_failed;
+		}
 
 		error = __split_vma(vms->vmi, vms->vma, vms->start, 1);
 		if (error)
@@ -1445,6 +1457,11 @@ static int vms_gather_munmap_vmas(struct vma_munmap_struct *vms,
 		}
 		/* Does it split the end? */
 		if (next->vm_end > vms->end) {
+			if ((next->vm_flags & VM_RESERVED_THP) &&
+			    !IS_ALIGNED(vms->end, HPAGE_PMD_SIZE)) {
+				error = -EINVAL;
+				goto end_split_failed;
+			}
 			error = __split_vma(vms->vmi, next, vms->end, 0);
 			if (error)
 				goto end_split_failed;
@@ -1465,6 +1482,11 @@ static int vms_gather_munmap_vmas(struct vma_munmap_struct *vms,
 		if (vma_test(next, VMA_ACCOUNT_BIT))
 			vms->nr_accounted += nrpages;
 
+		if (next->vm_flags & VM_RESERVED_THP)
+			vms->nr_reserved_thp +=
+				reserved_thp_hpage_nr(next->vm_start,
+						      next->vm_end);
+
 		if (is_exec_mapping(next->vm_flags))
 			vms->exec_vm += nrpages;
 		else if (is_stack_mapping(next->vm_flags))
@@ -1560,6 +1582,7 @@ static void init_vma_munmap(struct vma_munmap_struct *vms,
 	vms->uf = uf;
 	vms->vma_count = 0;
 	vms->nr_pages = vms->locked_vm = vms->nr_accounted = 0;
+	vms->nr_reserved_thp = 0;
 	vms->exec_vm = vms->stack_vm = vms->data_vm = 0;
 	vms->unmap_start = FIRST_USER_ADDRESS;
 	vms->unmap_end = USER_PGTABLES_CEILING;
diff --git a/mm/vma.h b/mm/vma.h
index 8e4b61a7304c68..68e44adee5c89e 100644
--- a/mm/vma.h
+++ b/mm/vma.h
@@ -48,6 +48,7 @@ struct vma_munmap_struct {
 	unsigned long nr_pages;         /* Number of pages being removed */
 	unsigned long locked_vm;        /* Number of locked pages */
 	unsigned long nr_accounted;     /* Number of VM_ACCOUNT pages */
+	unsigned long nr_reserved_thp;  /* Number of reserved PMD THP slots */
 	unsigned long exec_vm;
 	unsigned long stack_vm;
 	unsigned long data_vm;
diff --git a/tools/include/linux/gfp_types.h b/tools/include/linux/gfp_types.h
index 6c75df30a281d1..53a1d22fcf957e 100644
--- a/tools/include/linux/gfp_types.h
+++ b/tools/include/linux/gfp_types.h
@@ -33,7 +33,7 @@ enum {
 	___GFP_IO_BIT,
 	___GFP_FS_BIT,
 	___GFP_ZERO_BIT,
-	___GFP_UNUSED_BIT,	/* 0x200u unused */
+	___GFP_RESERVED_THP_BIT,
 	___GFP_DIRECT_RECLAIM_BIT,
 	___GFP_KSWAPD_RECLAIM_BIT,
 	___GFP_WRITE_BIT,
@@ -69,7 +69,7 @@ enum {
 #define ___GFP_IO		BIT(___GFP_IO_BIT)
 #define ___GFP_FS		BIT(___GFP_FS_BIT)
 #define ___GFP_ZERO		BIT(___GFP_ZERO_BIT)
-/* 0x200u unused */
+#define ___GFP_RESERVED_THP	BIT(___GFP_RESERVED_THP_BIT)
 #define ___GFP_DIRECT_RECLAIM	BIT(___GFP_DIRECT_RECLAIM_BIT)
 #define ___GFP_KSWAPD_RECLAIM	BIT(___GFP_KSWAPD_RECLAIM_BIT)
 #define ___GFP_WRITE		BIT(___GFP_WRITE_BIT)
diff --git a/tools/perf/builtin-kmem.c b/tools/perf/builtin-kmem.c
index e1b2f5bc1ba8d8..45732aaf1a525e 100644
--- a/tools/perf/builtin-kmem.c
+++ b/tools/perf/builtin-kmem.c
@@ -672,6 +672,7 @@ static const struct {
 	{ "__GFP_NORETRY",		"NR" },
 	{ "__GFP_COMP",			"C" },
 	{ "__GFP_ZERO",			"Z" },
+	{ "__GFP_RESERVED_THP",		"RTHP" },
 	{ "__GFP_NOMEMALLOC",		"NMA" },
 	{ "__GFP_MEMALLOC",		"MA" },
 	{ "__GFP_HARDWALL",		"HW" },
diff --git a/tools/testing/vma/include/dup.h b/tools/testing/vma/include/dup.h
index cf73bcd9bb9d52..022b7a56b6a9f3 100644
--- a/tools/testing/vma/include/dup.h
+++ b/tools/testing/vma/include/dup.h
@@ -160,6 +160,7 @@ enum {
 #endif
 	DECLARE_VMA_BIT(UFFD_MINOR, 41),
 	DECLARE_VMA_BIT(SEALED, 42),
+	DECLARE_VMA_BIT(RESERVED_THP, 43),
 	/* Flags that reuse flags above. */
 	DECLARE_VMA_BIT_ALIAS(PKEY_BIT0, HIGH_ARCH_0),
 	DECLARE_VMA_BIT_ALIAS(PKEY_BIT1, HIGH_ARCH_1),

3/2 2026/06/27 07:39 llm patch-evaluator
0m Model:
gemini-3.1-pro-preview

Args:
null

Results:
null

Tokens:
							input: 19828
							output: 174
							thoughts: 2751
						

Thoughts:
**Analysis of the "MADV_RESERVED_THP" Patch for Fuzzing**

Alright, let's break this down. We've got a new patch introducing `MADV_RESERVED_THP` and `VM_RESERVED_THP`, which is clearly a functional change to the core memory management. This screams "fuzz me!"

My initial instinct is to pinpoint the most interesting functions, specifically the ones that are directly touched by this new madvise flag. Modified or added functions are: `madvise_hugepage_policy`, `reserved_thp_madvise_aligned`, `reserved_thp_charge`, `reserved_thp_uncharge`, `reserved_thp_hpage_nr`, `reserved_thp_pageblocks`, `check_reserved_thp_alignment`, `vrm_calc_charge`, `unmap_source_vma`, `vms_gather_munmap_vmas`, `__split_vma`, `dup_mmap`, `tear_down_vmas`, `do_huge_pmd_anonymous_page`, `__do_huge_pmd_anonymous_page`, `vma_alloc_anon_folio_pmd`, and `hugepage_madvise`.

Looking at the configs, this patch definitely needs `TRANSPARENT_HUGEPAGE`. I see checks for `CONFIG_64BIT` within `madvise_hugepage_policy`, which makes sense. I see that this patch uses `HPAGE_PMD_SIZE`, which is part of the THP setup.

I want to trim down the list to be more targeted. I'll focus on: `madvise_hugepage_policy`, `reserved_thp_charge`, `reserved_thp_uncharge`, `reserved_thp_pageblocks`, `check_reserved_thp_alignment`, `hugepage_madvise`.

A quick check confirms a few functions' properties. `madvise_hugepage_policy` is static, `reserved_thp_charge`, `reserved_thp_uncharge`, `reserved_thp_pageblocks` are global.

Now, let's dig deeper. The `mm/reserved_thp.c` file is interesting. It has functions like `setup_reserved_thp_size`, `setup_reserved_thp_nr`, and `reserved_thp_init`. It seems to utilize `thp_reserved_size` and `thp_reserved_nr` boot parameters. This is a crucial observation. If `thp_reserved_nr` isn't provided, `reserved_thp_total` becomes zero.

A careful look at `reserved_thp_charge` reveals that when `reserved_thp_total` is zero, any allocation request will return `-ENOMEM`. This makes the `madvise(MADV_RESERVED_THP)` calls to fail. I wonder if there's a way to test it.

It looks like the `reserved_thp_total` can only be set at boot time via `thp_reserved_nr`, and there's no sysfs interface for changing it. If `reserved_thp_total == 0`, `VM_RESERVED_THP` will never get set, and all the modifications to `mmap.c`, `mremap.c`, `vma.c`, and `huge_memory.c` will be unreachable. This is critical.

Can we bypass this? It appears not. `madvise` requires the start and end of the region to be aligned with the `HPAGE_PMD_SIZE`, therefore, the number of hugepages requested will be at least one hugepage. And this request will fail if we haven't provided a boot parameter, and therefore have not set `reserved_thp_total`.

Even though we need that boot parameter, it's still worth fuzzing. The patch is still doing functional changes. Syzkaller might have the ability to set boot parameters or it's just a limitation, but we should return WorthFuzzing=true.

So, here's my focused list of symbols:
- `madvise_hugepage_policy`
- `reserved_thp_charge`
- `reserved_thp_uncharge`
- `reserved_thp_pageblocks`
- `check_reserved_thp_alignment`
- `vma_alloc_anon_folio_pmd`

And the key config requirement is:
- `TRANSPARENT_HUGEPAGE`

In short: the patch is introducing a new `MADV_RESERVED_THP


4/2 2026/06/27 07:40 tool set-results
0m Args:
{
  "EnableConfigs": [
    "TRANSPARENT_HUGEPAGE"
  ],
  "FocusSymbols": [
    "madvise_hugepage_policy",
    "reserved_thp_charge",
    "reserved_thp_uncharge",
    "reserved_thp_pageblocks",
    "check_reserved_thp_alignment",
    "vma_alloc_anon_folio_pmd"
  ],
  "Reasoning": "The patch introduces a new MADV_RESERVED_THP madvise behavior and VM_RESERVED_THP VMA flag, modifying core memory management logic including page fault handling, mremap, and VMA splitting. This is a functional change to the kernel's memory management subsystem that should be fuzzed.",
  "WorthFuzzing": true
}

Results:
{
  "EnableConfigs": [
    "TRANSPARENT_HUGEPAGE"
  ],
  "FocusSymbols": [
    "madvise_hugepage_policy",
    "reserved_thp_charge",
    "reserved_thp_uncharge",
    "reserved_thp_pageblocks",
    "check_reserved_thp_alignment",
    "vma_alloc_anon_folio_pmd"
  ],
  "Reasoning": "The patch introduces a new MADV_RESERVED_THP madvise behavior and VM_RESERVED_THP VMA flag, modifying core memory management logic including page fault handling, mremap, and VMA splitting. This is a functional change to the kernel's memory management subsystem that should be fuzzed.",
  "WorthFuzzing": true
}

LLM Calls Summary:
Total Calls Total Tokens Avg Tokens Total Duration (Seconds) Avg Duration (Seconds)
Tool Calls Summary:
Total Calls Total Duration (Seconds) Avg Duration (Seconds)