Now that x86 processes have a general mm-local region, the LDT-specific management of the higher-level pagetables can mostly be replaced by just using the generic mm-local API. Drop all management of pagetable allocation and freeing; that is now handled automatically by virtue of the pagetables being in the mm-local region. Drop explicit logic to map LDTs into the user pagetables under PTI; that also happens automatically for this region. Unify the sanity-checking logic between x86_64 and PAE: use the generic set_memory.c mechanism to walk pagetables. This means the sanity-checking is slightly more relaxed, since lookup_address_in_pgd() is more flexible than pgd_to_pmd_walk(), but this seems to be worth it for the simplified code. It means that ldt.c doesn't have to know about the exact structure of the mm-local region's pagetables. Signed-off-by: Brendan Jackman --- Documentation/arch/x86/x86_64/mm.rst | 4 +- arch/x86/Kconfig | 4 +- arch/x86/include/asm/mmu_context.h | 2 - arch/x86/kernel/ldt.c | 124 +++++------------------------------ 4 files changed, 20 insertions(+), 114 deletions(-) diff --git a/Documentation/arch/x86/x86_64/mm.rst b/Documentation/arch/x86/x86_64/mm.rst index a6cf05d51bd8c..fa2bb7bab6a42 100644 --- a/Documentation/arch/x86/x86_64/mm.rst +++ b/Documentation/arch/x86/x86_64/mm.rst @@ -53,7 +53,7 @@ Complete virtual memory map with 4-level page tables ____________________________________________________________|___________________________________________________________ | | | | ffff800000000000 | -128 TB | ffff87ffffffffff | 8 TB | ... guard hole, also reserved for hypervisor - ffff880000000000 | -120 TB | ffff887fffffffff | 0.5 TB | LDT remap for PTI + ffff880000000000 | -120 TB | ffff887fffffffff | 0.5 TB | MM-local kernel data. Includes LDT remap for PTI ffff888000000000 | -119.5 TB | ffffc87fffffffff | 64 TB | direct mapping of all physical memory (page_offset_base) ffffc88000000000 | -55.5 TB | ffffc8ffffffffff | 0.5 TB | ... unused hole ffffc90000000000 | -55 TB | ffffe8ffffffffff | 32 TB | vmalloc/ioremap space (vmalloc_base) @@ -123,7 +123,7 @@ Complete virtual memory map with 5-level page tables ____________________________________________________________|___________________________________________________________ | | | | ff00000000000000 | -64 PB | ff0fffffffffffff | 4 PB | ... guard hole, also reserved for hypervisor - ff10000000000000 | -60 PB | ff10ffffffffffff | 0.25 PB | LDT remap for PTI + ff10000000000000 | -60 PB | ff10ffffffffffff | 0.25 PB | MM-local kernel data. Includes LDT remap for PTI ff11000000000000 | -59.75 PB | ff90ffffffffffff | 32 PB | direct mapping of all physical memory (page_offset_base) ff91000000000000 | -27.75 PB | ff9fffffffffffff | 3.75 PB | ... unused hole ffa0000000000000 | -24 PB | ffd1ffffffffffff | 12.5 PB | vmalloc/ioremap space (vmalloc_base) diff --git a/arch/x86/Kconfig b/arch/x86/Kconfig index 3efab3524a6cf..33c1282bfbf93 100644 --- a/arch/x86/Kconfig +++ b/arch/x86/Kconfig @@ -132,8 +132,7 @@ config X86 select ARCH_SUPPORTS_LTO_CLANG select ARCH_SUPPORTS_LTO_CLANG_THIN select ARCH_SUPPORTS_RT - # LDT remap temporarily clashes with mm-local region, can't have both. - select ARCH_SUPPORTS_MM_LOCAL_REGION if X86_64 || X86_PAE && !MODIFY_LDT_SYSCALL + select ARCH_SUPPORTS_MM_LOCAL_REGION if X86_64 || X86_PAE select ARCH_USE_BUILTIN_BSWAP select ARCH_USE_CMPXCHG_LOCKREF select ARCH_USE_MEMTEST @@ -2280,6 +2279,7 @@ config CMDLINE_OVERRIDE config MODIFY_LDT_SYSCALL bool "Enable the LDT (local descriptor table)" if EXPERT default y + select MM_LOCAL_REGION if MITIGATION_PAGE_TABLE_ISOLATION || X86_PAE help Linux can allow user programs to install a per-process x86 Local Descriptor Table (LDT) using the modify_ldt(2) system diff --git a/arch/x86/include/asm/mmu_context.h b/arch/x86/include/asm/mmu_context.h index 3d4f54673014f..09b8d8e6a56ea 100644 --- a/arch/x86/include/asm/mmu_context.h +++ b/arch/x86/include/asm/mmu_context.h @@ -61,7 +61,6 @@ static inline void init_new_context_ldt(struct mm_struct *mm) } int ldt_dup_context(struct mm_struct *oldmm, struct mm_struct *mm); void destroy_context_ldt(struct mm_struct *mm); -void ldt_arch_exit_mmap(struct mm_struct *mm); #else /* CONFIG_MODIFY_LDT_SYSCALL */ static inline void init_new_context_ldt(struct mm_struct *mm) { } static inline int ldt_dup_context(struct mm_struct *oldmm, @@ -70,7 +69,6 @@ static inline int ldt_dup_context(struct mm_struct *oldmm, return 0; } static inline void destroy_context_ldt(struct mm_struct *mm) { } -static inline void ldt_arch_exit_mmap(struct mm_struct *mm) { } #endif #ifdef CONFIG_MODIFY_LDT_SYSCALL diff --git a/arch/x86/kernel/ldt.c b/arch/x86/kernel/ldt.c index 40c5bf97dd5cc..685664c1ee770 100644 --- a/arch/x86/kernel/ldt.c +++ b/arch/x86/kernel/ldt.c @@ -186,10 +186,16 @@ static struct ldt_struct *alloc_ldt_struct(unsigned int num_entries) #ifdef CONFIG_MITIGATION_PAGE_TABLE_ISOLATION -static void do_sanity_check(struct mm_struct *mm, - bool had_kernel_mapping, - bool had_user_mapping) +static void sanity_check_ldt_mapping(struct mm_struct *mm) { + pgd_t *k_pgd = pgd_offset(mm, LDT_BASE_ADDR); + pgd_t *u_pgd = kernel_to_user_pgdp(k_pgd); + unsigned int k_level, u_level; + bool had_kernel_mapping, had_user_mapping; + + had_kernel_mapping = lookup_address_in_pgd(k_pgd, LDT_BASE_ADDR, &k_level); + had_user_mapping = lookup_address_in_pgd(u_pgd, LDT_BASE_ADDR, &u_level); + if (mm->context.ldt) { /* * We already had an LDT. The top-level entry should already @@ -210,76 +216,6 @@ static void do_sanity_check(struct mm_struct *mm, } } -#ifdef CONFIG_X86_PAE - -static pmd_t *pgd_to_pmd_walk(pgd_t *pgd, unsigned long va) -{ - p4d_t *p4d; - pud_t *pud; - - if (pgd->pgd == 0) - return NULL; - - p4d = p4d_offset(pgd, va); - if (p4d_none(*p4d)) - return NULL; - - pud = pud_offset(p4d, va); - if (pud_none(*pud)) - return NULL; - - return pmd_offset(pud, va); -} - -static void map_ldt_struct_to_user(struct mm_struct *mm) -{ - pgd_t *k_pgd = pgd_offset(mm, LDT_BASE_ADDR); - pgd_t *u_pgd = kernel_to_user_pgdp(k_pgd); - pmd_t *k_pmd, *u_pmd; - - k_pmd = pgd_to_pmd_walk(k_pgd, LDT_BASE_ADDR); - u_pmd = pgd_to_pmd_walk(u_pgd, LDT_BASE_ADDR); - - if (boot_cpu_has(X86_FEATURE_PTI) && !mm->context.ldt) - set_pmd(u_pmd, *k_pmd); -} - -static void sanity_check_ldt_mapping(struct mm_struct *mm) -{ - pgd_t *k_pgd = pgd_offset(mm, LDT_BASE_ADDR); - pgd_t *u_pgd = kernel_to_user_pgdp(k_pgd); - bool had_kernel, had_user; - pmd_t *k_pmd, *u_pmd; - - k_pmd = pgd_to_pmd_walk(k_pgd, LDT_BASE_ADDR); - u_pmd = pgd_to_pmd_walk(u_pgd, LDT_BASE_ADDR); - had_kernel = (k_pmd->pmd != 0); - had_user = (u_pmd->pmd != 0); - - do_sanity_check(mm, had_kernel, had_user); -} - -#else /* !CONFIG_X86_PAE */ - -static void map_ldt_struct_to_user(struct mm_struct *mm) -{ - pgd_t *pgd = pgd_offset(mm, LDT_BASE_ADDR); - - if (boot_cpu_has(X86_FEATURE_PTI) && !mm->context.ldt) - set_pgd(kernel_to_user_pgdp(pgd), *pgd); -} - -static void sanity_check_ldt_mapping(struct mm_struct *mm) -{ - pgd_t *pgd = pgd_offset(mm, LDT_BASE_ADDR); - bool had_kernel = (pgd->pgd != 0); - bool had_user = (kernel_to_user_pgdp(pgd)->pgd != 0); - - do_sanity_check(mm, had_kernel, had_user); -} - -#endif /* CONFIG_X86_PAE */ - /* * If PTI is enabled, this maps the LDT into the kernelmode and * usermode tables for the given mm. @@ -290,7 +226,7 @@ map_ldt_struct(struct mm_struct *mm, struct ldt_struct *ldt, int slot) unsigned long va; bool is_vmalloc; spinlock_t *ptl; - int i, nr_pages; + int i, nr_pages, err; if (!boot_cpu_has(X86_FEATURE_PTI)) return 0; @@ -304,6 +240,10 @@ map_ldt_struct(struct mm_struct *mm, struct ldt_struct *ldt, int slot) /* Check if the current mappings are sane */ sanity_check_ldt_mapping(mm); + err = mm_local_region_init(mm); + if (err) + return err; + is_vmalloc = is_vmalloc_addr(ldt->entries); nr_pages = DIV_ROUND_UP(ldt->nr_entries * LDT_ENTRY_SIZE, PAGE_SIZE); @@ -339,9 +279,6 @@ map_ldt_struct(struct mm_struct *mm, struct ldt_struct *ldt, int slot) pte_unmap_unlock(ptep, ptl); } - /* Propagate LDT mapping to the user page-table */ - map_ldt_struct_to_user(mm); - ldt->slot = slot; return 0; } @@ -390,28 +327,6 @@ static void unmap_ldt_struct(struct mm_struct *mm, struct ldt_struct *ldt) } #endif /* CONFIG_MITIGATION_PAGE_TABLE_ISOLATION */ -static void free_ldt_pgtables(struct mm_struct *mm) -{ -#ifdef CONFIG_MITIGATION_PAGE_TABLE_ISOLATION - struct mmu_gather tlb; - unsigned long start = LDT_BASE_ADDR; - unsigned long end = LDT_END_ADDR; - - if (!boot_cpu_has(X86_FEATURE_PTI)) - return; - - /* - * Although free_pgd_range() is intended for freeing user - * page-tables, it also works out for kernel mappings on x86. - * We use tlb_gather_mmu_fullmm() to avoid confusing the - * range-tracking logic in __tlb_adjust_range(). - */ - tlb_gather_mmu_fullmm(&tlb, mm); - free_pgd_range(&tlb, start, end, start, end); - tlb_finish_mmu(&tlb); -#endif -} - /* After calling this, the LDT is immutable. */ static void finalize_ldt_struct(struct ldt_struct *ldt) { @@ -472,7 +387,6 @@ int ldt_dup_context(struct mm_struct *old_mm, struct mm_struct *mm) retval = map_ldt_struct(mm, new_ldt, 0); if (retval) { - free_ldt_pgtables(mm); free_ldt_struct(new_ldt); goto out_unlock; } @@ -494,11 +408,6 @@ void destroy_context_ldt(struct mm_struct *mm) mm->context.ldt = NULL; } -void ldt_arch_exit_mmap(struct mm_struct *mm) -{ - free_ldt_pgtables(mm); -} - static int read_ldt(void __user *ptr, unsigned long bytecount) { struct mm_struct *mm = current->mm; @@ -645,10 +554,9 @@ static int write_ldt(void __user *ptr, unsigned long bytecount, int oldmode) /* * This only can fail for the first LDT setup. If an LDT is * already installed then the PTE page is already - * populated. Mop up a half populated page table. + * populated. */ - if (!WARN_ON_ONCE(old_ldt)) - free_ldt_pgtables(mm); + WARN_ON_ONCE(old_ldt); free_ldt_struct(new_ldt); goto out_unlock; } -- 2.54.0