The locked_vm field in struct user_struct was originally introduced by commit 789f90fcf6b0 ("perf_counter: per user mlock gift") to provide an mlock-like budget for perf measured against mm->pinned_vm (since 2011). perf grants each user a 'gift' of perf_event_mlock_kb * online CPUs of buffer pages which is charged to user->locked_vm without any RLIMIT_MEMLOCK check. Only pages beyond the gift are checked against RLIMIT_MEMLOCK, per-mm, via mm->pinned_vm, which makes it different from the standard mlock() check, which is per-mm and made against RLIMIT_MEMLOCK. However, since this was introduced, a number of other components have utilised this field where a shared resource needed to be limited against RLIMIT_MEMLOCK. The other users are currently MSG_ZEROCOPY, io_uring, AF_XDP, iommufd, s390 KVM zPCI and most recently, secretmem in commit 97d34aa65c29 ("mm/secretmem: properly account locked pages"). This field is shared between all of these, but because the gift is not checked against RLIMIT_MEMLOCK, user->locked_vm alone can exceed it - 16 CPUs * 516 KiB is already more than the default 8 MiB - after which every other user of the field fails unconditionally. This was reported by a user who saw secretmem failing while running a simultaneous perf record. Resolve the issue by simply giving perf its own field separate from the rest. BPF used user->locked_vm up until commit 80ee81e0403c ("bpf: Eliminate rlimit-based memory accounting infra for bpf maps") which landed in 5.11. Thus also eliminate the now-defunct ifdef for CONFIG_BPF_SYSCALL for user->locked_vm. This issue exists between all of the components which use user->locked_vm and perf, however secretmem is the first user who has caused a report over many years and has been backported, so target the fix at that. Fixes: 97d34aa65c29 ("mm/secretmem: properly account locked pages") Reported-by: Ameer Hamza Closes: https://lore.kernel.org/20261002175651.811343-1-ameer.hamza@truenas.com Cc: stable@vger.kernel.org Signed-off-by: Lorenzo Stoakes (ARM) --- include/linux/sched/user.h | 8 +++++--- kernel/events/core.c | 12 ++++++------ 2 files changed, 11 insertions(+), 9 deletions(-) diff --git a/include/linux/sched/user.h b/include/linux/sched/user.h index 8d7e5521f7cd..ec262b605437 100644 --- a/include/linux/sched/user.h +++ b/include/linux/sched/user.h @@ -23,11 +23,13 @@ struct user_struct { struct hlist_node uidhash_node; kuid_t uid; -#if defined(CONFIG_PERF_EVENTS) || defined(CONFIG_BPF_SYSCALL) || \ - defined(CONFIG_NET) || defined(CONFIG_IO_URING) || \ +#if defined(CONFIG_NET) || defined(CONFIG_IO_URING) || \ defined(CONFIG_VFIO_PCI_ZDEV_KVM) || IS_ENABLED(CONFIG_IOMMUFD) || \ defined(CONFIG_SECRETMEM) - atomic_long_t locked_vm; + atomic_long_t locked_vm; /* pinned pages charged to RLIMIT_MEMLOCK */ +#endif +#ifdef CONFIG_PERF_EVENTS + atomic_long_t perf_mlock; /* perf buffer pages within perf_event_mlock_kb */ #endif #ifdef CONFIG_WATCH_QUEUE atomic_t nr_watches; /* The number of watches this user currently has */ diff --git a/kernel/events/core.c b/kernel/events/core.c index 634d2ccbab82..a7da42f13d14 100644 --- a/kernel/events/core.c +++ b/kernel/events/core.c @@ -7059,7 +7059,7 @@ static void perf_mmap_close(struct vm_area_struct *vma) perf_pmu_output_stop(event); /* now it's safe to free the pages */ - atomic_long_sub(rb->aux_nr_pages - rb->aux_mmap_locked, &mmap_user->locked_vm); + atomic_long_sub(rb->aux_nr_pages - rb->aux_mmap_locked, &mmap_user->perf_mlock); atomic64_sub(rb->aux_mmap_locked, &vma->vm_mm->pinned_vm); /* this has to be the last one */ @@ -7242,11 +7242,11 @@ static bool perf_mmap_calc_limits(struct vm_area_struct *vma, long *user_extra, /* Increase the limit linearly with more CPUs */ user_lock_limit *= num_online_cpus(); - user_locked = atomic_long_read(&user->locked_vm); + user_locked = atomic_long_read(&user->perf_mlock); /* * sysctl_perf_event_mlock may have changed, so that - * user->locked_vm > user_lock_limit + * user->perf_mlock > user_lock_limit */ if (user_locked > user_lock_limit) user_locked = user_lock_limit; @@ -7254,7 +7254,7 @@ static bool perf_mmap_calc_limits(struct vm_area_struct *vma, long *user_extra, if (user_locked > user_lock_limit) { /* - * charge locked_vm until it hits user_lock_limit; + * charge perf_mlock until it hits user_lock_limit; * charge the rest from pinned_vm */ *extra = user_locked - user_lock_limit; @@ -7272,7 +7272,7 @@ static void perf_mmap_account(struct vm_area_struct *vma, long user_extra, long { struct user_struct *user = current_user(); - atomic_long_add(user_extra, &user->locked_vm); + atomic_long_add(user_extra, &user->perf_mlock); atomic64_add(extra, &vma->vm_mm->pinned_vm); } @@ -7281,7 +7281,7 @@ static void perf_mmap_unaccount(struct vm_area_struct *vma, struct perf_buffer * struct user_struct *user = rb->mmap_user; atomic_long_sub((perf_data_size(rb) >> PAGE_SHIFT) + 1 - rb->mmap_locked, - &user->locked_vm); + &user->perf_mlock); atomic64_sub(rb->mmap_locked, &vma->vm_mm->pinned_vm); } --- base-commit: e767a4ea70a3992c37ed604157d32f0dfbf9b1e3 change-id: 20261003-perf-locked-vm-fix-8ea78402cbbd Cheers, -- Lorenzo Stoakes (ARM)