From b0c4582862c2feae7c9d49fe77aff6215cd1cea9 Mon Sep 17 00:00:00 2001 From: "Harry Yoo (Oracle)" Date: Wed, 29 Jul 2026 17:20:09 +0900 Subject: [PATCH 1/9] mm/slab, slub_kunit: register kprobe to trigger _nolock APIs Since kmalloc_nolock() always fails in NMI and hardirq contexts on PREEMPT_RT, slub_kunit cannot properly test _nolock() APIs. Register a kprobe pre-handler to invoke kmalloc_nolock() and kfree_nolock() in the middle of the slab allocator. However, do not register the handler on UP kernels because that use case is not well supported [1] in the kernel. To attach the pre-handler while s->cpu_sheaves->lock or n->list_lock is held, add a wrapper function for lockdep_assert_held() that calls a no-op function slab_attach_kprobe_locked() on debug builds. The function is optimized away when neither CONFIG_PROVE_LOCKING nor CONFIG_DEBUG_VM is selected and register_kprobe() fails. The function calls barrier() to prevent the compiler from optimizing away its callsites. Otherwise, the compiler may consider the function does not have any side effect and remove callsites. Compared to using plain kprobe, this has two advantages: 1) it avoids hardcoding function names in the test, and 2) it can trigger those APIs in the middle of a function, where the lock is expected to be held as annotated with lockdep. While it was proposed [2] to use kunit function redirection to test this, it is currently infeasible as some lock helpers don't have symbols. Factor out the nested loop that calls kmalloc and friends to test_kmalloc_kfree(), and call them in test_kmalloc_kfree_nolock_{perf,kprobe}(), each being an independent test case. During the refactoring, drop alloc_fail handling as it doesn't provide much benefits. Link: https://lore.kernel.org/linux-mm/20260427-nolock-api-fix-v2-0-a6b83a92d9a4@kernel.org [1] Link: https://lore.kernel.org/linux-mm/6edebc2b-5f5a-4b9c-9a4c-564310acee1b@kernel.org [2] Acked-by: Vlastimil Babka (SUSE) Reviewed-by: Shengming Hu Signed-off-by: Harry Yoo (Oracle) Link: https://patch.msgid.link/20260729-kfree_rcu_nolock-v5-1-a28cdcda9673@kernel.org Signed-off-by: Vlastimil Babka (SUSE) --- lib/tests/slub_kunit.c | 167 +++++++++++++++++++++++++++++------------ mm/slub.c | 36 ++++++--- 2 files changed, 148 insertions(+), 55 deletions(-) diff --git a/lib/tests/slub_kunit.c b/lib/tests/slub_kunit.c index fa6d31dbca16..8c2b9911471e 100644 --- a/lib/tests/slub_kunit.c +++ b/lib/tests/slub_kunit.c @@ -8,6 +8,7 @@ #include #include #include +#include #include "../mm/slab.h" static struct kunit_resource resource; @@ -292,7 +293,7 @@ static void test_krealloc_redzone_zeroing(struct kunit *test) kmem_cache_destroy(s); } -#ifdef CONFIG_PERF_EVENTS +#if defined(CONFIG_PERF_EVENTS) || (defined(CONFIG_KPROBES) && defined(CONFIG_SMP)) #define NR_ITERATIONS 1000 #define NR_OBJECTS 1000 static void *objects[NR_OBJECTS]; @@ -302,26 +303,40 @@ struct test_nolock_context { int callback_count; int alloc_ok; int alloc_fail; +#ifdef CONFIG_PERF_EVENTS struct perf_event *event; +#endif +#if defined(CONFIG_KPROBES) && defined(CONFIG_SMP) + struct kprobe kprobe; +#endif }; -static struct perf_event_attr hw_attr = { - .type = PERF_TYPE_HARDWARE, - .config = PERF_COUNT_HW_CPU_CYCLES, - .size = sizeof(struct perf_event_attr), - .pinned = 1, - .disabled = 1, - .freq = 1, - .sample_freq = 100000, -}; +static void test_kmalloc_kfree(void) +{ + int i, j; -static void overflow_handler_test_kmalloc_kfree_nolock(struct perf_event *event, - struct perf_sample_data *data, - struct pt_regs *regs) + for (i = 0; i < NR_ITERATIONS; i++) { + for (j = 0; j < NR_OBJECTS; j++) { + gfp_t gfp = (i % 2) ? GFP_KERNEL : GFP_KERNEL_ACCOUNT; + + objects[j] = kmalloc(64, gfp); + if (!objects[j]) { + j--; + while (j >= 0) + kfree(objects[j--]); + return; + } + } + + for (j = 0; j < NR_OBJECTS; j++) + kfree(objects[j]); + } +} + +static void test_nolock(struct test_nolock_context *ctx) { void *objp; gfp_t gfp; - struct test_nolock_context *ctx = event->overflow_handler_context; /* __GFP_ACCOUNT to test kmalloc_nolock() in alloc_slab_obj_exts() */ gfp = (ctx->callback_count % 2) ? 0 : __GFP_ACCOUNT; @@ -335,47 +350,104 @@ static void overflow_handler_test_kmalloc_kfree_nolock(struct perf_event *event, kfree_nolock(objp); ctx->callback_count++; } +#endif -static void test_kmalloc_kfree_nolock(struct kunit *test) +#ifdef CONFIG_PERF_EVENTS +static struct perf_event_attr hw_attr = { + .type = PERF_TYPE_HARDWARE, + .config = PERF_COUNT_HW_CPU_CYCLES, + .size = sizeof(struct perf_event_attr), + .pinned = 1, + .disabled = 1, + .freq = 1, + .sample_freq = 100000, +}; + +static void overflow_handler_test_nolock(struct perf_event *event, + struct perf_sample_data *data, + struct pt_regs *regs) +{ + struct test_nolock_context *ctx = event->overflow_handler_context; + + test_nolock(ctx); +} + +static bool enable_perf_events(struct test_nolock_context *ctx) { - int i, j; - struct test_nolock_context ctx = { .test = test }; struct perf_event *event; - bool alloc_fail = false; event = perf_event_create_kernel_counter(&hw_attr, -1, current, - overflow_handler_test_kmalloc_kfree_nolock, - &ctx); + overflow_handler_test_nolock, + ctx); + if (IS_ERR(event)) - kunit_skip(test, "Failed to create perf event"); - ctx.event = event; - perf_event_enable(ctx.event); - for (i = 0; i < NR_ITERATIONS; i++) { - for (j = 0; j < NR_OBJECTS; j++) { - gfp_t gfp = (i % 2) ? GFP_KERNEL : GFP_KERNEL_ACCOUNT; + return false; - objects[j] = kmalloc(64, gfp); - if (!objects[j]) { - j--; - while (j >= 0) - kfree(objects[j--]); - alloc_fail = true; - goto cleanup; - } - } - for (j = 0; j < NR_OBJECTS; j++) - kfree(objects[j]); - } + ctx->event = event; + perf_event_enable(ctx->event); + return true; +} -cleanup: - perf_event_disable(ctx.event); - perf_event_release_kernel(ctx.event); +static void disable_perf_events(struct test_nolock_context *ctx) +{ + kunit_info(ctx->test, "HW perf events: callback_count: %d, alloc_ok: %d, alloc_fail: %d\n", + ctx->callback_count, ctx->alloc_ok, ctx->alloc_fail); - kunit_info(test, "callback_count: %d, alloc_ok: %d, alloc_fail: %d\n", - ctx.callback_count, ctx.alloc_ok, ctx.alloc_fail); + perf_event_disable(ctx->event); + perf_event_release_kernel(ctx->event); +} - if (alloc_fail) - kunit_skip(test, "Allocation failed"); +static void test_kmalloc_kfree_nolock_perf(struct kunit *test) +{ + struct test_nolock_context ctx = { .test = test }; + + if (!enable_perf_events(&ctx)) + kunit_skip(test, "Failed to enable perf event, skipping"); + + test_kmalloc_kfree(); + + disable_perf_events(&ctx); + KUNIT_EXPECT_EQ(test, 0, slab_errors); +} +#endif + +#if defined(CONFIG_KPROBES) && defined(CONFIG_SMP) +static int slab_kprobe_pre_handler(struct kprobe *p, struct pt_regs *regs) +{ + struct test_nolock_context *ctx; + + ctx = container_of(p, struct test_nolock_context, kprobe); + test_nolock(ctx); + return 0; +} + +static bool register_slab_kprobes(struct test_nolock_context *ctx) +{ + ctx->kprobe.symbol_name = "slab_attach_kprobe_locked"; + ctx->kprobe.pre_handler = slab_kprobe_pre_handler; + + if (register_kprobe(&ctx->kprobe)) + return false; + return true; +} + +static void unregister_slab_kprobes(struct test_nolock_context *ctx) +{ + kunit_info(ctx->test, "kprobes: callback_count: %d, alloc_ok: %d, alloc_fail: %d\n", + ctx->callback_count, ctx->alloc_ok, ctx->alloc_fail); + unregister_kprobe(&ctx->kprobe); +} + +static void test_kmalloc_kfree_nolock_kprobe(struct kunit *test) +{ + struct test_nolock_context ctx = { .test = test }; + + if (!register_slab_kprobes(&ctx)) + kunit_skip(test, "Failed to register kprobe, skipping"); + + test_kmalloc_kfree(); + + unregister_slab_kprobes(&ctx); KUNIT_EXPECT_EQ(test, 0, slab_errors); } #endif @@ -405,7 +477,10 @@ static struct kunit_case test_cases[] = { KUNIT_CASE(test_leak_destroy), KUNIT_CASE(test_krealloc_redzone_zeroing), #ifdef CONFIG_PERF_EVENTS - KUNIT_CASE_SLOW(test_kmalloc_kfree_nolock), + KUNIT_CASE_SLOW(test_kmalloc_kfree_nolock_perf), +#endif +#if defined(CONFIG_KPROBES) && defined(CONFIG_SMP) + KUNIT_CASE_SLOW(test_kmalloc_kfree_nolock_kprobe), #endif {} }; diff --git a/mm/slub.c b/mm/slub.c index 0337e60db5ac..8273780fb4ae 100644 --- a/mm/slub.c +++ b/mm/slub.c @@ -908,6 +908,24 @@ static inline unsigned int obj_exts_offset_in_object(struct kmem_cache *s) } #endif +/* + * A no-op function used to attach kprobe handlers in slub_kunit tests. + * The barrier is needed to prevent the compiler from optimizing out callsites. + */ +#if defined(CONFIG_DEBUG_VM) || defined(CONFIG_PROVE_LOCKING) +static noinline void slab_attach_kprobe_locked(void) +{ + barrier(); +} +#else +static inline void slab_attach_kprobe_locked(void) { } +#endif + +#define slab_lockdep_assert_held(lock) do { \ + lockdep_assert_held(lock); \ + slab_attach_kprobe_locked(); \ +} while (0) + #ifdef CONFIG_SLUB_DEBUG /* @@ -1665,7 +1683,7 @@ static void add_full(struct kmem_cache *s, if (!(s->flags & SLAB_STORE_USER)) return; - lockdep_assert_held(&n->list_lock); + slab_lockdep_assert_held(&n->list_lock); list_add(&slab->slab_list, &n->full); } @@ -1674,7 +1692,7 @@ static void remove_full(struct kmem_cache *s, struct kmem_cache_node *n, struct if (!(s->flags & SLAB_STORE_USER)) return; - lockdep_assert_held(&n->list_lock); + slab_lockdep_assert_held(&n->list_lock); list_del(&slab->slab_list); } @@ -2840,7 +2858,7 @@ static unsigned int __sheaf_flush_main_batch(struct kmem_cache *s) void *objects[PCS_BATCH_MAX]; struct slab_sheaf *sheaf; - lockdep_assert_held(this_cpu_ptr(&s->cpu_sheaves->lock)); + slab_lockdep_assert_held(this_cpu_ptr(&s->cpu_sheaves->lock)); pcs = this_cpu_ptr(s->cpu_sheaves); sheaf = pcs->main; @@ -3519,7 +3537,7 @@ __add_partial(struct kmem_cache_node *n, struct slab *slab, enum add_mode mode) static inline void add_partial(struct kmem_cache_node *n, struct slab *slab, enum add_mode mode) { - lockdep_assert_held(&n->list_lock); + slab_lockdep_assert_held(&n->list_lock); __add_partial(n, slab, mode); } @@ -3533,7 +3551,7 @@ static inline void clear_node_partial_state(struct kmem_cache_node *n, static inline void remove_partial(struct kmem_cache_node *n, struct slab *slab) { - lockdep_assert_held(&n->list_lock); + slab_lockdep_assert_held(&n->list_lock); list_del(&slab->slab_list); clear_node_partial_state(n, slab); } @@ -3549,7 +3567,7 @@ static void *alloc_single_from_partial(struct kmem_cache *s, { void *object; - lockdep_assert_held(&n->list_lock); + slab_lockdep_assert_held(&n->list_lock); #ifdef CONFIG_SLUB_DEBUG if (s->flags & SLAB_CONSISTENCY_CHECKS) { @@ -4620,7 +4638,7 @@ __pcs_replace_empty_main(struct kmem_cache *s, struct slub_percpu_sheaves *pcs, struct node_barn *barn; bool allow_spin; - lockdep_assert_held(this_cpu_ptr(&s->cpu_sheaves->lock)); + slab_lockdep_assert_held(this_cpu_ptr(&s->cpu_sheaves->lock)); /* Bootstrap or debug cache, back off */ if (unlikely(!cache_has_sheaves(s))) { @@ -5763,7 +5781,7 @@ static void __pcs_install_empty_sheaf(struct kmem_cache *s, struct slub_percpu_sheaves *pcs, struct slab_sheaf *empty, struct node_barn *barn) { - lockdep_assert_held(this_cpu_ptr(&s->cpu_sheaves->lock)); + slab_lockdep_assert_held(this_cpu_ptr(&s->cpu_sheaves->lock)); /* This is what we expect to find if nobody interrupted us. */ if (likely(!pcs->spare)) { @@ -5814,7 +5832,7 @@ __pcs_replace_full_main(struct kmem_cache *s, struct slub_percpu_sheaves *pcs, bool put_fail; restart: - lockdep_assert_held(this_cpu_ptr(&s->cpu_sheaves->lock)); + slab_lockdep_assert_held(this_cpu_ptr(&s->cpu_sheaves->lock)); /* Bootstrap or debug cache, back off */ if (unlikely(!cache_has_sheaves(s))) { From f3521fbec2b2839698eba2b7013ed20119416e6d Mon Sep 17 00:00:00 2001 From: "Harry Yoo (Oracle)" Date: Wed, 29 Jul 2026 17:20:10 +0900 Subject: [PATCH 2/9] mm/slab: handle the !allow_spin case in kfree_rcu_sheaf() Teach kfree_rcu_sheaf() how to handle the !allow_spin case. Try to get an empty sheaf from pcs->spare or the barn even when spinning is not allowed. Unlike __pcs_replace_full_main(), try harder to allocate an empty sheaf because the fallback path will be more expensive than kfree_nolock(). Now that slab has internal alloc_flags to describe context, introduce free_flags analogously and convert free_flags to alloc_flags when allocating memory in the free path. When trylock fails or the kernel observes non-NULL pcs->rcu_free after lock acquisition, free the sheaf instead of putting it to the barn. This is rare and not worth complicating the code. Since call_rcu() cannot be called in an unknown context, kfree_rcu_sheaf() fails when the rcu sheaf becomes full. Link: https://lore.kernel.org/linux-mm/872bd673-3d45-4111-8a41-31185db3ece5@kernel.org Reviewed-by: Vlastimil Babka (SUSE) Signed-off-by: Harry Yoo (Oracle) Link: https://patch.msgid.link/20260729-kfree_rcu_nolock-v5-2-a28cdcda9673@kernel.org Reviewed-by: Shengming Hu Signed-off-by: Vlastimil Babka (SUSE) --- mm/slab.h | 18 +++++++++++++++++- mm/slab_common.c | 2 +- mm/slub.c | 34 ++++++++++++++++++++++++++-------- 3 files changed, 44 insertions(+), 10 deletions(-) diff --git a/mm/slab.h b/mm/slab.h index f5e336b6b6b0..ba08da09da36 100644 --- a/mm/slab.h +++ b/mm/slab.h @@ -24,11 +24,27 @@ #define SLAB_ALLOC_NO_RECURSE 0x04 /* prevent kmalloc() recursion */ #define SLAB_ALLOC_NO_OBJ_EXT 0x08 /* prevent obj_exts array allocation */ +#define SLAB_FREE_DEFAULT 0x00 /* no flags */ +#define SLAB_FREE_NOLOCK 0x01 /* spinning not allowed */ + +static inline unsigned int to_alloc_flags(unsigned int free_flags) +{ + if (free_flags & SLAB_FREE_NOLOCK) + return SLAB_ALLOC_NOLOCK; + else + return SLAB_ALLOC_DEFAULT; +} + static inline bool alloc_flags_allow_spinning(const unsigned int alloc_flags) { return !(alloc_flags & SLAB_ALLOC_NOLOCK); } +static inline bool free_flags_allow_spinning(const unsigned int free_flags) +{ + return !(free_flags & SLAB_FREE_NOLOCK); +} + void *__kmalloc_flags_noprof(DECL_TOKEN_PARAMS(size, token), gfp_t flags, unsigned int alloc_flags, int node) __assume_kmalloc_alignment __alloc_size(1); @@ -436,7 +452,7 @@ static inline bool is_kmalloc_normal(struct kmem_cache *s) return !(s->flags & (SLAB_CACHE_DMA|SLAB_ACCOUNT|SLAB_RECLAIM_ACCOUNT|SLAB_NO_OBJ_EXT)); } -bool __kfree_rcu_sheaf(struct kmem_cache *s, void *obj); +bool __kfree_rcu_sheaf(struct kmem_cache *s, void *obj, unsigned int free_flags); void flush_all_rcu_sheaves(void); void flush_rcu_sheaves_on_cache(struct kmem_cache *s); diff --git a/mm/slab_common.c b/mm/slab_common.c index 03ecac12cd86..2475fc0e5e41 100644 --- a/mm/slab_common.c +++ b/mm/slab_common.c @@ -1622,7 +1622,7 @@ static bool kfree_rcu_sheaf(void *obj) s = slab->slab_cache; if (likely(!IS_ENABLED(CONFIG_NUMA) || slab_nid(slab) == numa_mem_id())) - return __kfree_rcu_sheaf(s, obj); + return __kfree_rcu_sheaf(s, obj, SLAB_FREE_DEFAULT); return false; } diff --git a/mm/slub.c b/mm/slub.c index 8273780fb4ae..325f3818cd6f 100644 --- a/mm/slub.c +++ b/mm/slub.c @@ -2789,7 +2789,8 @@ static inline struct slab_sheaf *alloc_empty_sheaf(struct kmem_cache *s, return __alloc_empty_sheaf(s, gfp, alloc_flags, s->sheaf_capacity); } -static void free_empty_sheaf(struct kmem_cache *s, struct slab_sheaf *sheaf) +static void __free_empty_sheaf(struct kmem_cache *s, struct slab_sheaf *sheaf, + unsigned int free_flags) { /* * If the sheaf was created with SLAB_ALLOC_NO_RECURSE flag then its @@ -2801,11 +2802,20 @@ static void free_empty_sheaf(struct kmem_cache *s, struct slab_sheaf *sheaf) mark_obj_codetag_empty(sheaf); VM_WARN_ON_ONCE(sheaf->size > 0); - kfree(sheaf); + + if (unlikely(free_flags & SLAB_FREE_NOLOCK)) + kfree_nolock(sheaf); + else + kfree(sheaf); stat(s, SHEAF_FREE); } +static void free_empty_sheaf(struct kmem_cache *s, struct slab_sheaf *sheaf) +{ + __free_empty_sheaf(s, sheaf, SLAB_FREE_DEFAULT); +} + static unsigned int refill_objects(struct kmem_cache *s, void **p, gfp_t gfp, unsigned int min, unsigned int max); @@ -6042,10 +6052,11 @@ static void rcu_free_sheaf(struct rcu_head *head) */ static DEFINE_WAIT_OVERRIDE_MAP(kfree_rcu_sheaf_map, LD_WAIT_CONFIG); -bool __kfree_rcu_sheaf(struct kmem_cache *s, void *obj) +bool __kfree_rcu_sheaf(struct kmem_cache *s, void *obj, unsigned int free_flags) { struct slub_percpu_sheaves *pcs; struct slab_sheaf *rcu_sheaf; + bool allow_spin = free_flags_allow_spinning(free_flags); if (WARN_ON_ONCE(IS_ENABLED(CONFIG_PREEMPT_RT))) return false; @@ -6058,9 +6069,10 @@ bool __kfree_rcu_sheaf(struct kmem_cache *s, void *obj) pcs = this_cpu_ptr(s->cpu_sheaves); if (unlikely(!pcs->rcu_free)) { - struct slab_sheaf *empty; struct node_barn *barn; + unsigned int alloc_flags = to_alloc_flags(free_flags); + gfp_t gfp = allow_spin ? GFP_NOWAIT : __GFP_NOWARN; /* Bootstrap or debug cache, fall back */ if (unlikely(!cache_has_sheaves(s))) { @@ -6080,7 +6092,7 @@ bool __kfree_rcu_sheaf(struct kmem_cache *s, void *obj) goto fail; } - empty = barn_get_empty_sheaf(barn, true); + empty = barn_get_empty_sheaf(barn, allow_spin); if (empty) { pcs->rcu_free = empty; @@ -6089,20 +6101,20 @@ bool __kfree_rcu_sheaf(struct kmem_cache *s, void *obj) local_unlock(&s->cpu_sheaves->lock); - empty = alloc_empty_sheaf(s, GFP_NOWAIT, SLAB_ALLOC_DEFAULT); + empty = alloc_empty_sheaf(s, gfp, alloc_flags); if (!empty) goto fail; if (!local_trylock(&s->cpu_sheaves->lock)) { - barn_put_empty_sheaf(barn, empty); + __free_empty_sheaf(s, empty, free_flags); goto fail; } pcs = this_cpu_ptr(s->cpu_sheaves); if (unlikely(pcs->rcu_free)) - barn_put_empty_sheaf(barn, empty); + __free_empty_sheaf(s, empty, free_flags); else pcs->rcu_free = empty; } @@ -6120,6 +6132,12 @@ bool __kfree_rcu_sheaf(struct kmem_cache *s, void *obj) if (likely(rcu_sheaf->size < s->sheaf_capacity)) { rcu_sheaf = NULL; } else { + if (unlikely(!allow_spin)) { + /* call_rcu() cannot be called in an unknown context */ + rcu_sheaf->size--; + local_unlock(&s->cpu_sheaves->lock); + goto fail; + } pcs->rcu_free = NULL; rcu_sheaf->node = numa_node_id(); } From f8e0c305995fa2931e77a6e669e8362458ae4eba Mon Sep 17 00:00:00 2001 From: "Harry Yoo (Oracle)" Date: Wed, 29 Jul 2026 17:20:11 +0900 Subject: [PATCH 3/9] mm/slab: use call_rcu() in unknown context if irqs are enabled call_rcu() disables IRQs with local_irq_save() to protect its per-cpu data structures. Therefore, if IRQs are not disabled, they cannot be corrupted by reentrance into call_rcu(). So fall back to the deferred path only when !allow_spin && irqs_disabled(). The RCU subsystem does not guarantee this contractually, and this optimization relies on RCU's implementation details. Ideally, it should be removed once call_rcu_nolock() is supported by the RCU subsystem. Link: https://lore.kernel.org/linux-mm/CAADnVQKRVD5ZSnEKbZZU7w86gHbGHUug2pvzpgZTngNS+fg4rw@mail.gmail.com Suggested-by: Alexei Starovoitov Signed-off-by: Harry Yoo (Oracle) Link: https://patch.msgid.link/20260729-kfree_rcu_nolock-v5-3-a28cdcda9673@kernel.org Reviewed-by: Shengming Hu Signed-off-by: Vlastimil Babka (SUSE) --- mm/slub.c | 8 ++++++-- 1 file changed, 6 insertions(+), 2 deletions(-) diff --git a/mm/slub.c b/mm/slub.c index 325f3818cd6f..31f94bdad3b0 100644 --- a/mm/slub.c +++ b/mm/slub.c @@ -6132,8 +6132,12 @@ bool __kfree_rcu_sheaf(struct kmem_cache *s, void *obj, unsigned int free_flags) if (likely(rcu_sheaf->size < s->sheaf_capacity)) { rcu_sheaf = NULL; } else { - if (unlikely(!allow_spin)) { - /* call_rcu() cannot be called in an unknown context */ + /* + * With !allow_spin, we might have interrupted call_rcu()'s + * IRQ-disabled critical section. If IRQs are not disabled, + * we know that's not the case. + */ + if (unlikely(!allow_spin && irqs_disabled())) { rcu_sheaf->size--; local_unlock(&s->cpu_sheaves->lock); goto fail; From 69abda97a09b2e7df26d21cb534f4930201b4186 Mon Sep 17 00:00:00 2001 From: "Harry Yoo (Oracle)" Date: Wed, 29 Jul 2026 17:20:12 +0900 Subject: [PATCH 4/9] mm/slab: extend deferred free mechanism to handle rcu sheaves __kfree_rcu_sheaf() cannot invoke call_rcu() when spinning is not allowed and IRQs are disabled. To relax the limitation, extend the deferred free fallback so that a full rcu sheaf can be submitted to call_rcu() via the existing IRQ work. Since the deferred mechanism does more than deferred freeing of objects, rename the struct to deferred_percpu_work and adjust names accordingly. When a sheaf is queued on an IRQ work, it is detached from pcs->rcu_free but call_rcu() is not invoked until the irq_work runs. To keep the kvfree_rcu barrier's promise, call irq_work_sync() on each CPU before calling rcu_barrier(). In the meantime, remove the TODO item as apparently there is no simple and effective way to achieve that. This is because, unlike sheaves, kfree_rcu() batches objects from different caches together. Suggested-by: Alexei Starovoitov Reviewed-by: Pedro Falcato Reviewed-by: Vlastimil Babka (SUSE) Signed-off-by: Harry Yoo (Oracle) Link: https://patch.msgid.link/20260729-kfree_rcu_nolock-v5-4-a28cdcda9673@kernel.org Signed-off-by: Vlastimil Babka (SUSE) --- mm/slab.h | 2 +- mm/slab_common.c | 7 ++-- mm/slub.c | 85 ++++++++++++++++++++++++++++-------------------- 3 files changed, 53 insertions(+), 41 deletions(-) diff --git a/mm/slab.h b/mm/slab.h index ba08da09da36..ddcf59230d2f 100644 --- a/mm/slab.h +++ b/mm/slab.h @@ -786,7 +786,7 @@ void __kmem_obj_info(struct kmem_obj_info *kpp, void *object, struct slab *slab) void __check_heap_object(const void *ptr, unsigned long n, const struct slab *slab, bool to_user); -void defer_free_barrier(void); +void deferred_work_barrier(void); static inline bool slub_debug_orig_size(struct kmem_cache *s) { diff --git a/mm/slab_common.c b/mm/slab_common.c index 2475fc0e5e41..ed49b9abfab2 100644 --- a/mm/slab_common.c +++ b/mm/slab_common.c @@ -551,7 +551,7 @@ void kmem_cache_destroy(struct kmem_cache *s) } /* Wait for deferred work from kmalloc/kfree_nolock() */ - defer_free_barrier(); + deferred_work_barrier(); cpus_read_lock(); mutex_lock(&slab_mutex); @@ -2130,13 +2130,10 @@ void kvfree_rcu_barrier_on_cache(struct kmem_cache *s) cpus_read_lock(); flush_rcu_sheaves_on_cache(s); cpus_read_unlock(); + deferred_work_barrier(); rcu_barrier(); } - /* - * TODO: Introduce a version of __kvfree_rcu_barrier() that works - * on a specific slab cache. - */ __kvfree_rcu_barrier(); } EXPORT_SYMBOL_GPL(kvfree_rcu_barrier_on_cache); diff --git a/mm/slub.c b/mm/slub.c index 31f94bdad3b0..c0cc6d3126c7 100644 --- a/mm/slub.c +++ b/mm/slub.c @@ -418,6 +418,8 @@ struct slab_sheaf { union { struct rcu_head rcu_head; struct list_head barn_list; + /* only used to defer call_rcu() in unknown context */ + struct llist_node llnode; /* only used for prefilled sheafs */ struct { unsigned int capacity; @@ -4046,6 +4048,20 @@ static void flush_all(struct kmem_cache *s) cpus_read_unlock(); } +struct deferred_percpu_work { + struct llist_head objects; + struct llist_head rcu_sheaves; + struct irq_work work; +}; + +static void deferred_percpu_work_fn(struct irq_work *work); + +static DEFINE_PER_CPU(struct deferred_percpu_work, deferred_percpu_work) = { + .objects = LLIST_HEAD_INIT(objects), + .rcu_sheaves = LLIST_HEAD_INIT(rcu_sheaves), + .work = IRQ_WORK_INIT(deferred_percpu_work_fn), +}; + static void flush_rcu_sheaf(struct work_struct *w) { struct slub_percpu_sheaves *pcs; @@ -4117,6 +4133,7 @@ void flush_all_rcu_sheaves(void) mutex_unlock(&slab_mutex); cpus_read_unlock(); + deferred_work_barrier(); rcu_barrier(); } @@ -6132,16 +6149,6 @@ bool __kfree_rcu_sheaf(struct kmem_cache *s, void *obj, unsigned int free_flags) if (likely(rcu_sheaf->size < s->sheaf_capacity)) { rcu_sheaf = NULL; } else { - /* - * With !allow_spin, we might have interrupted call_rcu()'s - * IRQ-disabled critical section. If IRQs are not disabled, - * we know that's not the case. - */ - if (unlikely(!allow_spin && irqs_disabled())) { - rcu_sheaf->size--; - local_unlock(&s->cpu_sheaves->lock); - goto fail; - } pcs->rcu_free = NULL; rcu_sheaf->node = numa_node_id(); } @@ -6150,8 +6157,22 @@ bool __kfree_rcu_sheaf(struct kmem_cache *s, void *obj, unsigned int free_flags) * we flush before local_unlock to make sure a racing * flush_all_rcu_sheaves() doesn't miss this sheaf */ - if (rcu_sheaf) - call_rcu(&rcu_sheaf->rcu_head, rcu_free_sheaf); + if (rcu_sheaf) { + /* + * With !allow_spin, we might have interrupted call_rcu()'s + * IRQ-disabled critical section. If IRQs are not disabled, + * we know that's not the case. + */ + if (unlikely(!allow_spin && irqs_disabled())) { + struct deferred_percpu_work *dpw; + + dpw = this_cpu_ptr(&deferred_percpu_work); + if (llist_add(&rcu_sheaf->llnode, &dpw->rcu_sheaves)) + irq_work_queue(&dpw->work); + } else { + call_rcu(&rcu_sheaf->rcu_head, rcu_free_sheaf); + } + } local_unlock(&s->cpu_sheaves->lock); @@ -6336,31 +6357,21 @@ static void free_to_pcs_bulk(struct kmem_cache *s, size_t size, void **p) } } -struct defer_free { - struct llist_head objects; - struct irq_work work; -}; - -static void free_deferred_objects(struct irq_work *work); - -static DEFINE_PER_CPU(struct defer_free, defer_free_objects) = { - .objects = LLIST_HEAD_INIT(objects), - .work = IRQ_WORK_INIT(free_deferred_objects), -}; - /* * In PREEMPT_RT irq_work runs in per-cpu kthread, so it's safe * to take sleeping spin_locks from __slab_free(). * In !PREEMPT_RT irq_work will run after local_unlock_irqrestore(). */ -static void free_deferred_objects(struct irq_work *work) +static void deferred_percpu_work_fn(struct irq_work *work) { - struct defer_free *df = container_of(work, struct defer_free, work); - struct llist_head *objs = &df->objects; + struct deferred_percpu_work *dpw; + struct llist_head *objs, *rcu_sheaves; struct llist_node *llnode, *pos, *t; + struct slab_sheaf *sheaf, *next; - if (llist_empty(objs)) - return; + dpw = container_of(work, struct deferred_percpu_work, work); + rcu_sheaves = &dpw->rcu_sheaves; + objs = &dpw->objects; llnode = llist_del_all(objs); llist_for_each_safe(pos, t, llnode) { @@ -6384,27 +6395,31 @@ static void free_deferred_objects(struct irq_work *work) __slab_free(s, slab, x, x, 1, _THIS_IP_); stat(s, FREE_SLOWPATH); } + + llnode = llist_del_all(rcu_sheaves); + llist_for_each_entry_safe(sheaf, next, llnode, llnode) + call_rcu(&sheaf->rcu_head, rcu_free_sheaf); } static void defer_free(struct kmem_cache *s, void *head) { - struct defer_free *df; + struct deferred_percpu_work *dpw; guard(preempt)(); head = kasan_reset_tag(head); - df = this_cpu_ptr(&defer_free_objects); - if (llist_add(head + s->offset, &df->objects)) - irq_work_queue(&df->work); + dpw = this_cpu_ptr(&deferred_percpu_work); + if (llist_add(head + s->offset, &dpw->objects)) + irq_work_queue(&dpw->work); } -void defer_free_barrier(void) +void deferred_work_barrier(void) { int cpu; for_each_possible_cpu(cpu) - irq_work_sync(&per_cpu_ptr(&defer_free_objects, cpu)->work); + irq_work_sync(&per_cpu_ptr(&deferred_percpu_work, cpu)->work); } static __fastpath_inline From 2a8bb29ec9b202d3d7bd094c800e6990fee278ba Mon Sep 17 00:00:00 2001 From: "Harry Yoo (Oracle)" Date: Wed, 29 Jul 2026 17:20:13 +0900 Subject: [PATCH 5/9] mm/slab: allow kfree_rcu_sheaf() on PREEMPT_RT As suggested by Vlastimil Babka [1], kfree_rcu_sheaf() can be used on PREEMPT_RT if we always assume spinning is not allowed on PREEMPT_RT. This is because local_trylock and spinlock_t are safe to use with trylock and unlock as long as the kernel does not spin and the context is not NMI and not hardirq. Now that __kfree_rcu_sheaf() knows how to handle SLAB_FREE_NOLOCK, relax the limitation and try the sheaves path on PREEMPT_RT as well. Keep the lockdep map on non RT kernels. However, do not use the lockdep map on PREEMPT_RT to avoid suppressing valid lockdep warnings. As pointed by Vlastimil Babka [2], on PREEMPT_RT it is unnecessary to defer call_rcu() under IRQ-disabled section or raw spinlock. However, let us avoid adding more complexity as the scenario is not supposed to be common on PREEMPT_RT, with a hope that call_rcu_nolock() will be soon supported in RCU. Link: https://lore.kernel.org/linux-mm/6811cc17-8ee4-48c8-8cbf-6bf4d9f98162@kernel.org [1] Link: https://lore.kernel.org/linux-mm/40591888-3a87-433e-b3d2-cda1cab543be@kernel.org [2] Suggested-by: Vlastimil Babka (SUSE) Reviewed-by: Vlastimil Babka (SUSE) Signed-off-by: Harry Yoo (Oracle) Link: https://patch.msgid.link/20260729-kfree_rcu_nolock-v5-5-a28cdcda9673@kernel.org Signed-off-by: Vlastimil Babka (SUSE) --- mm/slab_common.c | 12 ++++++++++-- mm/slub.c | 17 ++++++++++------- 2 files changed, 20 insertions(+), 9 deletions(-) diff --git a/mm/slab_common.c b/mm/slab_common.c index ed49b9abfab2..e776fc4516c8 100644 --- a/mm/slab_common.c +++ b/mm/slab_common.c @@ -1612,6 +1612,14 @@ static bool kfree_rcu_sheaf(void *obj) { struct kmem_cache *s; struct slab *slab; + unsigned int free_flags = SLAB_FREE_DEFAULT; + + /* + * It is not safe to spin on PREEMPT_RT because the kernel might be + * holding a raw spinlock and slab acquires sleeping locks. + */ + if (IS_ENABLED(CONFIG_PREEMPT_RT)) + free_flags = SLAB_FREE_NOLOCK; if (is_vmalloc_addr(obj)) return false; @@ -1622,7 +1630,7 @@ static bool kfree_rcu_sheaf(void *obj) s = slab->slab_cache; if (likely(!IS_ENABLED(CONFIG_NUMA) || slab_nid(slab) == numa_mem_id())) - return __kfree_rcu_sheaf(s, obj, SLAB_FREE_DEFAULT); + return __kfree_rcu_sheaf(s, obj, free_flags); return false; } @@ -1971,7 +1979,7 @@ void kvfree_call_rcu(struct rcu_head *head, void *ptr) if (!head) might_sleep(); - if (!IS_ENABLED(CONFIG_PREEMPT_RT) && kfree_rcu_sheaf(ptr)) + if (kfree_rcu_sheaf(ptr)) return; // Queue the object but don't yet schedule the batch. diff --git a/mm/slub.c b/mm/slub.c index c0cc6d3126c7..4a07ba92b64f 100644 --- a/mm/slub.c +++ b/mm/slub.c @@ -6060,12 +6060,13 @@ static void rcu_free_sheaf(struct rcu_head *head) * kvfree_call_rcu() can be called while holding a raw_spinlock_t. Since * __kfree_rcu_sheaf() may acquire a spinlock_t (sleeping lock on PREEMPT_RT), * this would violate lock nesting rules. Therefore, kvfree_call_rcu() avoids - * this problem by bypassing the sheaves layer entirely on PREEMPT_RT. + * this problem by passing SLAB_FREE_NOLOCK on PREEMPT_RT. * * However, lockdep still complains that it is invalid to acquire spinlock_t * while holding raw_spinlock_t, even on !PREEMPT_RT where spinlock_t is a * spinning lock. Tell lockdep that acquiring spinlock_t is valid here - * by temporarily raising the wait-type to LD_WAIT_CONFIG. + * by temporarily raising the wait-type to LD_WAIT_CONFIG. Skip the lockdep map + * on PREEMPT_RT to avoid suppressing valid lockdep warnings. */ static DEFINE_WAIT_OVERRIDE_MAP(kfree_rcu_sheaf_map, LD_WAIT_CONFIG); @@ -6075,10 +6076,10 @@ bool __kfree_rcu_sheaf(struct kmem_cache *s, void *obj, unsigned int free_flags) struct slab_sheaf *rcu_sheaf; bool allow_spin = free_flags_allow_spinning(free_flags); - if (WARN_ON_ONCE(IS_ENABLED(CONFIG_PREEMPT_RT))) - return false; + VM_WARN_ON_ONCE(IS_ENABLED(CONFIG_PREEMPT_RT) && allow_spin); - lock_map_acquire_try(&kfree_rcu_sheaf_map); + if (!IS_ENABLED(CONFIG_PREEMPT_RT)) + lock_map_acquire_try(&kfree_rcu_sheaf_map); if (!local_trylock(&s->cpu_sheaves->lock)) goto fail; @@ -6177,12 +6178,14 @@ bool __kfree_rcu_sheaf(struct kmem_cache *s, void *obj, unsigned int free_flags) local_unlock(&s->cpu_sheaves->lock); stat(s, FREE_RCU_SHEAF); - lock_map_release(&kfree_rcu_sheaf_map); + if (!IS_ENABLED(CONFIG_PREEMPT_RT)) + lock_map_release(&kfree_rcu_sheaf_map); return true; fail: stat(s, FREE_RCU_SHEAF_FAIL); - lock_map_release(&kfree_rcu_sheaf_map); + if (!IS_ENABLED(CONFIG_PREEMPT_RT)) + lock_map_release(&kfree_rcu_sheaf_map); return false; } From acc6fdade62c11d822f7b73f16092ebd365fc1c2 Mon Sep 17 00:00:00 2001 From: "Harry Yoo (Oracle)" Date: Wed, 29 Jul 2026 17:20:14 +0900 Subject: [PATCH 6/9] mm/slab: introduce struct kvfree_rcu_head for kvfree_rcu batching rcu_head is overkill for kvfree_rcu() because the callback function is always either kfree(), vfree(), or free_large_kmalloc(), and thus there is no need for a function pointer. kvfree_rcu batching reuses the field to store the start address of an object, however, this is not strictly needed because we can calculate the start address in the slowpath. For the purpose of kvfree_rcu batching, it is sufficient to implement a linked list using a single pointer. Introduce a new struct called kvfree_rcu_head (the name was suggested by Vlastimil Babka), which is similar to rcu_head but is only a single pointer to build a linked list, without a function pointer, when CONFIG_KVFREE_RCU_BATCHED=y. When kvfree_rcu is not batched, kvfree_rcu_head is the same size as rcu_head. Note that shrinking struct kvfree_rcu_head on CONFIG_KVFREE_RCU_BATCHED=n kernels would inevitably require additional complexity and also some sort of batching (which defeats the purpose of the config option) because it cannot fall back to call_rcu(). For now there are no user-visible changes to the API. k[v]free_rcu() simply casts rcu_head to kvfree_rcu_head. While this does not affect the API, it allows kfree_rcu_nolock() to reuse kvfree_rcu batching as a fallback when trylock or sheaf allocation fails. Stop storing the object pointer in rcu_head.func and instead calculate the object's start address in kvfree_rcu_list(). Factor out the existing logic to calculate the start address from kvfree_rcu_cb() to kvmalloc_obj_start_addr(). To avoid losing the KASAN tag, calculate the offset and subtract it from the address of the kvfree_rcu_head. Signed-off-by: Harry Yoo (Oracle) Link: https://patch.msgid.link/20260729-kfree_rcu_nolock-v5-6-a28cdcda9673@kernel.org Signed-off-by: Vlastimil Babka (SUSE) --- include/linux/rcupdate.h | 10 ++++++---- include/linux/types.h | 10 ++++++++++ include/trace/events/rcu.h | 2 +- mm/slab.h | 32 ++++++++++++++++++++++++++++++ mm/slab_common.c | 21 ++++++++++---------- mm/slub.c | 40 +++++++++----------------------------- 6 files changed, 68 insertions(+), 47 deletions(-) diff --git a/include/linux/rcupdate.h b/include/linux/rcupdate.h index 5e95acc33989..ef5bb6981133 100644 --- a/include/linux/rcupdate.h +++ b/include/linux/rcupdate.h @@ -1098,19 +1098,21 @@ static inline void rcu_read_unlock_migrate(void) /* * In mm/slab_common.c, no suitable header to include here. */ -void kvfree_call_rcu(struct rcu_head *head, void *ptr); +void kvfree_call_rcu(struct kvfree_rcu_head *head, void *ptr); /* * The BUILD_BUG_ON() makes sure the rcu_head offset can be handled. See the * comment of kfree_rcu() for details. */ -#define kvfree_rcu_arg_2(ptr, rhf) \ +#define kvfree_rcu_arg_2(ptr, kvrhf) \ do { \ typeof (ptr) ___p = (ptr); \ + struct kvfree_rcu_head *___head; \ \ if (___p) { \ - BUILD_BUG_ON(offsetof(typeof(*(ptr)), rhf) >= 4096); \ - kvfree_call_rcu(&((___p)->rhf), (void *) (___p)); \ + BUILD_BUG_ON(offsetof(typeof(*(ptr)), kvrhf) >= 4096); \ + ___head = (struct kvfree_rcu_head *) &(___p)->kvrhf; \ + kvfree_call_rcu(___head, (void *) (___p)); \ } \ } while (0) diff --git a/include/linux/types.h b/include/linux/types.h index 93166b0b0617..7d1d305a763e 100644 --- a/include/linux/types.h +++ b/include/linux/types.h @@ -255,6 +255,16 @@ struct callback_head { } __attribute__((aligned(sizeof(void *)))); #define rcu_head callback_head +#ifdef CONFIG_KVFREE_RCU_BATCHED +struct kvfree_rcu_head { + struct kvfree_rcu_head *next; +}; +#else +struct kvfree_rcu_head { + struct rcu_head head; +}; +#endif + typedef void (*rcu_callback_t)(struct rcu_head *head); typedef void (*call_rcu_func_t)(struct rcu_head *head, rcu_callback_t func); diff --git a/include/trace/events/rcu.h b/include/trace/events/rcu.h index 5fbdabe3faea..a74027126a2f 100644 --- a/include/trace/events/rcu.h +++ b/include/trace/events/rcu.h @@ -625,7 +625,7 @@ TRACE_EVENT_RCU(rcu_invoke_callback, */ TRACE_EVENT_RCU(rcu_invoke_kvfree_callback, - TP_PROTO(const char *rcuname, struct rcu_head *rhp, unsigned long offset), + TP_PROTO(const char *rcuname, struct kvfree_rcu_head *rhp, unsigned long offset), TP_ARGS(rcuname, rhp, offset), diff --git a/mm/slab.h b/mm/slab.h index ddcf59230d2f..1dea45402794 100644 --- a/mm/slab.h +++ b/mm/slab.h @@ -352,6 +352,37 @@ static inline int objs_per_slab(const struct kmem_cache *cache, return slab->objects; } +/* + * kvfree_rcu_head offset can be only less than page size. + * Calculate the start address while preserving the KASAN tag. + */ +static inline void *kvmalloc_obj_start_addr(void *head) +{ + unsigned long offset; + + if (unlikely(is_vmalloc_addr(head))) { + offset = offset_in_page(head); + } else { + struct slab *slab = virt_to_slab(head); + + if (!slab) { + offset = offset_in_page(head); + } else if (is_kfence_address(head)) { + offset = head - kfence_object_start(head); + } else { + struct kmem_cache *s = slab->slab_cache; + unsigned int idx = __obj_to_index(s, slab_address(slab), head); + void *obj = slab_address(slab) + s->size * idx; + + obj = fixup_red_left(s, obj); + obj = kasan_reset_tag(obj); + offset = kasan_reset_tag(head) - obj; + } + } + + return head - offset; +} + /* * State of the slab allocator. * @@ -787,6 +818,7 @@ void __check_heap_object(const void *ptr, unsigned long n, const struct slab *slab, bool to_user); void deferred_work_barrier(void); +void defer_kfree_rcu(struct kvfree_rcu_head *head); static inline bool slub_debug_orig_size(struct kmem_cache *s) { diff --git a/mm/slab_common.c b/mm/slab_common.c index e776fc4516c8..661f1fc24c2d 100644 --- a/mm/slab_common.c +++ b/mm/slab_common.c @@ -1282,11 +1282,11 @@ EXPORT_TRACEPOINT_SYMBOL(kmem_cache_free); #ifndef CONFIG_KVFREE_RCU_BATCHED -void kvfree_call_rcu(struct rcu_head *head, void *ptr) +void kvfree_call_rcu(struct kvfree_rcu_head *head, void *ptr) { if (head) { kasan_record_aux_stack(ptr); - call_rcu(head, kvfree_rcu_cb); + call_rcu(&head->head, kvfree_rcu_cb); return; } @@ -1363,7 +1363,7 @@ struct kvfree_rcu_bulk_data { struct kfree_rcu_cpu_work { struct rcu_work rcu_work; - struct rcu_head *head_free; + struct kvfree_rcu_head *head_free; struct rcu_gp_oldstate head_free_gp_snap; struct list_head bulk_head_free[FREE_N_CHANNELS]; struct kfree_rcu_cpu *krcp; @@ -1399,7 +1399,7 @@ struct kfree_rcu_cpu_work { struct kfree_rcu_cpu { // Objects queued on a linked list // through their rcu_head structures. - struct rcu_head *head; + struct kvfree_rcu_head *head; unsigned long head_gp_snap; atomic_t head_count; @@ -1540,12 +1540,12 @@ kvfree_rcu_bulk(struct kfree_rcu_cpu *krcp, } static void -kvfree_rcu_list(struct rcu_head *head) +kvfree_rcu_list(struct kvfree_rcu_head *head) { - struct rcu_head *next; + struct kvfree_rcu_head *next; for (; head; head = next) { - void *ptr = (void *) head->func; + void *ptr = kvmalloc_obj_start_addr(head); unsigned long offset = (void *) head - ptr; next = head->next; @@ -1569,7 +1569,7 @@ static void kfree_rcu_work(struct work_struct *work) unsigned long flags; struct kvfree_rcu_bulk_data *bnode, *n; struct list_head bulk_head[FREE_N_CHANNELS]; - struct rcu_head *head; + struct kvfree_rcu_head *head; struct kfree_rcu_cpu *krcp; struct kfree_rcu_cpu_work *krwp; struct rcu_gp_oldstate head_gp_snap; @@ -1700,7 +1700,7 @@ kvfree_rcu_drain_ready(struct kfree_rcu_cpu *krcp) { struct list_head bulk_ready[FREE_N_CHANNELS]; struct kvfree_rcu_bulk_data *bnode, *n; - struct rcu_head *head_ready = NULL; + struct kvfree_rcu_head *head_ready = NULL; unsigned long flags; int i; @@ -1963,7 +1963,7 @@ void __init kfree_rcu_scheduler_running(void) * be free'd in workqueue context. This allows us to: batch requests together to * reduce the number of grace periods during heavy kfree_rcu()/kvfree_rcu() load. */ -void kvfree_call_rcu(struct rcu_head *head, void *ptr) +void kvfree_call_rcu(struct kvfree_rcu_head *head, void *ptr) { unsigned long flags; struct kfree_rcu_cpu *krcp; @@ -2001,7 +2001,6 @@ void kvfree_call_rcu(struct rcu_head *head, void *ptr) // Inline if kvfree_rcu(one_arg) call. goto unlock_return; - head->func = ptr; head->next = krcp->head; WRITE_ONCE(krcp->head, head); atomic_inc(&krcp->head_count); diff --git a/mm/slub.c b/mm/slub.c index 4a07ba92b64f..0a7024c8c090 100644 --- a/mm/slub.c +++ b/mm/slub.c @@ -6681,43 +6681,21 @@ static void free_large_kmalloc(struct page *page, void *object) */ void kvfree_rcu_cb(struct rcu_head *head) { - void *obj = head; - struct page *page; - struct slab *slab; - struct kmem_cache *s; - void *slab_addr; + void *obj; + + obj = kvmalloc_obj_start_addr(head); if (is_vmalloc_addr(obj)) { - obj = (void *) PAGE_ALIGN_DOWN((unsigned long)obj); vfree(obj); - return; - } - - page = virt_to_page(obj); - slab = page_slab(page); - if (!slab) { - /* - * rcu_head offset can be only less than page size so no need to - * consider allocation order - */ - obj = (void *) PAGE_ALIGN_DOWN((unsigned long)obj); - free_large_kmalloc(page, obj); - return; - } - - s = slab->slab_cache; - slab_addr = slab_address(slab); - - if (is_kfence_address(obj)) { - obj = kfence_object_start(obj); } else { - unsigned int idx = __obj_to_index(s, slab_addr, obj); + struct page *page = virt_to_page(obj); + struct slab *slab = page_slab(page); - obj = slab_addr + s->size * idx; - obj = fixup_red_left(s, obj); + if (slab) + slab_free(slab->slab_cache, slab, obj, _RET_IP_); + else + free_large_kmalloc(page, obj); } - - slab_free(s, slab, obj, _RET_IP_); } /** From 3bc999d944b35dead1755b2bde1911cd5892225e Mon Sep 17 00:00:00 2001 From: "Harry Yoo (Oracle)" Date: Wed, 29 Jul 2026 17:20:15 +0900 Subject: [PATCH 7/9] mm/slab: introduce kfree_rcu_nolock() Currently, k[v]free_rcu() cannot be called in unknown context since it could lead to a deadlock when called in the middle of k[v]free_rcu(). Make users' lives easier by introducing kfree_rcu_nolock() variant, now that kfree_rcu_sheaf() is available on PREEMPT_RT and __kfree_rcu_sheaf() handles unknown context. When sheaves path fails, kfree_rcu_nolock() falls back to defer_kfree_rcu() that uses an irq work to free the object via kvfree_call_rcu(). In most cases, the sheaves path is expected to succeed and therefore it's unnecessary to introduce additional complexity to the existing kvfree_rcu batching by teaching it how to handle unknown context. Since defer_kfree_rcu() can be called on caches without sheaves, move deferred_work_barrier() and rcu_barrier() outside the branch in kvfree_rcu_barrier_on_cache(). Now that deferred kvfree_rcu objects are submitted to kvfree_call_rcu() after deferred_work_barrier() and may end up in RCU sheaves, deferred_work_barrier() must be invoked before flush_rcu_sheaves_on_cache(). Since the RCU sheaf path has not been used on !KVFREE_RCU_BATCHED kernels, always fall back when kvfree_rcu() is not batched, for consistency. kvfree_rcu_barrier{,_on_cache()}() on !KVFREE_RCU_BATCHED are moved to mm/slab_common.c to invoke deferred_work_barrier() before rcu_barrier(). Signed-off-by: Harry Yoo (Oracle) Link: https://patch.msgid.link/20260729-kfree_rcu_nolock-v5-7-a28cdcda9673@kernel.org Signed-off-by: Vlastimil Babka (SUSE) --- include/linux/rcupdate.h | 22 +++++++++++++++++++ include/linux/slab.h | 16 +++----------- mm/slab_common.c | 47 ++++++++++++++++++++++++++++++++++++++-- mm/slub.c | 28 ++++++++++++++++++++++-- 4 files changed, 96 insertions(+), 17 deletions(-) diff --git a/include/linux/rcupdate.h b/include/linux/rcupdate.h index ef5bb6981133..aede77bd0387 100644 --- a/include/linux/rcupdate.h +++ b/include/linux/rcupdate.h @@ -1099,6 +1099,7 @@ static inline void rcu_read_unlock_migrate(void) * In mm/slab_common.c, no suitable header to include here. */ void kvfree_call_rcu(struct kvfree_rcu_head *head, void *ptr); +void kfree_call_rcu_nolock(struct kvfree_rcu_head *head, void *ptr); /* * The BUILD_BUG_ON() makes sure the rcu_head offset can be handled. See the @@ -1124,6 +1125,27 @@ do { \ kvfree_call_rcu(NULL, (void *) (___p)); \ } while (0) +/** + * kfree_rcu_nolock() - a version of kfree_rcu() that can be called in any context. + * @ptr: pointer to kfree for double-argument invocations. + * @kvrhf: the name of the struct kvfree_rcu_head within the type of @ptr. + * + * With KVFREE_RCU_BATCHED, kfree_rcu_nolock() tries hard to free objects + * without any deferred processing, but may still defer freeing. + * Large kmalloc and vmalloc objects are always deferred. + * + * kfree_rcu_nolock() supports 2-arg variant only. + */ +#define kfree_rcu_nolock(ptr, kvrhf) \ +do { \ + typeof (ptr) ___p = (ptr); \ + \ + if (___p) { \ + BUILD_BUG_ON(offsetof(typeof(*(ptr)), kvrhf) >= 4096); \ + kfree_call_rcu_nolock(&((___p)->kvrhf), (void *) (___p)); \ + } \ +} while (0) + /* * Place this after a lock-acquisition primitive to guarantee that * an UNLOCK+LOCK pair acts as a full barrier. This guarantee applies diff --git a/include/linux/slab.h b/include/linux/slab.h index 32c9f8ed7ae2..066a25cc6966 100644 --- a/include/linux/slab.h +++ b/include/linux/slab.h @@ -1427,25 +1427,15 @@ extern void kvfree_sensitive(const void *addr, size_t len); unsigned int kmem_cache_size(struct kmem_cache *s); #ifndef CONFIG_KVFREE_RCU_BATCHED -static inline void kvfree_rcu_barrier(void) -{ - rcu_barrier(); -} - -static inline void kvfree_rcu_barrier_on_cache(struct kmem_cache *s) -{ - rcu_barrier(); -} - static inline void kfree_rcu_scheduler_running(void) { } #else +void kfree_rcu_scheduler_running(void); +#endif + void kvfree_rcu_barrier(void); void kvfree_rcu_barrier_on_cache(struct kmem_cache *s); -void kfree_rcu_scheduler_running(void); -#endif - /** * kmalloc_size_roundup - Report allocation bucket size for the given size * diff --git a/mm/slab_common.c b/mm/slab_common.c index 661f1fc24c2d..39ca4b4a182a 100644 --- a/mm/slab_common.c +++ b/mm/slab_common.c @@ -1280,6 +1280,33 @@ EXPORT_TRACEPOINT_SYMBOL(kmem_cache_alloc); EXPORT_TRACEPOINT_SYMBOL(kfree); EXPORT_TRACEPOINT_SYMBOL(kmem_cache_free); +void kfree_call_rcu_nolock(struct kvfree_rcu_head *head, void *ptr) +{ + struct slab *slab; + + if (!IS_ENABLED(CONFIG_KVFREE_RCU_BATCHED)) + goto fallback; + + if (unlikely(is_vmalloc_addr(ptr))) + goto fallback; + + slab = virt_to_slab(ptr); + if (unlikely(!slab)) + goto fallback; + + if (unlikely(IS_ENABLED(CONFIG_NUMA) && slab_nid(slab) != numa_mem_id())) + goto fallback; + + if (unlikely(!__kfree_rcu_sheaf(slab->slab_cache, ptr, SLAB_FREE_NOLOCK))) + goto fallback; + + return; + +fallback: + defer_kfree_rcu(head); +} +EXPORT_SYMBOL_GPL(kfree_call_rcu_nolock); + #ifndef CONFIG_KVFREE_RCU_BATCHED void kvfree_call_rcu(struct kvfree_rcu_head *head, void *ptr) @@ -1297,6 +1324,20 @@ void kvfree_call_rcu(struct kvfree_rcu_head *head, void *ptr) } EXPORT_SYMBOL_GPL(kvfree_call_rcu); +void kvfree_rcu_barrier(void) +{ + deferred_work_barrier(); + rcu_barrier(); +} +EXPORT_SYMBOL_GPL(kvfree_rcu_barrier); + +void kvfree_rcu_barrier_on_cache(struct kmem_cache *s) +{ + deferred_work_barrier(); + rcu_barrier(); +} +EXPORT_SYMBOL_GPL(kvfree_rcu_barrier_on_cache); + void __init kvfree_rcu_init(void) { } @@ -2133,14 +2174,16 @@ EXPORT_SYMBOL_GPL(kvfree_rcu_barrier); */ void kvfree_rcu_barrier_on_cache(struct kmem_cache *s) { + /* kfree_rcu_nolock() might have deferred frees even without sheaves */ + deferred_work_barrier(); + if (cache_has_sheaves(s)) { cpus_read_lock(); flush_rcu_sheaves_on_cache(s); cpus_read_unlock(); - deferred_work_barrier(); - rcu_barrier(); } + rcu_barrier(); __kvfree_rcu_barrier(); } EXPORT_SYMBOL_GPL(kvfree_rcu_barrier_on_cache); diff --git a/mm/slub.c b/mm/slub.c index 0a7024c8c090..b9aeb02a880f 100644 --- a/mm/slub.c +++ b/mm/slub.c @@ -4050,6 +4050,7 @@ static void flush_all(struct kmem_cache *s) struct deferred_percpu_work { struct llist_head objects; + struct llist_head objects_by_rcu; struct llist_head rcu_sheaves; struct irq_work work; }; @@ -4058,6 +4059,7 @@ static void deferred_percpu_work_fn(struct irq_work *work); static DEFINE_PER_CPU(struct deferred_percpu_work, deferred_percpu_work) = { .objects = LLIST_HEAD_INIT(objects), + .objects_by_rcu = LLIST_HEAD_INIT(objects_by_rcu), .rcu_sheaves = LLIST_HEAD_INIT(rcu_sheaves), .work = IRQ_WORK_INIT(deferred_percpu_work_fn), }; @@ -4121,6 +4123,8 @@ void flush_all_rcu_sheaves(void) { struct kmem_cache *s; + deferred_work_barrier(); + cpus_read_lock(); mutex_lock(&slab_mutex); @@ -4133,7 +4137,6 @@ void flush_all_rcu_sheaves(void) mutex_unlock(&slab_mutex); cpus_read_unlock(); - deferred_work_barrier(); rcu_barrier(); } @@ -6368,13 +6371,14 @@ static void free_to_pcs_bulk(struct kmem_cache *s, size_t size, void **p) static void deferred_percpu_work_fn(struct irq_work *work) { struct deferred_percpu_work *dpw; - struct llist_head *objs, *rcu_sheaves; + struct llist_head *objs, *objs_by_rcu, *rcu_sheaves; struct llist_node *llnode, *pos, *t; struct slab_sheaf *sheaf, *next; dpw = container_of(work, struct deferred_percpu_work, work); rcu_sheaves = &dpw->rcu_sheaves; objs = &dpw->objects; + objs_by_rcu = &dpw->objects_by_rcu; llnode = llist_del_all(objs); llist_for_each_safe(pos, t, llnode) { @@ -6399,6 +6403,14 @@ static void deferred_percpu_work_fn(struct irq_work *work) stat(s, FREE_SLOWPATH); } + llnode = llist_del_all(objs_by_rcu); + llist_for_each_safe(pos, t, llnode) { + void *head = pos; + void *objp = kvmalloc_obj_start_addr(head); + + kvfree_call_rcu(head, objp); + } + llnode = llist_del_all(rcu_sheaves); llist_for_each_entry_safe(sheaf, next, llnode, llnode) call_rcu(&sheaf->rcu_head, rcu_free_sheaf); @@ -6417,6 +6429,18 @@ static void defer_free(struct kmem_cache *s, void *head) irq_work_queue(&dpw->work); } +void defer_kfree_rcu(struct kvfree_rcu_head *head) +{ + struct deferred_percpu_work *dpw; + + guard(preempt)(); + + dpw = this_cpu_ptr(&deferred_percpu_work); + if (llist_add((struct llist_node *)head, &dpw->objects_by_rcu)) + irq_work_queue(&dpw->work); +} + +/* Must be called before flush_rcu_sheaves_on_cache() */ void deferred_work_barrier(void) { int cpu; From 7df60eeb6736013ee1555a19e261a7d14e84f250 Mon Sep 17 00:00:00 2001 From: "Harry Yoo (Oracle)" Date: Wed, 29 Jul 2026 17:20:16 +0900 Subject: [PATCH 8/9] slub_kunit: extend the test for kfree_rcu_nolock() When slub_kunit is not built-in, call kfree_rcu() and kfree_rcu_nolock() to test kfree_rcu_nolock() in slub_kunit. Rename the test case as the test covers more _nolock() APIs. Acked-by: Vlastimil Babka (SUSE) Signed-off-by: Harry Yoo (Oracle) Link: https://patch.msgid.link/20260729-kfree_rcu_nolock-v5-8-a28cdcda9673@kernel.org Signed-off-by: Vlastimil Babka (SUSE) --- lib/tests/slub_kunit.c | 47 +++++++++++++++++++++++++++--------------- 1 file changed, 30 insertions(+), 17 deletions(-) diff --git a/lib/tests/slub_kunit.c b/lib/tests/slub_kunit.c index 8c2b9911471e..e3b63f0338d5 100644 --- a/lib/tests/slub_kunit.c +++ b/lib/tests/slub_kunit.c @@ -162,7 +162,10 @@ static void test_kmalloc_redzone_access(struct kunit *test) } struct test_kfree_rcu_struct { - struct rcu_head rcu; + union { + struct rcu_head rcu; + struct kvfree_rcu_head kvrcu; + }; }; static void test_kfree_rcu(struct kunit *test) @@ -296,7 +299,7 @@ static void test_krealloc_redzone_zeroing(struct kunit *test) #if defined(CONFIG_PERF_EVENTS) || (defined(CONFIG_KPROBES) && defined(CONFIG_SMP)) #define NR_ITERATIONS 1000 #define NR_OBJECTS 1000 -static void *objects[NR_OBJECTS]; +static struct test_kfree_rcu_struct *objects[NR_OBJECTS]; struct test_nolock_context { struct kunit *test; @@ -311,15 +314,16 @@ struct test_nolock_context { #endif }; -static void test_kmalloc_kfree(void) +static void test_kmalloc_and_friends(void) { int i, j; + bool can_use_kfree_rcu = !IS_BUILTIN(CONFIG_SLUB_KUNIT_TEST); for (i = 0; i < NR_ITERATIONS; i++) { for (j = 0; j < NR_OBJECTS; j++) { - gfp_t gfp = (i % 2) ? GFP_KERNEL : GFP_KERNEL_ACCOUNT; + gfp_t gfp = (i & 1) ? GFP_KERNEL : GFP_KERNEL_ACCOUNT; - objects[j] = kmalloc(64, gfp); + objects[j] = kmalloc_obj(*objects[j], gfp); if (!objects[j]) { j--; while (j >= 0) @@ -328,26 +332,35 @@ static void test_kmalloc_kfree(void) } } - for (j = 0; j < NR_OBJECTS; j++) - kfree(objects[j]); + for (j = 0; j < NR_OBJECTS; j++) { + if (can_use_kfree_rcu && (i & 2)) + kfree_rcu(objects[j], rcu); + else + kfree(objects[j]); + } } } static void test_nolock(struct test_nolock_context *ctx) { - void *objp; + struct test_kfree_rcu_struct *objp; gfp_t gfp; + bool can_use_kfree_rcu = !IS_BUILTIN(CONFIG_SLUB_KUNIT_TEST); /* __GFP_ACCOUNT to test kmalloc_nolock() in alloc_slab_obj_exts() */ - gfp = (ctx->callback_count % 2) ? 0 : __GFP_ACCOUNT; - objp = kmalloc_nolock(64, gfp, NUMA_NO_NODE); + gfp = (ctx->callback_count & 1) ? 0 : __GFP_ACCOUNT; + objp = kmalloc_nolock(sizeof(*objp), gfp, NUMA_NO_NODE); if (objp) ctx->alloc_ok++; else ctx->alloc_fail++; - kfree_nolock(objp); + if (can_use_kfree_rcu && (ctx->callback_count & 2)) + kfree_rcu_nolock(objp, kvrcu); + else + kfree_nolock(objp); + ctx->callback_count++; } #endif @@ -397,14 +410,14 @@ static void disable_perf_events(struct test_nolock_context *ctx) perf_event_release_kernel(ctx->event); } -static void test_kmalloc_kfree_nolock_perf(struct kunit *test) +static void test_kmalloc_nolock_and_friends_perf(struct kunit *test) { struct test_nolock_context ctx = { .test = test }; if (!enable_perf_events(&ctx)) kunit_skip(test, "Failed to enable perf event, skipping"); - test_kmalloc_kfree(); + test_kmalloc_and_friends(); disable_perf_events(&ctx); KUNIT_EXPECT_EQ(test, 0, slab_errors); @@ -438,14 +451,14 @@ static void unregister_slab_kprobes(struct test_nolock_context *ctx) unregister_kprobe(&ctx->kprobe); } -static void test_kmalloc_kfree_nolock_kprobe(struct kunit *test) +static void test_kmalloc_nolock_and_friends_kprobe(struct kunit *test) { struct test_nolock_context ctx = { .test = test }; if (!register_slab_kprobes(&ctx)) kunit_skip(test, "Failed to register kprobe, skipping"); - test_kmalloc_kfree(); + test_kmalloc_and_friends(); unregister_slab_kprobes(&ctx); KUNIT_EXPECT_EQ(test, 0, slab_errors); @@ -477,10 +490,10 @@ static struct kunit_case test_cases[] = { KUNIT_CASE(test_leak_destroy), KUNIT_CASE(test_krealloc_redzone_zeroing), #ifdef CONFIG_PERF_EVENTS - KUNIT_CASE_SLOW(test_kmalloc_kfree_nolock_perf), + KUNIT_CASE_SLOW(test_kmalloc_nolock_and_friends_perf), #endif #if defined(CONFIG_KPROBES) && defined(CONFIG_SMP) - KUNIT_CASE_SLOW(test_kmalloc_kfree_nolock_kprobe), + KUNIT_CASE_SLOW(test_kmalloc_nolock_and_friends_kprobe), #endif {} }; From 648294a02bfcd0eddae51877e3b30f8bbb2d4bb6 Mon Sep 17 00:00:00 2001 From: "Vlastimil Babka (SUSE)" Date: Thu, 30 Jul 2026 16:32:46 +0200 Subject: [PATCH 9/9] mm/slab: stop exporting kvfree_rcu_barrier[_on_cache]() No module code calls these functions directly. Seems it was always the case. Remove the exports. Acked-by: Paul E. McKenney Reviewed-by: Harry Yoo Link: https://patch.msgid.link/20260730-unexport-barriers-v1-1-852f6641abe9@kernel.org Signed-off-by: Vlastimil Babka (SUSE) --- mm/slab_common.c | 4 ---- 1 file changed, 4 deletions(-) diff --git a/mm/slab_common.c b/mm/slab_common.c index 39ca4b4a182a..e71a1b346b36 100644 --- a/mm/slab_common.c +++ b/mm/slab_common.c @@ -1329,14 +1329,12 @@ void kvfree_rcu_barrier(void) deferred_work_barrier(); rcu_barrier(); } -EXPORT_SYMBOL_GPL(kvfree_rcu_barrier); void kvfree_rcu_barrier_on_cache(struct kmem_cache *s) { deferred_work_barrier(); rcu_barrier(); } -EXPORT_SYMBOL_GPL(kvfree_rcu_barrier_on_cache); void __init kvfree_rcu_init(void) { @@ -2163,7 +2161,6 @@ void kvfree_rcu_barrier(void) flush_all_rcu_sheaves(); __kvfree_rcu_barrier(); } -EXPORT_SYMBOL_GPL(kvfree_rcu_barrier); /** * kvfree_rcu_barrier_on_cache - Wait for in-flight kvfree_rcu() calls on a @@ -2186,7 +2183,6 @@ void kvfree_rcu_barrier_on_cache(struct kmem_cache *s) rcu_barrier(); __kvfree_rcu_barrier(); } -EXPORT_SYMBOL_GPL(kvfree_rcu_barrier_on_cache); static unsigned long kfree_rcu_shrink_count(struct shrinker *shrink, struct shrink_control *sc)