diff --git a/Documentation/ABI/testing/sysfs-kernel-slab b/Documentation/ABI/testing/sysfs-kernel-slab index b26e4299f822..c52034e3e794 100644 --- a/Documentation/ABI/testing/sysfs-kernel-slab +++ b/Documentation/ABI/testing/sysfs-kernel-slab @@ -113,8 +113,10 @@ KernelVersion: 2.6.22 Contact: Pekka Enberg , Christoph Lameter Description: - The cpu_slabs file is read-only and displays how many cpu slabs - are active and their NUMA locality. + The cpu_slabs file is read-only. It is deprecated and always + reads "0" since the removal of per-cpu slabs in Linux 7.0. It + previously displayed how many cpu slabs were active and their + NUMA locality. The file is kept for backwards compatibility. What: /sys/kernel/slab//cpuslab_flush Date: April 2009 @@ -509,12 +511,16 @@ What: /sys/kernel/slab//slabs_cpu_partial Date: Aug 2011 Contact: Christoph Lameter Description: - This read-only file shows the number of partialli allocated - frozen slabs. + This read-only file is deprecated and always reads "0(0)" since + the removal of per-cpu partial slabs in Linux 7.0. It previously + showed the number of partially allocated frozen slabs. The file + is kept for backwards compatibility. What: /sys/kernel/slab//cpu_partial Date: Aug 2011 Contact: Christoph Lameter Description: - This read-only file shows the number of per cpu partial - pages to keep around. + This file is deprecated and always reads "0" since the removal of + per-cpu partial slabs in Linux 7.0. It previously showed the + number of per-cpu partial pages to keep around. The file is kept + for backwards compatibility. diff --git a/Documentation/admin-guide/kernel-parameters.txt b/Documentation/admin-guide/kernel-parameters.txt index 66c1c879a0b2..caa517836ece 100644 --- a/Documentation/admin-guide/kernel-parameters.txt +++ b/Documentation/admin-guide/kernel-parameters.txt @@ -4014,6 +4014,11 @@ Kernel parameters Note that even when enabled, there are a few cases where the feature is not effective. + mempool_debug [MM] + Enable mempool debugging. This enables element + poison checking when freeing elements back to the + pool. Useful for debugging mempool corruption. + memtest= [KNL,X86,ARM,M68K,PPC,RISCV,EARLY] Enable memtest Format: default : 0 diff --git a/Documentation/mm/allocation-profiling.rst b/Documentation/mm/allocation-profiling.rst index 5389d241176a..d02eb54ee8f2 100644 --- a/Documentation/mm/allocation-profiling.rst +++ b/Documentation/mm/allocation-profiling.rst @@ -112,3 +112,10 @@ To do so: - Then, use the following form for your allocations: alloc_hooks_tag(ht->your_saved_tag, kmalloc_noprof(...)) + +Notes +===== + +- When a slab object is allocated from KFENCE, its accounting is skipped. + KFENCE allocations are rare and limited to a small number, so this omission + is negligible. diff --git a/include/linux/memcontrol.h b/include/linux/memcontrol.h index 8170bb8066a2..71d045fe127f 100644 --- a/include/linux/memcontrol.h +++ b/include/linux/memcontrol.h @@ -1461,19 +1461,6 @@ static inline void mem_cgroup_flush_workqueue(void) { } static inline int mem_cgroup_init(void) { return 0; } #endif /* CONFIG_MEMCG */ -/* - * Extended information for slab objects stored as an array in page->memcg_data - * if MEMCG_DATA_OBJEXTS is set. - */ -struct slabobj_ext { -#ifdef CONFIG_MEMCG - struct obj_cgroup *objcg; -#endif -#ifdef CONFIG_MEM_ALLOC_PROFILING - union codetag_ref ref; -#endif -} __aligned(8); - static inline struct lruvec *parent_lruvec(struct lruvec *lruvec) { struct mem_cgroup *memcg; diff --git a/include/linux/rcupdate.h b/include/linux/rcupdate.h index 9a741ce05885..44c07a66edff 100644 --- a/include/linux/rcupdate.h +++ b/include/linux/rcupdate.h @@ -1107,19 +1107,22 @@ static inline void rcu_read_unlock_migrate(void) /* * In mm/slab_common.c, no suitable header to include here. */ -void kvfree_call_rcu(struct rcu_head *head, void *ptr); +void kvfree_call_rcu(struct kvfree_rcu_head *head, void *ptr); +void kfree_call_rcu_nolock(struct kvfree_rcu_head *head, void *ptr); /* * The BUILD_BUG_ON() makes sure the rcu_head offset can be handled. See the * comment of kfree_rcu() for details. */ -#define kvfree_rcu_arg_2(ptr, rhf) \ +#define kvfree_rcu_arg_2(ptr, kvrhf) \ do { \ typeof (ptr) ___p = (ptr); \ + struct kvfree_rcu_head *___head; \ \ if (___p) { \ - BUILD_BUG_ON(offsetof(typeof(*(ptr)), rhf) >= 4096); \ - kvfree_call_rcu(&((___p)->rhf), (void *) (___p)); \ + BUILD_BUG_ON(offsetof(typeof(*(ptr)), kvrhf) >= 4096); \ + ___head = (struct kvfree_rcu_head *) &(___p)->kvrhf; \ + kvfree_call_rcu(___head, (void *) (___p)); \ } \ } while (0) @@ -1131,6 +1134,27 @@ do { \ kvfree_call_rcu(NULL, (void *) (___p)); \ } while (0) +/** + * kfree_rcu_nolock() - a version of kfree_rcu() that can be called in any context. + * @ptr: pointer to kfree for double-argument invocations. + * @kvrhf: the name of the struct kvfree_rcu_head within the type of @ptr. + * + * With KVFREE_RCU_BATCHED, kfree_rcu_nolock() tries hard to free objects + * without any deferred processing, but may still defer freeing. + * Large kmalloc and vmalloc objects are always deferred. + * + * kfree_rcu_nolock() supports 2-arg variant only. + */ +#define kfree_rcu_nolock(ptr, kvrhf) \ +do { \ + typeof (ptr) ___p = (ptr); \ + \ + if (___p) { \ + BUILD_BUG_ON(offsetof(typeof(*(ptr)), kvrhf) >= 4096); \ + kfree_call_rcu_nolock(&((___p)->kvrhf), (void *) (___p)); \ + } \ +} while (0) + /* * Place this after a lock-acquisition primitive to guarantee that * an UNLOCK+LOCK pair acts as a full barrier. This guarantee applies diff --git a/include/linux/slab.h b/include/linux/slab.h index 32c9f8ed7ae2..cda126def67a 100644 --- a/include/linux/slab.h +++ b/include/linux/slab.h @@ -45,6 +45,7 @@ enum _slab_flag_bits { #endif #ifdef CONFIG_MEMCG _SLAB_ACCOUNT, + _SLAB_MAY_ACCOUNT, #endif #ifdef CONFIG_KASAN_GENERIC _SLAB_KASAN, @@ -204,8 +205,10 @@ enum _slab_flag_bits { */ #ifdef CONFIG_MEMCG # define SLAB_ACCOUNT __SLAB_FLAG_BIT(_SLAB_ACCOUNT) +# define SLAB_MAY_ACCOUNT __SLAB_FLAG_BIT(_SLAB_MAY_ACCOUNT) #else # define SLAB_ACCOUNT __SLAB_FLAG_UNUSED +# define SLAB_MAY_ACCOUNT __SLAB_FLAG_UNUSED #endif #ifdef CONFIG_KASAN_GENERIC @@ -1427,25 +1430,15 @@ extern void kvfree_sensitive(const void *addr, size_t len); unsigned int kmem_cache_size(struct kmem_cache *s); #ifndef CONFIG_KVFREE_RCU_BATCHED -static inline void kvfree_rcu_barrier(void) -{ - rcu_barrier(); -} - -static inline void kvfree_rcu_barrier_on_cache(struct kmem_cache *s) -{ - rcu_barrier(); -} - static inline void kfree_rcu_scheduler_running(void) { } #else +void kfree_rcu_scheduler_running(void); +#endif + void kvfree_rcu_barrier(void); void kvfree_rcu_barrier_on_cache(struct kmem_cache *s); -void kfree_rcu_scheduler_running(void); -#endif - /** * kmalloc_size_roundup - Report allocation bucket size for the given size * diff --git a/include/linux/types.h b/include/linux/types.h index bc5dda2a3d86..53e0adca4b9f 100644 --- a/include/linux/types.h +++ b/include/linux/types.h @@ -257,6 +257,16 @@ struct callback_head { } __attribute__((aligned(sizeof(void *)))); #define rcu_head callback_head +#ifdef CONFIG_KVFREE_RCU_BATCHED +struct kvfree_rcu_head { + struct kvfree_rcu_head *next; +}; +#else +struct kvfree_rcu_head { + struct rcu_head head; +}; +#endif + typedef void (*rcu_callback_t)(struct rcu_head *head); typedef void (*call_rcu_func_t)(struct rcu_head *head, rcu_callback_t func); diff --git a/include/trace/events/rcu.h b/include/trace/events/rcu.h index c84309c38834..991fd6f41e4b 100644 --- a/include/trace/events/rcu.h +++ b/include/trace/events/rcu.h @@ -626,7 +626,7 @@ TRACE_EVENT_RCU(rcu_invoke_callback, */ TRACE_EVENT_RCU(rcu_invoke_kvfree_callback, - TP_PROTO(const char *rcuname, struct rcu_head *rhp, unsigned long offset), + TP_PROTO(const char *rcuname, struct kvfree_rcu_head *rhp, unsigned long offset), TP_ARGS(rcuname, rhp, offset), diff --git a/lib/tests/slub_kunit.c b/lib/tests/slub_kunit.c index fa6d31dbca16..e3b63f0338d5 100644 --- a/lib/tests/slub_kunit.c +++ b/lib/tests/slub_kunit.c @@ -8,6 +8,7 @@ #include #include #include +#include #include "../mm/slab.h" static struct kunit_resource resource; @@ -161,7 +162,10 @@ static void test_kmalloc_redzone_access(struct kunit *test) } struct test_kfree_rcu_struct { - struct rcu_head rcu; + union { + struct rcu_head rcu; + struct kvfree_rcu_head kvrcu; + }; }; static void test_kfree_rcu(struct kunit *test) @@ -292,19 +296,76 @@ static void test_krealloc_redzone_zeroing(struct kunit *test) kmem_cache_destroy(s); } -#ifdef CONFIG_PERF_EVENTS +#if defined(CONFIG_PERF_EVENTS) || (defined(CONFIG_KPROBES) && defined(CONFIG_SMP)) #define NR_ITERATIONS 1000 #define NR_OBJECTS 1000 -static void *objects[NR_OBJECTS]; +static struct test_kfree_rcu_struct *objects[NR_OBJECTS]; struct test_nolock_context { struct kunit *test; int callback_count; int alloc_ok; int alloc_fail; +#ifdef CONFIG_PERF_EVENTS struct perf_event *event; +#endif +#if defined(CONFIG_KPROBES) && defined(CONFIG_SMP) + struct kprobe kprobe; +#endif }; +static void test_kmalloc_and_friends(void) +{ + int i, j; + bool can_use_kfree_rcu = !IS_BUILTIN(CONFIG_SLUB_KUNIT_TEST); + + for (i = 0; i < NR_ITERATIONS; i++) { + for (j = 0; j < NR_OBJECTS; j++) { + gfp_t gfp = (i & 1) ? GFP_KERNEL : GFP_KERNEL_ACCOUNT; + + objects[j] = kmalloc_obj(*objects[j], gfp); + if (!objects[j]) { + j--; + while (j >= 0) + kfree(objects[j--]); + return; + } + } + + for (j = 0; j < NR_OBJECTS; j++) { + if (can_use_kfree_rcu && (i & 2)) + kfree_rcu(objects[j], rcu); + else + kfree(objects[j]); + } + } +} + +static void test_nolock(struct test_nolock_context *ctx) +{ + struct test_kfree_rcu_struct *objp; + gfp_t gfp; + bool can_use_kfree_rcu = !IS_BUILTIN(CONFIG_SLUB_KUNIT_TEST); + + /* __GFP_ACCOUNT to test kmalloc_nolock() in alloc_slab_obj_exts() */ + gfp = (ctx->callback_count & 1) ? 0 : __GFP_ACCOUNT; + objp = kmalloc_nolock(sizeof(*objp), gfp, NUMA_NO_NODE); + + if (objp) + ctx->alloc_ok++; + else + ctx->alloc_fail++; + + if (can_use_kfree_rcu && (ctx->callback_count & 2)) + kfree_rcu_nolock(objp, kvrcu); + else + kfree_nolock(objp); + + ctx->callback_count++; +} +#endif + +#ifdef CONFIG_PERF_EVENTS static struct perf_event_attr hw_attr = { .type = PERF_TYPE_HARDWARE, .config = PERF_COUNT_HW_CPU_CYCLES, @@ -315,67 +376,91 @@ static struct perf_event_attr hw_attr = { .sample_freq = 100000, }; -static void overflow_handler_test_kmalloc_kfree_nolock(struct perf_event *event, - struct perf_sample_data *data, - struct pt_regs *regs) +static void overflow_handler_test_nolock(struct perf_event *event, + struct perf_sample_data *data, + struct pt_regs *regs) { - void *objp; - gfp_t gfp; struct test_nolock_context *ctx = event->overflow_handler_context; - /* __GFP_ACCOUNT to test kmalloc_nolock() in alloc_slab_obj_exts() */ - gfp = (ctx->callback_count % 2) ? 0 : __GFP_ACCOUNT; - objp = kmalloc_nolock(64, gfp, NUMA_NO_NODE); - - if (objp) - ctx->alloc_ok++; - else - ctx->alloc_fail++; - - kfree_nolock(objp); - ctx->callback_count++; + test_nolock(ctx); } -static void test_kmalloc_kfree_nolock(struct kunit *test) +static bool enable_perf_events(struct test_nolock_context *ctx) { - int i, j; - struct test_nolock_context ctx = { .test = test }; struct perf_event *event; - bool alloc_fail = false; event = perf_event_create_kernel_counter(&hw_attr, -1, current, - overflow_handler_test_kmalloc_kfree_nolock, - &ctx); + overflow_handler_test_nolock, + ctx); + if (IS_ERR(event)) - kunit_skip(test, "Failed to create perf event"); - ctx.event = event; - perf_event_enable(ctx.event); - for (i = 0; i < NR_ITERATIONS; i++) { - for (j = 0; j < NR_OBJECTS; j++) { - gfp_t gfp = (i % 2) ? GFP_KERNEL : GFP_KERNEL_ACCOUNT; + return false; - objects[j] = kmalloc(64, gfp); - if (!objects[j]) { - j--; - while (j >= 0) - kfree(objects[j--]); - alloc_fail = true; - goto cleanup; - } - } - for (j = 0; j < NR_OBJECTS; j++) - kfree(objects[j]); - } + ctx->event = event; + perf_event_enable(ctx->event); + return true; +} -cleanup: - perf_event_disable(ctx.event); - perf_event_release_kernel(ctx.event); +static void disable_perf_events(struct test_nolock_context *ctx) +{ + kunit_info(ctx->test, "HW perf events: callback_count: %d, alloc_ok: %d, alloc_fail: %d\n", + ctx->callback_count, ctx->alloc_ok, ctx->alloc_fail); - kunit_info(test, "callback_count: %d, alloc_ok: %d, alloc_fail: %d\n", - ctx.callback_count, ctx.alloc_ok, ctx.alloc_fail); + perf_event_disable(ctx->event); + perf_event_release_kernel(ctx->event); +} - if (alloc_fail) - kunit_skip(test, "Allocation failed"); +static void test_kmalloc_nolock_and_friends_perf(struct kunit *test) +{ + struct test_nolock_context ctx = { .test = test }; + + if (!enable_perf_events(&ctx)) + kunit_skip(test, "Failed to enable perf event, skipping"); + + test_kmalloc_and_friends(); + + disable_perf_events(&ctx); + KUNIT_EXPECT_EQ(test, 0, slab_errors); +} +#endif + +#if defined(CONFIG_KPROBES) && defined(CONFIG_SMP) +static int slab_kprobe_pre_handler(struct kprobe *p, struct pt_regs *regs) +{ + struct test_nolock_context *ctx; + + ctx = container_of(p, struct test_nolock_context, kprobe); + test_nolock(ctx); + return 0; +} + +static bool register_slab_kprobes(struct test_nolock_context *ctx) +{ + ctx->kprobe.symbol_name = "slab_attach_kprobe_locked"; + ctx->kprobe.pre_handler = slab_kprobe_pre_handler; + + if (register_kprobe(&ctx->kprobe)) + return false; + return true; +} + +static void unregister_slab_kprobes(struct test_nolock_context *ctx) +{ + kunit_info(ctx->test, "kprobes: callback_count: %d, alloc_ok: %d, alloc_fail: %d\n", + ctx->callback_count, ctx->alloc_ok, ctx->alloc_fail); + unregister_kprobe(&ctx->kprobe); +} + +static void test_kmalloc_nolock_and_friends_kprobe(struct kunit *test) +{ + struct test_nolock_context ctx = { .test = test }; + + if (!register_slab_kprobes(&ctx)) + kunit_skip(test, "Failed to register kprobe, skipping"); + + test_kmalloc_and_friends(); + + unregister_slab_kprobes(&ctx); KUNIT_EXPECT_EQ(test, 0, slab_errors); } #endif @@ -405,7 +490,10 @@ static struct kunit_case test_cases[] = { KUNIT_CASE(test_leak_destroy), KUNIT_CASE(test_krealloc_redzone_zeroing), #ifdef CONFIG_PERF_EVENTS - KUNIT_CASE_SLOW(test_kmalloc_kfree_nolock), + KUNIT_CASE_SLOW(test_kmalloc_nolock_and_friends_perf), +#endif +#if defined(CONFIG_KPROBES) && defined(CONFIG_SMP) + KUNIT_CASE_SLOW(test_kmalloc_nolock_and_friends_kprobe), #endif {} }; diff --git a/mm/kfence/core.c b/mm/kfence/core.c index 6577bd76954e..90925c646c4c 100644 --- a/mm/kfence/core.c +++ b/mm/kfence/core.c @@ -636,11 +636,6 @@ static unsigned long kfence_init_pool(void) page = pfn_to_page(start_pfn + i); __SetPageSlab(page); -#ifdef CONFIG_MEMCG - struct slab *slab = page_slab(page); - slab->obj_exts = (unsigned long)&kfence_metadata_init[i / 2 - 1].obj_exts | - MEMCG_DATA_OBJEXTS; -#endif } /* @@ -704,10 +699,6 @@ static unsigned long kfence_init_pool(void) continue; page = pfn_to_page(start_pfn + i); -#ifdef CONFIG_MEMCG - struct slab *slab = page_slab(page); - slab->obj_exts = 0; -#endif __ClearPageSlab(page); } @@ -1248,9 +1239,6 @@ void __kfence_free(void *addr) { struct kfence_metadata *meta = addr_to_metadata((unsigned long)addr); -#ifdef CONFIG_MEMCG - KFENCE_WARN_ON(meta->obj_exts.objcg); -#endif /* * If the objects of the cache are SLAB_TYPESAFE_BY_RCU, defer freeing * the object, as the object page may be recycled for other-typed diff --git a/mm/kfence/kfence.h b/mm/kfence/kfence.h index 1f618f9b0d12..e6b4bf349ff7 100644 --- a/mm/kfence/kfence.h +++ b/mm/kfence/kfence.h @@ -102,9 +102,6 @@ struct kfence_metadata { struct kfence_track free_track __guarded_by(&lock); /* For updating alloc_covered on frees. */ u32 alloc_stack_hash __guarded_by(&lock); -#ifdef CONFIG_MEMCG - struct slabobj_ext obj_exts; -#endif }; #define KFENCE_METADATA_SIZE PAGE_ALIGN(sizeof(struct kfence_metadata) * \ diff --git a/mm/kfence/kfence_test.c b/mm/kfence/kfence_test.c index de2d0f7d62b1..9867c03ef0ae 100644 --- a/mm/kfence/kfence_test.c +++ b/mm/kfence/kfence_test.c @@ -295,7 +295,7 @@ static void *test_alloc(struct kunit *test, size_t size, gfp_t gfp, enum allocat * memcg accounting works correctly. */ KUNIT_EXPECT_EQ(test, obj_to_index(s, slab, alloc), 0U); - KUNIT_EXPECT_EQ(test, objs_per_slab(s, slab), 1); + KUNIT_EXPECT_EQ(test, ((unsigned int)slab->objects), 1); if (policy == ALLOCATE_ANY) return alloc; diff --git a/mm/memcontrol.c b/mm/memcontrol.c index 69b37f63a307..1ebceade4021 100644 --- a/mm/memcontrol.c +++ b/mm/memcontrol.c @@ -2870,18 +2870,19 @@ struct mem_cgroup *mem_cgroup_from_obj_slab(struct slab *slab, void *p) */ unsigned long obj_exts; struct slabobj_ext *obj_ext; - unsigned int off; + struct obj_cgroup *objcg; obj_exts = slab_obj_exts(slab); if (!obj_exts) return NULL; - get_slab_obj_exts(obj_exts); - off = obj_to_index(slab->slab_cache, slab, p); - obj_ext = slab_obj_ext(slab, obj_exts, off); - if (obj_ext->objcg) { - struct obj_cgroup *objcg = obj_ext->objcg; + if (!slab_needs_objcg(slab)) + return NULL; + get_slab_obj_exts(obj_exts); + obj_ext = slab_obj_ext(slab->slab_cache, slab, obj_exts, p); + objcg = slab_obj_ext_objcg(slab, obj_ext); + if (objcg) { put_slab_obj_exts(obj_exts); return obj_cgroup_memcg(objcg); } @@ -3543,7 +3544,6 @@ bool __memcg_slab_post_alloc_hook(struct kmem_cache *s, struct list_lru *lru, size_t obj_size = obj_full_size(s); struct obj_cgroup *objcg; struct slab *slab; - unsigned long off; size_t i; /* @@ -3585,9 +3585,11 @@ bool __memcg_slab_post_alloc_hook(struct kmem_cache *s, struct list_lru *lru, slab = virt_to_slab(p[i]); - if (!slab_obj_exts(slab) && - alloc_slab_obj_exts(slab, s, flags, slab_alloc_flags)) { - continue; + if (!slab_obj_exts(slab)) { + if (is_kfence_address(p[i])) + continue; + if (alloc_slab_obj_exts(slab, s, flags, slab_alloc_flags)) + continue; } /* @@ -3618,10 +3620,11 @@ bool __memcg_slab_post_alloc_hook(struct kmem_cache *s, struct list_lru *lru, obj_exts = slab_obj_exts(slab); get_slab_obj_exts(obj_exts); - off = obj_to_index(s, slab, p[i]); - obj_ext = slab_obj_ext(slab, obj_exts, off); + obj_ext = slab_obj_ext(s, slab, obj_exts, p[i]); + obj_cgroup_get(objcg); - obj_ext->objcg = objcg; + slab_obj_ext_set_objcg(slab, obj_ext, objcg); + put_slab_obj_exts(obj_exts); } @@ -3637,15 +3640,13 @@ void __memcg_slab_free_hook(struct kmem_cache *s, struct slab *slab, struct obj_cgroup *objcg; struct slabobj_ext *obj_ext; struct obj_stock_pcp *stock; - unsigned int off; - off = obj_to_index(s, slab, p[i]); - obj_ext = slab_obj_ext(slab, obj_exts, off); - objcg = obj_ext->objcg; + obj_ext = slab_obj_ext(s, slab, obj_exts, p[i]); + objcg = slab_obj_ext_objcg(slab, obj_ext); if (!objcg) continue; - obj_ext->objcg = NULL; + slab_obj_ext_set_objcg(slab, obj_ext, NULL); stock = trylock_stock(); __refill_obj_stock(objcg, stock, obj_size, true); diff --git a/mm/mempool.c b/mm/mempool.c index 473a029fa31f..cb74e718b2c6 100644 --- a/mm/mempool.c +++ b/mm/mempool.c @@ -16,11 +16,28 @@ #include #include #include +#include +#include #include "slab.h" static DECLARE_FAULT_ATTR(fail_mempool_alloc); static DECLARE_FAULT_ATTR(fail_mempool_alloc_bulk); +/* + * Debugging support for mempool using static key. + * + * This allows enabling mempool debug at boot time via: + * mempool_debug + */ +static DEFINE_STATIC_KEY_FALSE(mempool_debug_enabled); + +static int __init mempool_debug_setup(char *str) +{ + static_branch_enable(&mempool_debug_enabled); + return 1; +} +__setup("mempool_debug", mempool_debug_setup); + static int __init mempool_faul_inject_init(void) { int error; @@ -37,7 +54,6 @@ static int __init mempool_faul_inject_init(void) } late_initcall(mempool_faul_inject_init); -#ifdef CONFIG_SLUB_DEBUG_ON static void poison_error(struct mempool *pool, void *element, size_t size, size_t byte) { @@ -140,14 +156,6 @@ static void poison_element(struct mempool *pool, void *element) #endif } } -#else /* CONFIG_SLUB_DEBUG_ON */ -static inline void check_element(struct mempool *pool, void *element) -{ -} -static inline void poison_element(struct mempool *pool, void *element) -{ -} -#endif /* CONFIG_SLUB_DEBUG_ON */ static __always_inline bool kasan_poison_element(struct mempool *pool, void *element) @@ -175,7 +183,10 @@ static void kasan_unpoison_element(struct mempool *pool, void *element) static __always_inline void add_element(struct mempool *pool, void *element) { BUG_ON(pool->min_nr != 0 && pool->curr_nr >= pool->min_nr); - poison_element(pool, element); + + if (static_branch_unlikely(&mempool_debug_enabled)) + poison_element(pool, element); + if (kasan_poison_element(pool, element)) pool->elements[pool->curr_nr++] = element; } @@ -186,7 +197,9 @@ static void *remove_element(struct mempool *pool) BUG_ON(pool->curr_nr < 0); kasan_unpoison_element(pool, element); - check_element(pool, element); + + if (static_branch_unlikely(&mempool_debug_enabled)) + check_element(pool, element); return element; } diff --git a/mm/slab.h b/mm/slab.h index c24c3daaa869..8fd6835e4235 100644 --- a/mm/slab.h +++ b/mm/slab.h @@ -24,11 +24,27 @@ #define SLAB_ALLOC_NO_RECURSE 0x04 /* prevent kmalloc() recursion */ #define SLAB_ALLOC_NO_OBJ_EXT 0x08 /* prevent obj_exts array allocation */ +#define SLAB_FREE_DEFAULT 0x00 /* no flags */ +#define SLAB_FREE_NOLOCK 0x01 /* spinning not allowed */ + +static inline unsigned int to_alloc_flags(unsigned int free_flags) +{ + if (free_flags & SLAB_FREE_NOLOCK) + return SLAB_ALLOC_NOLOCK; + else + return SLAB_ALLOC_DEFAULT; +} + static inline bool alloc_flags_allow_spinning(const unsigned int alloc_flags) { return !(alloc_flags & SLAB_ALLOC_NOLOCK); } +static inline bool free_flags_allow_spinning(const unsigned int free_flags) +{ + return !(free_flags & SLAB_FREE_NOLOCK); +} + void *__kmalloc_flags_noprof(DECL_TOKEN_PARAMS(size, token), gfp_t flags, unsigned int alloc_flags, int node) __assume_kmalloc_alignment __alloc_size(1); @@ -81,10 +97,11 @@ struct freelist_counters { #ifdef CONFIG_64BIT /* * Some optimizations use free bits in 'counters' field - * to save memory. In case ->stride field is not available, - * such optimizations are disabled. + * to save memory or CPU. If these free bits are not + * available, such optimizations are disabled. */ - unsigned int stride; + unsigned obj_exts_in_object:1; + unsigned obj_exts_needs_objcg:1; #endif }; }; @@ -330,10 +347,35 @@ static inline unsigned int obj_to_index(const struct kmem_cache *cache, return __obj_to_index(cache, slab_address(slab), obj); } -static inline int objs_per_slab(const struct kmem_cache *cache, - const struct slab *slab) +/* + * kvfree_rcu_head offset can be only less than page size. + * Calculate the start address while preserving the KASAN tag. + */ +static inline void *kvmalloc_obj_start_addr(void *head) { - return slab->objects; + unsigned long offset; + + if (unlikely(is_vmalloc_addr(head))) { + offset = offset_in_page(head); + } else { + struct slab *slab = virt_to_slab(head); + + if (!slab) { + offset = offset_in_page(head); + } else if (is_kfence_address(head)) { + offset = head - kfence_object_start(head); + } else { + struct kmem_cache *s = slab->slab_cache; + unsigned int idx = __obj_to_index(s, slab_address(slab), head); + void *obj = slab_address(slab) + s->size * idx; + + obj = fixup_red_left(s, obj); + obj = kasan_reset_tag(obj); + offset = kasan_reset_tag(head) - obj; + } + } + + return head - offset; } /* @@ -436,7 +478,7 @@ static inline bool is_kmalloc_normal(struct kmem_cache *s) return !(s->flags & (SLAB_CACHE_DMA|SLAB_ACCOUNT|SLAB_RECLAIM_ACCOUNT|SLAB_NO_OBJ_EXT)); } -bool __kfree_rcu_sheaf(struct kmem_cache *s, void *obj); +bool __kfree_rcu_sheaf(struct kmem_cache *s, void *obj, unsigned int free_flags); void flush_all_rcu_sheaves(void); void flush_rcu_sheaves_on_cache(struct kmem_cache *s); @@ -555,6 +597,93 @@ static inline bool need_kmalloc_no_objext(void) return false; } +/* + * Extended information for slab objects stored as a pointer to an array in + * slab->obj_exts (aliasing page->memcg_data) if MEMCG_DATA_OBJEXTS is set. + */ +struct slabobj_ext { + /* + * All elements of the union should be pointer-sized to avoid memory + * waste + */ + union { +#ifdef CONFIG_MEMCG + struct obj_cgroup *_objcg; +#endif +#ifdef CONFIG_MEM_ALLOC_PROFILING + union codetag_ref _ctref; +#endif + }; +} __aligned(8); + +#ifdef CONFIG_MEM_ALLOC_PROFILING +DECLARE_STATIC_KEY_MAYBE(CONFIG_MEM_ALLOC_PROFILING_ENABLED_BY_DEFAULT, + slab_obj_ext_has_codetag_key); + +static inline bool slab_obj_ext_has_codetag(void) +{ + return static_branch_maybe(CONFIG_MEM_ALLOC_PROFILING_ENABLED_BY_DEFAULT, + &slab_obj_ext_has_codetag_key); +} +#else +static inline bool slab_obj_ext_has_codetag(void) +{ + return false; +} +#endif + +#ifdef CONFIG_MEMCG +static inline bool cache_needs_objcg(struct kmem_cache *cache) +{ + return (cache->flags & SLAB_MAY_ACCOUNT); +} + +static inline bool slab_needs_objcg(struct slab *slab) +{ +#ifdef CONFIG_64BIT + return slab->obj_exts_needs_objcg; +#else + return cache_needs_objcg(slab->slab_cache); +#endif +} +#else +static inline bool cache_needs_objcg(struct kmem_cache *cache) +{ + return false; +} + +static inline bool slab_needs_objcg(struct slab *slab) +{ + return false; +} +#endif + +static inline size_t cache_obj_ext_size(struct kmem_cache *s) +{ + size_t sz = 0; + + if (cache_needs_objcg(s)) + sz += 1; + + if (slab_obj_ext_has_codetag()) + sz += 1; + + return sizeof(struct slabobj_ext) * sz; +} + +static inline size_t slab_obj_ext_size(struct slab *slab) +{ + size_t sz = 0; + + if (slab_needs_objcg(slab)) + sz += 1; + + if (slab_obj_ext_has_codetag()) + sz += 1; + + return sizeof(struct slabobj_ext) * sz; +} + #ifdef CONFIG_SLAB_OBJ_EXT /* @@ -572,7 +701,7 @@ static inline bool need_kmalloc_no_objext(void) * obj_exts = slab_obj_exts(slab); * if (obj_exts) { * get_slab_obj_exts(obj_exts); - * obj_ext = slab_obj_ext(slab, obj_exts, obj_to_index(s, slab, obj)); + * obj_ext = slab_obj_ext(s, slab, obj_exts, obj); * // do something with obj_ext * put_slab_obj_exts(obj_exts); * } @@ -610,48 +739,94 @@ static inline void put_slab_obj_exts(unsigned long obj_exts) } #ifdef CONFIG_64BIT -static inline void slab_set_stride(struct slab *slab, unsigned int stride) +static inline bool obj_exts_in_object(struct slab *slab) { - slab->stride = stride; -} -static inline unsigned int slab_get_stride(struct slab *slab) -{ - return slab->stride; + /* + * Note we cannot rely on the SLAB_OBJ_EXT_IN_OBJ flag here and need to + * check the per-slab bit. A cache can have SLAB_OBJ_EXT_IN_OBJ set, but + * allocations within_slab_leftover are preferred. And those may be + * possible or not depending on the particular slab's size. + */ + return slab->obj_exts_in_object; } #else -static inline void slab_set_stride(struct slab *slab, unsigned int stride) +static inline bool obj_exts_in_object(struct slab *slab) { - VM_WARN_ON_ONCE(stride != sizeof(struct slabobj_ext)); -} -static inline unsigned int slab_get_stride(struct slab *slab) -{ - return sizeof(struct slabobj_ext); + return false; } #endif /* * slab_obj_ext - get the pointer to the slab object extension metadata * associated with an object in a slab. + * @s: cache that the slab belongs to * @slab: a pointer to the slab struct * @obj_exts: a pointer to the object extension vector - * @index: an index of the object + * @obj: a pointer to the object * * Returns a pointer to the object extension associated with the object. * Must be called within a section covered by get/put_slab_obj_exts(). */ -static inline struct slabobj_ext *slab_obj_ext(struct slab *slab, - unsigned long obj_exts, - unsigned int index) +static inline struct slabobj_ext * +slab_obj_ext(struct kmem_cache *s, struct slab *slab, unsigned long obj_exts, + const void *obj) { struct slabobj_ext *obj_ext; + unsigned int index; + unsigned int stride; VM_WARN_ON_ONCE(obj_exts != slab_obj_exts(slab)); - obj_ext = (struct slabobj_ext *)(obj_exts + - slab_get_stride(slab) * index); + /* + * KFENCE objects have NULL obj_exts and thus can't reach this + * and we don't need obj_to_index() + */ + index = __obj_to_index(s, slab_address(slab), obj); + + if (!obj_exts_in_object(slab)) + stride = slab_obj_ext_size(slab); + else + stride = s->size; + + obj_ext = (struct slabobj_ext *)(obj_exts + index * stride); + return kasan_reset_tag(obj_ext); } +#ifdef CONFIG_MEMCG +static inline struct obj_cgroup * +slab_obj_ext_objcg(struct slab *slab, struct slabobj_ext *obj_ext) +{ + VM_WARN_ON_ONCE(!slab_needs_objcg(slab)); + + /* if objcg exists, it comes first, so we don't need to do anything */ + return obj_ext->_objcg; +} + +static inline void +slab_obj_ext_set_objcg(struct slab *slab, struct slabobj_ext *obj_ext, + struct obj_cgroup *objcg) +{ + VM_WARN_ON_ONCE(!slab_needs_objcg(slab)); + + /* if objcg exists, it comes first, so we don't need to do anything */ + obj_ext->_objcg = objcg; +} +#endif + +#ifdef CONFIG_MEM_ALLOC_PROFILING +static inline union codetag_ref * +slab_obj_ext_codetag_ref(struct slab *slab, struct slabobj_ext *obj_ext) +{ + VM_WARN_ON_ONCE(!slab_obj_ext_has_codetag()); + + if (slab_needs_objcg(slab)) + obj_ext += 1; + + return &obj_ext->_ctref; +} +#endif + int alloc_slab_obj_exts(struct slab *slab, struct kmem_cache *s, gfp_t gfp, unsigned int alloc_flags); @@ -662,16 +837,17 @@ static inline unsigned long slab_obj_exts(struct slab *slab) return 0; } -static inline struct slabobj_ext *slab_obj_ext(struct slab *slab, - unsigned long obj_exts, - unsigned int index) +static inline struct slabobj_ext * +slab_obj_ext(struct kmem_cache *s, struct slab *slab, unsigned long obj_exts, + const void *obj) { return NULL; } -static inline void slab_set_stride(struct slab *slab, unsigned int stride) { } -static inline unsigned int slab_get_stride(struct slab *slab) { return 0; } - +static inline bool obj_exts_in_object(struct slab *slab) +{ + return false; +} #endif /* CONFIG_SLAB_OBJ_EXT */ @@ -770,7 +946,8 @@ void __kmem_obj_info(struct kmem_obj_info *kpp, void *object, struct slab *slab) void __check_heap_object(const void *ptr, unsigned long n, const struct slab *slab, bool to_user); -void defer_free_barrier(void); +void deferred_work_barrier(void); +void defer_kfree_rcu(struct kvfree_rcu_head *head); static inline bool slub_debug_orig_size(struct kmem_cache *s) { diff --git a/mm/slab_common.c b/mm/slab_common.c index 657fd75776ea..b19ba1b31484 100644 --- a/mm/slab_common.c +++ b/mm/slab_common.c @@ -52,7 +52,7 @@ struct kmem_cache *kmem_cache; SLAB_OBJ_EXT_IN_OBJ) #define SLAB_MERGE_SAME (SLAB_RECLAIM_ACCOUNT | SLAB_CACHE_DMA | \ - SLAB_CACHE_DMA32 | SLAB_ACCOUNT) + SLAB_CACHE_DMA32 | SLAB_ACCOUNT | SLAB_MAY_ACCOUNT) /* * Merge control. If this is set then no merging of slab caches will occur. @@ -359,6 +359,13 @@ struct kmem_cache *__kmem_cache_create_args(const char *name, goto out_unlock; } + /* + * For now we assume any cache can be used with __GFP_ACCOUNT and thus + * may need to store objcg pointers for objects + */ + if (!mem_cgroup_kmem_disabled()) + flags |= SLAB_MAY_ACCOUNT; + /* Fail closed on bad usersize of useroffset values. */ if (!IS_ENABLED(CONFIG_HARDENED_USERCOPY) || WARN_ON(!args->usersize && args->useroffset) || @@ -551,7 +558,7 @@ void kmem_cache_destroy(struct kmem_cache *s) } /* Wait for deferred work from kmalloc/kfree_nolock() */ - defer_free_barrier(); + deferred_work_barrier(); cpus_read_lock(); mutex_lock(&slab_mutex); @@ -984,11 +991,20 @@ new_kmalloc_cache(int idx, enum kmalloc_cache_type type) #endif /* - * If CONFIG_MEMCG is enabled, disable cache merging for - * KMALLOC_NORMAL caches. + * If memcg_kmem is enabled and this is a KMALLOC_NORMAL cache and not + * aliased with any other type, make sure it's never merged with any other + * cache. + * + * In other cases the kmalloc cache may end up being used for a + * __GFP_ACCOUNT allocation so mark it as such. The exception is a + * KMALLOC_NO_OBJ_EXT cache. */ - if (IS_ENABLED(CONFIG_MEMCG) && (type == KMALLOC_NORMAL)) - flags |= SLAB_NO_MERGE; + if (!mem_cgroup_kmem_disabled()) { + if (type == KMALLOC_NORMAL && KMALLOC_RECLAIM != KMALLOC_NORMAL) + flags |= SLAB_NO_MERGE; + else if (!(flags & SLAB_NO_OBJ_EXT)) + flags |= SLAB_MAY_ACCOUNT; + } if (minalign > ARCH_KMALLOC_MINALIGN) { aligned_size = ALIGN(aligned_size, minalign); @@ -1280,13 +1296,40 @@ EXPORT_TRACEPOINT_SYMBOL(kmem_cache_alloc); EXPORT_TRACEPOINT_SYMBOL(kfree); EXPORT_TRACEPOINT_SYMBOL(kmem_cache_free); +void kfree_call_rcu_nolock(struct kvfree_rcu_head *head, void *ptr) +{ + struct slab *slab; + + if (!IS_ENABLED(CONFIG_KVFREE_RCU_BATCHED)) + goto fallback; + + if (unlikely(is_vmalloc_addr(ptr))) + goto fallback; + + slab = virt_to_slab(ptr); + if (unlikely(!slab)) + goto fallback; + + if (unlikely(IS_ENABLED(CONFIG_NUMA) && slab_nid(slab) != numa_mem_id())) + goto fallback; + + if (unlikely(!__kfree_rcu_sheaf(slab->slab_cache, ptr, SLAB_FREE_NOLOCK))) + goto fallback; + + return; + +fallback: + defer_kfree_rcu(head); +} +EXPORT_SYMBOL_GPL(kfree_call_rcu_nolock); + #ifndef CONFIG_KVFREE_RCU_BATCHED -void kvfree_call_rcu(struct rcu_head *head, void *ptr) +void kvfree_call_rcu(struct kvfree_rcu_head *head, void *ptr) { if (head) { kasan_record_aux_stack(ptr); - call_rcu(head, kvfree_rcu_cb); + call_rcu(&head->head, kvfree_rcu_cb); return; } @@ -1297,6 +1340,18 @@ void kvfree_call_rcu(struct rcu_head *head, void *ptr) } EXPORT_SYMBOL_GPL(kvfree_call_rcu); +void kvfree_rcu_barrier(void) +{ + deferred_work_barrier(); + rcu_barrier(); +} + +void kvfree_rcu_barrier_on_cache(struct kmem_cache *s) +{ + deferred_work_barrier(); + rcu_barrier(); +} + void __init kvfree_rcu_init(void) { } @@ -1363,7 +1418,7 @@ struct kvfree_rcu_bulk_data { struct kfree_rcu_cpu_work { struct rcu_work rcu_work; - struct rcu_head *head_free; + struct kvfree_rcu_head *head_free; struct rcu_gp_seq head_free_gp_snap; struct list_head bulk_head_free[FREE_N_CHANNELS]; struct kfree_rcu_cpu *krcp; @@ -1399,7 +1454,7 @@ struct kfree_rcu_cpu_work { struct kfree_rcu_cpu { // Objects queued on a linked list // through their rcu_head structures. - struct rcu_head *head; + struct kvfree_rcu_head *head; unsigned long head_gp_snap; atomic_t head_count; @@ -1540,12 +1595,12 @@ kvfree_rcu_bulk(struct kfree_rcu_cpu *krcp, } static void -kvfree_rcu_list(struct rcu_head *head) +kvfree_rcu_list(struct kvfree_rcu_head *head) { - struct rcu_head *next; + struct kvfree_rcu_head *next; for (; head; head = next) { - void *ptr = (void *) head->func; + void *ptr = kvmalloc_obj_start_addr(head); unsigned long offset = (void *) head - ptr; next = head->next; @@ -1569,7 +1624,7 @@ static void kfree_rcu_work(struct work_struct *work) unsigned long flags; struct kvfree_rcu_bulk_data *bnode, *n; struct list_head bulk_head[FREE_N_CHANNELS]; - struct rcu_head *head; + struct kvfree_rcu_head *head; struct kfree_rcu_cpu *krcp; struct kfree_rcu_cpu_work *krwp; struct rcu_gp_seq head_gp_snap; @@ -1612,6 +1667,14 @@ static bool kfree_rcu_sheaf(void *obj) { struct kmem_cache *s; struct slab *slab; + unsigned int free_flags = SLAB_FREE_DEFAULT; + + /* + * It is not safe to spin on PREEMPT_RT because the kernel might be + * holding a raw spinlock and slab acquires sleeping locks. + */ + if (IS_ENABLED(CONFIG_PREEMPT_RT)) + free_flags = SLAB_FREE_NOLOCK; if (is_vmalloc_addr(obj)) return false; @@ -1622,7 +1685,7 @@ static bool kfree_rcu_sheaf(void *obj) s = slab->slab_cache; if (likely(!IS_ENABLED(CONFIG_NUMA) || slab_nid(slab) == numa_mem_id())) - return __kfree_rcu_sheaf(s, obj); + return __kfree_rcu_sheaf(s, obj, free_flags); return false; } @@ -1692,7 +1755,7 @@ kvfree_rcu_drain_ready(struct kfree_rcu_cpu *krcp) { struct list_head bulk_ready[FREE_N_CHANNELS]; struct kvfree_rcu_bulk_data *bnode, *n; - struct rcu_head *head_ready = NULL; + struct kvfree_rcu_head *head_ready = NULL; unsigned long flags; int i; @@ -1955,7 +2018,7 @@ void __init kfree_rcu_scheduler_running(void) * be free'd in workqueue context. This allows us to: batch requests together to * reduce the number of grace periods during heavy kfree_rcu()/kvfree_rcu() load. */ -void kvfree_call_rcu(struct rcu_head *head, void *ptr) +void kvfree_call_rcu(struct kvfree_rcu_head *head, void *ptr) { unsigned long flags; struct kfree_rcu_cpu *krcp; @@ -1971,7 +2034,7 @@ void kvfree_call_rcu(struct rcu_head *head, void *ptr) if (!head) might_sleep(); - if (!IS_ENABLED(CONFIG_PREEMPT_RT) && kfree_rcu_sheaf(ptr)) + if (kfree_rcu_sheaf(ptr)) return; // Queue the object but don't yet schedule the batch. @@ -1993,7 +2056,6 @@ void kvfree_call_rcu(struct rcu_head *head, void *ptr) // Inline if kvfree_rcu(one_arg) call. goto unlock_return; - head->func = ptr; head->next = krcp->head; WRITE_ONCE(krcp->head, head); atomic_inc(&krcp->head_count); @@ -2115,7 +2177,6 @@ void kvfree_rcu_barrier(void) flush_all_rcu_sheaves(); __kvfree_rcu_barrier(); } -EXPORT_SYMBOL_GPL(kvfree_rcu_barrier); /** * kvfree_rcu_barrier_on_cache - Wait for in-flight kvfree_rcu() calls on a @@ -2126,20 +2187,18 @@ EXPORT_SYMBOL_GPL(kvfree_rcu_barrier); */ void kvfree_rcu_barrier_on_cache(struct kmem_cache *s) { + /* kfree_rcu_nolock() might have deferred frees even without sheaves */ + deferred_work_barrier(); + if (cache_has_sheaves(s)) { cpus_read_lock(); flush_rcu_sheaves_on_cache(s); cpus_read_unlock(); - rcu_barrier(); } - /* - * TODO: Introduce a version of __kvfree_rcu_barrier() that works - * on a specific slab cache. - */ + rcu_barrier(); __kvfree_rcu_barrier(); } -EXPORT_SYMBOL_GPL(kvfree_rcu_barrier_on_cache); static unsigned long kfree_rcu_shrink_count(struct shrinker *shrink, struct shrink_control *sc) diff --git a/mm/slub.c b/mm/slub.c index 422bc3e12c02..f9b56cb439e7 100644 --- a/mm/slub.c +++ b/mm/slub.c @@ -214,6 +214,11 @@ DEFINE_STATIC_KEY_FALSE(slub_debug_enabled); static DEFINE_STATIC_KEY_FALSE(strict_numa); #endif +#ifdef CONFIG_MEM_ALLOC_PROFILING +DEFINE_STATIC_KEY_MAYBE(CONFIG_MEM_ALLOC_PROFILING_ENABLED_BY_DEFAULT, + slab_obj_ext_has_codetag_key); +#endif + /* Structure holding extra parameters for slab allocations */ struct slab_alloc_context { unsigned long caller_addr; @@ -334,14 +339,20 @@ enum track_item { TRACK_ALLOC, TRACK_FREE }; #ifdef SLAB_SUPPORTS_SYSFS static int sysfs_slab_add(struct kmem_cache *); +static int __init slab_kset_init(void); +static void __init slab_sysfs_process_aliases(void); #else static inline int sysfs_slab_add(struct kmem_cache *s) { return 0; } +static inline int slab_kset_init(void) { return 0; } +static inline void slab_sysfs_process_aliases(void) { } #endif #if defined(CONFIG_DEBUG_FS) && defined(CONFIG_SLUB_DEBUG) static void debugfs_slab_add(struct kmem_cache *); +static void __init slab_debugfs_root_init(void); #else static inline void debugfs_slab_add(struct kmem_cache *s) { } +static inline void slab_debugfs_root_init(void) { } #endif enum add_mode { @@ -419,6 +430,8 @@ struct slab_sheaf { union { struct rcu_head rcu_head; struct list_head barn_list; + /* only used to defer call_rcu() in unknown context */ + struct llist_node llnode; /* only used for prefilled sheafs */ struct { unsigned int capacity; @@ -804,7 +817,7 @@ static inline bool need_slab_obj_exts(struct kmem_cache *s) static inline unsigned int obj_exts_size_in_slab(struct slab *slab) { - return sizeof(struct slabobj_ext) * slab->objects; + return slab_obj_ext_size(slab) * slab->objects; } static inline unsigned long obj_exts_offset_in_slab(struct kmem_cache *s, @@ -871,18 +884,6 @@ static inline bool obj_exts_in_slab(struct kmem_cache *s, struct slab *slab) #endif #if defined(CONFIG_SLAB_OBJ_EXT) && defined(CONFIG_64BIT) -static bool obj_exts_in_object(struct kmem_cache *s, struct slab *slab) -{ - /* - * Note we cannot rely on the SLAB_OBJ_EXT_IN_OBJ flag here and need to - * check the stride. A cache can have SLAB_OBJ_EXT_IN_OBJ set, but - * allocations within_slab_leftover are preferred. And those may be - * possible or not depending on the particular slab's size. - */ - return obj_exts_in_slab(s, slab) && - (slab_get_stride(slab) == s->size); -} - static unsigned int obj_exts_offset_in_object(struct kmem_cache *s) { unsigned int offset = get_info_end(s); @@ -897,18 +898,40 @@ static unsigned int obj_exts_offset_in_object(struct kmem_cache *s) return offset; } -#else -static inline bool obj_exts_in_object(struct kmem_cache *s, struct slab *slab) -{ - return false; -} +static inline void slab_set_obj_exts_in_object(struct slab *slab) +{ + slab->obj_exts_in_object = 1; +} +#else static inline unsigned int obj_exts_offset_in_object(struct kmem_cache *s) { return 0; } + +static inline void slab_set_obj_exts_in_object(struct slab *slab) +{ +} #endif +/* + * A no-op function used to attach kprobe handlers in slub_kunit tests. + * The barrier is needed to prevent the compiler from optimizing out callsites. + */ +#if defined(CONFIG_DEBUG_VM) || defined(CONFIG_PROVE_LOCKING) +static noinline void slab_attach_kprobe_locked(void) +{ + barrier(); +} +#else +static inline void slab_attach_kprobe_locked(void) { } +#endif + +#define slab_lockdep_assert_held(lock) do { \ + lockdep_assert_held(lock); \ + slab_attach_kprobe_locked(); \ +} while (0) + #ifdef CONFIG_SLUB_DEBUG /* @@ -1207,8 +1230,8 @@ static void print_trailer(struct kmem_cache *s, struct slab *slab, u8 *p) off += kasan_metadata_size(s, false); - if (obj_exts_in_object(s, slab)) - off += sizeof(struct slabobj_ext); + if (obj_exts_in_object(slab)) + off += slab_obj_ext_size(slab); if (off != size_from_object(s)) /* Beginning of the filler is the free pointer */ @@ -1412,8 +1435,8 @@ static int check_pad_bytes(struct kmem_cache *s, struct slab *slab, u8 *p) off += kasan_metadata_size(s, false); - if (obj_exts_in_object(s, slab)) - off += sizeof(struct slabobj_ext); + if (obj_exts_in_object(slab)) + off += slab_obj_ext_size(slab); if (size_from_object(s) == off) return 1; @@ -1440,7 +1463,7 @@ slab_pad_check(struct kmem_cache *s, struct slab *slab) length = slab_size(slab); end = start + length; - if (obj_exts_in_slab(s, slab) && !obj_exts_in_object(s, slab)) { + if (obj_exts_in_slab(s, slab) && !obj_exts_in_object(slab)) { remainder = length; remainder -= obj_exts_offset_in_slab(s, slab); remainder -= obj_exts_size_in_slab(slab); @@ -1666,7 +1689,7 @@ static void add_full(struct kmem_cache *s, if (!(s->flags & SLAB_STORE_USER)) return; - lockdep_assert_held(&n->list_lock); + slab_lockdep_assert_held(&n->list_lock); list_add(&slab->slab_list, &n->full); } @@ -1675,7 +1698,7 @@ static void remove_full(struct kmem_cache *s, struct kmem_cache_node *n, struct if (!(s->flags & SLAB_STORE_USER)) return; - lockdep_assert_held(&n->list_lock); + slab_lockdep_assert_held(&n->list_lock); list_del(&slab->slab_list); } @@ -2068,23 +2091,27 @@ static inline void mark_obj_codetag_empty(const void *obj) struct slab *obj_slab; unsigned long slab_exts; + if (!slab_obj_ext_has_codetag()) + return; + obj_slab = virt_to_slab(obj); slab_exts = slab_obj_exts(obj_slab); if (slab_exts) { - get_slab_obj_exts(slab_exts); - unsigned int offs = obj_to_index(obj_slab->slab_cache, - obj_slab, obj); - struct slabobj_ext *ext = slab_obj_ext(obj_slab, - slab_exts, offs); + struct slabobj_ext *ext; + union codetag_ref *ref; - if (unlikely(is_codetag_empty(&ext->ref))) { + get_slab_obj_exts(slab_exts); + ext = slab_obj_ext(obj_slab->slab_cache, obj_slab, slab_exts, obj); + ref = slab_obj_ext_codetag_ref(obj_slab, ext); + + if (unlikely(is_codetag_empty(ref))) { put_slab_obj_exts(slab_exts); return; } /* codetag should be NULL here */ - WARN_ON(ext->ref.ct); - set_codetag_empty(&ext->ref); + WARN_ON(ref->ct); + set_codetag_empty(ref); put_slab_obj_exts(slab_exts); } } @@ -2094,19 +2121,29 @@ static inline bool mark_failed_objexts_alloc(struct slab *slab) return cmpxchg(&slab->obj_exts, 0, OBJEXTS_ALLOC_FAIL) == 0; } -static inline void handle_failed_objexts_alloc(unsigned long obj_exts, - struct slabobj_ext *vec, unsigned int objects) +static inline void handle_failed_objexts_alloc(struct slab *slab, + unsigned long obj_exts, struct slabobj_ext *vec) { + unsigned int stride; + + if (!slab_obj_ext_has_codetag()) + return; + /* * If vector previously failed to allocate then we have live * objects with no tag reference. Mark all references in this * vector as empty to avoid warnings later on. */ - if (obj_exts == OBJEXTS_ALLOC_FAIL) { - unsigned int i; + if (obj_exts != OBJEXTS_ALLOC_FAIL) + return; - for (i = 0; i < objects; i++) - set_codetag_empty(&vec[i].ref); + stride = slab_obj_ext_size(slab) / sizeof(*vec); + + for (unsigned int i = 0; i < slab->objects; i++) { + union codetag_ref *ref = slab_obj_ext_codetag_ref(slab, vec); + + set_codetag_empty(ref); + vec += stride; } } @@ -2114,8 +2151,8 @@ static inline void handle_failed_objexts_alloc(unsigned long obj_exts, static inline void mark_obj_codetag_empty(const void *obj) {} static inline bool mark_failed_objexts_alloc(struct slab *slab) { return false; } -static inline void handle_failed_objexts_alloc(unsigned long obj_exts, - struct slabobj_ext *vec, unsigned int objects) {} +static inline void handle_failed_objexts_alloc(struct slab *slab, + unsigned long obj_exts, struct slabobj_ext *vec) {} #endif /* CONFIG_MEM_ALLOC_PROFILING_DEBUG */ @@ -2128,12 +2165,11 @@ int alloc_slab_obj_exts(struct slab *slab, struct kmem_cache *s, gfp_t gfp, unsigned int alloc_flags) { const bool allow_spin = alloc_flags_allow_spinning(alloc_flags); - unsigned int objects = objs_per_slab(s, slab); bool new_slab = alloc_flags & SLAB_ALLOC_NEW_SLAB; unsigned long new_exts; unsigned long old_exts; struct slabobj_ext *vec; - size_t sz = sizeof(struct slabobj_ext) * slab->objects; + size_t sz = slab_obj_ext_size(slab) * slab->objects; gfp &= ~OBJCGS_CLEAR_MASK; /* @@ -2184,7 +2220,7 @@ int alloc_slab_obj_exts(struct slab *slab, struct kmem_cache *s, #endif retry: old_exts = READ_ONCE(slab->obj_exts); - handle_failed_objexts_alloc(old_exts, vec, objects); + handle_failed_objexts_alloc(slab, old_exts, vec); if (new_slab) { /* @@ -2251,9 +2287,6 @@ static void alloc_slab_obj_exts_early(struct kmem_cache *s, struct slab *slab) void *addr; unsigned long obj_exts; - /* Initialize stride early to avoid memory ordering issues */ - slab_set_stride(slab, sizeof(struct slabobj_ext)); - if (!need_slab_obj_exts(s)) return; @@ -2279,15 +2312,14 @@ static void alloc_slab_obj_exts_early(struct kmem_cache *s, struct slab *slab) get_slab_obj_exts(obj_exts); for_each_object(addr, s, slab_address(slab), slab->objects) - memset(kasan_reset_tag(addr) + offset, 0, - sizeof(struct slabobj_ext)); + memset(kasan_reset_tag(addr) + offset, 0, slab_obj_ext_size(slab)); put_slab_obj_exts(obj_exts); #ifdef CONFIG_MEMCG obj_exts |= MEMCG_DATA_OBJEXTS; #endif slab->obj_exts = obj_exts; - slab_set_stride(slab, s->size); + slab_set_obj_exts_in_object(slab); } } @@ -2324,11 +2356,15 @@ static inline unsigned long prepare_slab_obj_exts_hook(struct kmem_cache *s, struct slab *slab, gfp_t flags, unsigned int alloc_flags, void *p) { - if (!slab_obj_exts(slab) && - alloc_slab_obj_exts(slab, s, flags, alloc_flags)) { - pr_warn_once("%s, %s: Failed to create slab extension vector!\n", - __func__, s->name); - return 0; + if (!slab_obj_exts(slab)) { + if (is_kfence_address(p)) + return 0; + + if (alloc_slab_obj_exts(slab, s, flags, alloc_flags)) { + pr_warn_once("%s, %s: Failed to create slab extension vector!\n", + __func__, s->name); + return 0; + } } return slab_obj_exts(slab); @@ -2361,14 +2397,24 @@ __alloc_tagging_slab_alloc_hook(struct kmem_cache *s, void *object, gfp_t flags, * check should be added before alloc_tag_add(). */ if (obj_exts) { - unsigned int obj_idx = obj_to_index(s, slab, object); + union codetag_ref *ref; get_slab_obj_exts(obj_exts); - obj_ext = slab_obj_ext(slab, obj_exts, obj_idx); - alloc_tag_add(&obj_ext->ref, current->alloc_tag, s->size); + + obj_ext = slab_obj_ext(s, slab, obj_exts, object); + ref = slab_obj_ext_codetag_ref(slab, obj_ext); + + alloc_tag_add(ref, current->alloc_tag, s->size); + put_slab_obj_exts(obj_exts); } else { - alloc_tag_set_inaccurate(current->alloc_tag); + /* + * KFENCE allocations are rare and the amount of outstanding + * ones is limited to a small number so it's not worth setting + * tags as inaccurate because of them. + */ + if (!is_kfence_address(object)) + alloc_tag_set_inaccurate(current->alloc_tag); } } @@ -2385,7 +2431,6 @@ static noinline void __alloc_tagging_slab_free_hook(struct kmem_cache *s, struct slab *slab, void **p, int objects) { - int i; unsigned long obj_exts; /* slab->obj_exts might not be NULL if it was created for MEMCG accounting. */ @@ -2397,10 +2442,11 @@ __alloc_tagging_slab_free_hook(struct kmem_cache *s, struct slab *slab, void **p return; get_slab_obj_exts(obj_exts); - for (i = 0; i < objects; i++) { - unsigned int off = obj_to_index(s, slab, p[i]); + for (int i = 0; i < objects; i++) { + struct slabobj_ext *ext; - alloc_tag_sub(&slab_obj_ext(slab, obj_exts, off)->ref, s->size); + ext = slab_obj_ext(s, slab, obj_exts, p[i]); + alloc_tag_sub(slab_obj_ext_codetag_ref(slab, ext), s->size); } put_slab_obj_exts(obj_exts); } @@ -2413,6 +2459,25 @@ alloc_tagging_slab_free_hook(struct kmem_cache *s, struct slab *slab, void **p, __alloc_tagging_slab_free_hook(s, slab, p, objects); } +/* + * Make sure the static key used by slab_obj_ext_has_codetag() reflects the + * value of !mem_alloc_profiling_permanently_disabled() + * + * Any later mem alloc profiling shutdown won't be reflected in the static key + * because obj_exts with codetags might already exist. + */ +static void __init slab_obj_ext_has_codetag_init(void) +{ + bool need_codetag = !mem_alloc_profiling_permanently_disabled(); + + if (need_codetag != static_key_enabled(&slab_obj_ext_has_codetag_key)) { + if (need_codetag) + static_branch_enable(&slab_obj_ext_has_codetag_key); + else + static_branch_disable(&slab_obj_ext_has_codetag_key); + } +} + #else /* CONFIG_MEM_ALLOC_PROFILING */ static inline void @@ -2427,6 +2492,10 @@ alloc_tagging_slab_free_hook(struct kmem_cache *s, struct slab *slab, void **p, { } +static inline void slab_obj_ext_has_codetag_init(void) +{ +} + #endif /* CONFIG_MEM_ALLOC_PROFILING */ @@ -2472,6 +2541,9 @@ void memcg_slab_free_hook(struct kmem_cache *s, struct slab *slab, void **p, if (likely(!obj_exts)) return; + if (!slab_needs_objcg(slab)) + return; + get_slab_obj_exts(obj_exts); __memcg_slab_free_hook(s, slab, p, objects, obj_exts); put_slab_obj_exts(obj_exts); @@ -2485,7 +2557,6 @@ bool memcg_slab_post_charge(void *p, gfp_t flags) struct kmem_cache *s; struct page *page; struct slab *slab; - unsigned long off; page = virt_to_page(p); if (PageLargeKmalloc(page)) { @@ -2518,16 +2589,15 @@ bool memcg_slab_post_charge(void *p, gfp_t flags) * of slab_obj_exts being allocated from the same slab and thus the slab * becoming effectively unfreeable. */ - if (is_kmalloc_normal(s)) + if (!cache_needs_objcg(s)) return true; /* Ignore already charged objects. */ obj_exts = slab_obj_exts(slab); if (obj_exts) { get_slab_obj_exts(obj_exts); - off = obj_to_index(s, slab, p); - obj_ext = slab_obj_ext(slab, obj_exts, off); - if (unlikely(obj_ext->objcg)) { + obj_ext = slab_obj_ext(s, slab, obj_exts, p); + if (unlikely(slab_obj_ext_objcg(slab, obj_ext))) { put_slab_obj_exts(obj_exts); return true; } @@ -2772,7 +2842,8 @@ static inline struct slab_sheaf *alloc_empty_sheaf(struct kmem_cache *s, return __alloc_empty_sheaf(s, gfp, alloc_flags, s->sheaf_capacity); } -static void free_empty_sheaf(struct kmem_cache *s, struct slab_sheaf *sheaf) +static void __free_empty_sheaf(struct kmem_cache *s, struct slab_sheaf *sheaf, + unsigned int free_flags) { /* * If the sheaf was created with SLAB_ALLOC_NO_RECURSE flag then its @@ -2784,11 +2855,20 @@ static void free_empty_sheaf(struct kmem_cache *s, struct slab_sheaf *sheaf) mark_obj_codetag_empty(sheaf); VM_WARN_ON_ONCE(sheaf->size > 0); - kfree(sheaf); + + if (unlikely(free_flags & SLAB_FREE_NOLOCK)) + kfree_nolock(sheaf); + else + kfree(sheaf); stat(s, SHEAF_FREE); } +static void free_empty_sheaf(struct kmem_cache *s, struct slab_sheaf *sheaf) +{ + __free_empty_sheaf(s, sheaf, SLAB_FREE_DEFAULT); +} + static unsigned int refill_objects(struct kmem_cache *s, void **p, gfp_t gfp, unsigned int min, unsigned int max); @@ -2841,7 +2921,7 @@ static unsigned int __sheaf_flush_main_batch(struct kmem_cache *s) void *objects[PCS_BATCH_MAX]; struct slab_sheaf *sheaf; - lockdep_assert_held(this_cpu_ptr(&s->cpu_sheaves->lock)); + slab_lockdep_assert_held(this_cpu_ptr(&s->cpu_sheaves->lock)); pcs = this_cpu_ptr(s->cpu_sheaves); sheaf = pcs->main; @@ -3393,10 +3473,15 @@ static struct slab *allocate_slab(struct kmem_cache *s, gfp_t flags, stat(s, ORDER_FALLBACK); } - slab->objects = oo_objects(oo); - slab->inuse = 0; - slab->frozen = 0; + /* Initializes frozen, inuse, and any extra 64bit-only flags */ + slab->counters = 0; + slab->objects = oo_objects(oo); + +#ifdef CONFIG_64BIT + if (cache_needs_objcg(s)) + slab->obj_exts_needs_objcg = 1; +#endif slab->slab_cache = s; kasan_poison_slab(slab); @@ -3521,7 +3606,7 @@ __add_partial(struct kmem_cache_node *n, struct slab *slab, enum add_mode mode) static inline void add_partial(struct kmem_cache_node *n, struct slab *slab, enum add_mode mode) { - lockdep_assert_held(&n->list_lock); + slab_lockdep_assert_held(&n->list_lock); __add_partial(n, slab, mode); } @@ -3535,7 +3620,7 @@ static inline void clear_node_partial_state(struct kmem_cache_node *n, static inline void remove_partial(struct kmem_cache_node *n, struct slab *slab) { - lockdep_assert_held(&n->list_lock); + slab_lockdep_assert_held(&n->list_lock); list_del(&slab->slab_list); clear_node_partial_state(n, slab); } @@ -3551,7 +3636,7 @@ static void *alloc_single_from_partial(struct kmem_cache *s, { void *object; - lockdep_assert_held(&n->list_lock); + slab_lockdep_assert_held(&n->list_lock); #ifdef CONFIG_SLUB_DEBUG if (s->flags & SLAB_CONSISTENCY_CHECKS) { @@ -4020,6 +4105,22 @@ static void flush_all(struct kmem_cache *s) cpus_read_unlock(); } +struct deferred_percpu_work { + struct llist_head objects; + struct llist_head objects_by_rcu; + struct llist_head rcu_sheaves; + struct irq_work work; +}; + +static void deferred_percpu_work_fn(struct irq_work *work); + +static DEFINE_PER_CPU(struct deferred_percpu_work, deferred_percpu_work) = { + .objects = LLIST_HEAD_INIT(objects), + .objects_by_rcu = LLIST_HEAD_INIT(objects_by_rcu), + .rcu_sheaves = LLIST_HEAD_INIT(rcu_sheaves), + .work = IRQ_WORK_INIT(deferred_percpu_work_fn), +}; + static void flush_rcu_sheaf(struct work_struct *w) { struct slub_percpu_sheaves *pcs; @@ -4079,6 +4180,8 @@ void flush_all_rcu_sheaves(void) { struct kmem_cache *s; + deferred_work_barrier(); + cpus_read_lock(); mutex_lock(&slab_mutex); @@ -4500,11 +4603,8 @@ static void *___slab_alloc(struct kmem_cache *s, gfp_t gfpflags, int node, return object; } -static void *__slab_alloc_node(struct kmem_cache *s, gfp_t gfpflags, int node, - const struct slab_alloc_context *ac) +static __always_inline int apply_strict_numa_policy(int node) { - void *object; - #ifdef CONFIG_NUMA if (static_branch_unlikely(&strict_numa) && node == NUMA_NO_NODE) { @@ -4525,10 +4625,7 @@ static void *__slab_alloc_node(struct kmem_cache *s, gfp_t gfpflags, int node, } } #endif - - object = ___slab_alloc(s, gfpflags, node, ac); - - return object; + return node; } static __fastpath_inline @@ -4622,7 +4719,7 @@ __pcs_replace_empty_main(struct kmem_cache *s, struct slub_percpu_sheaves *pcs, struct node_barn *barn; bool allow_spin; - lockdep_assert_held(this_cpu_ptr(&s->cpu_sheaves->lock)); + slab_lockdep_assert_held(this_cpu_ptr(&s->cpu_sheaves->lock)); /* Bootstrap or debug cache, back off */ if (unlikely(!cache_has_sheaves(s))) { @@ -4733,28 +4830,6 @@ void *alloc_from_pcs(struct kmem_cache *s, gfp_t gfp, unsigned int alloc_flags, bool node_requested; void *object; -#ifdef CONFIG_NUMA - if (static_branch_unlikely(&strict_numa) && - node == NUMA_NO_NODE) { - - struct mempolicy *mpol = current->mempolicy; - - if (mpol) { - /* - * Special BIND rule support. If the local node - * is in permitted set then do not redirect - * to a particular node. - * Otherwise we apply the memory policy to get - * the node we need to allocate on. - */ - if (mpol->mode != MPOL_BIND || - !node_isset(numa_mem_id(), mpol->nodes)) - - node = mempolicy_slab_node(); - } - } -#endif - node_requested = IS_ENABLED(CONFIG_NUMA) && node != NUMA_NO_NODE; /* @@ -4904,10 +4979,12 @@ static __fastpath_inline void *slab_alloc_node(struct kmem_cache *s, if (unlikely(object)) goto out; + node = apply_strict_numa_policy(node); + object = alloc_from_pcs(s, gfpflags, ac->alloc_flags, node); if (unlikely(!object)) - object = __slab_alloc_node(s, gfpflags, node, ac); + object = ___slab_alloc(s, gfpflags, node, ac); maybe_wipe_obj_freeptr(s, object); @@ -5149,7 +5226,7 @@ void kmem_cache_return_sheaf(struct kmem_cache *s, gfp_t gfp, * simply flush and free it. */ if (!barn || data_race(barn->nr_full) >= MAX_FULL_SHEAVES || - refill_sheaf(s, sheaf, gfp)) { + refill_sheaf(s, sheaf, gfp | __GFP_NOMEMALLOC | __GFP_NOWARN)) { sheaf_flush_unused(s, sheaf); free_empty_sheaf(s, sheaf); return; @@ -5383,6 +5460,8 @@ static void *__kmalloc_nolock_noprof(DECL_TOKEN_PARAMS(size, token), gfp_t gfp_f if (!can_spin_trylock()) return NULL; + node = apply_strict_numa_policy(node); + retry: if (unlikely(size > KMALLOC_MAX_CACHE_SIZE)) return NULL; @@ -5409,10 +5488,10 @@ static void *__kmalloc_nolock_noprof(DECL_TOKEN_PARAMS(size, token), gfp_t gfp_f /* * Do not call slab_alloc_node(), since trylock mode isn't * compatible with slab_pre_alloc_hook/should_failslab and - * kfence_alloc. Hence call __slab_alloc_node() (at most twice) + * kfence_alloc. Hence call ___slab_alloc() (at most twice) * and slab_post_alloc_hook() directly. */ - ret = __slab_alloc_node(s, gfp_flags, node, ac); + ret = ___slab_alloc(s, gfp_flags, node, ac); /* * It's possible we failed due to trylock as we preempted someone with @@ -5758,7 +5837,7 @@ static void __pcs_install_empty_sheaf(struct kmem_cache *s, struct slub_percpu_sheaves *pcs, struct slab_sheaf *empty, struct node_barn *barn) { - lockdep_assert_held(this_cpu_ptr(&s->cpu_sheaves->lock)); + slab_lockdep_assert_held(this_cpu_ptr(&s->cpu_sheaves->lock)); /* This is what we expect to find if nobody interrupted us. */ if (likely(!pcs->spare)) { @@ -5809,7 +5888,7 @@ __pcs_replace_full_main(struct kmem_cache *s, struct slub_percpu_sheaves *pcs, bool put_fail; restart: - lockdep_assert_held(this_cpu_ptr(&s->cpu_sheaves->lock)); + slab_lockdep_assert_held(this_cpu_ptr(&s->cpu_sheaves->lock)); /* Bootstrap or debug cache, back off */ if (unlikely(!cache_has_sheaves(s))) { @@ -6010,24 +6089,26 @@ static void rcu_free_sheaf(struct rcu_head *head) * kvfree_call_rcu() can be called while holding a raw_spinlock_t. Since * __kfree_rcu_sheaf() may acquire a spinlock_t (sleeping lock on PREEMPT_RT), * this would violate lock nesting rules. Therefore, kvfree_call_rcu() avoids - * this problem by bypassing the sheaves layer entirely on PREEMPT_RT. + * this problem by passing SLAB_FREE_NOLOCK on PREEMPT_RT. * * However, lockdep still complains that it is invalid to acquire spinlock_t * while holding raw_spinlock_t, even on !PREEMPT_RT where spinlock_t is a * spinning lock. Tell lockdep that acquiring spinlock_t is valid here - * by temporarily raising the wait-type to LD_WAIT_CONFIG. + * by temporarily raising the wait-type to LD_WAIT_CONFIG. Skip the lockdep map + * on PREEMPT_RT to avoid suppressing valid lockdep warnings. */ static DEFINE_WAIT_OVERRIDE_MAP(kfree_rcu_sheaf_map, LD_WAIT_CONFIG); -bool __kfree_rcu_sheaf(struct kmem_cache *s, void *obj) +bool __kfree_rcu_sheaf(struct kmem_cache *s, void *obj, unsigned int free_flags) { struct slub_percpu_sheaves *pcs; struct slab_sheaf *rcu_sheaf; + bool allow_spin = free_flags_allow_spinning(free_flags); - if (WARN_ON_ONCE(IS_ENABLED(CONFIG_PREEMPT_RT))) - return false; + VM_WARN_ON_ONCE(IS_ENABLED(CONFIG_PREEMPT_RT) && allow_spin); - lock_map_acquire_try(&kfree_rcu_sheaf_map); + if (!IS_ENABLED(CONFIG_PREEMPT_RT)) + lock_map_acquire_try(&kfree_rcu_sheaf_map); if (!local_trylock(&s->cpu_sheaves->lock)) goto fail; @@ -6035,9 +6116,10 @@ bool __kfree_rcu_sheaf(struct kmem_cache *s, void *obj) pcs = this_cpu_ptr(s->cpu_sheaves); if (unlikely(!pcs->rcu_free)) { - struct slab_sheaf *empty; struct node_barn *barn; + unsigned int alloc_flags = to_alloc_flags(free_flags); + gfp_t gfp = allow_spin ? GFP_NOWAIT : __GFP_NOWARN; /* Bootstrap or debug cache, fall back */ if (unlikely(!cache_has_sheaves(s))) { @@ -6057,7 +6139,7 @@ bool __kfree_rcu_sheaf(struct kmem_cache *s, void *obj) goto fail; } - empty = barn_get_empty_sheaf(barn, true); + empty = barn_get_empty_sheaf(barn, allow_spin); if (empty) { pcs->rcu_free = empty; @@ -6066,20 +6148,20 @@ bool __kfree_rcu_sheaf(struct kmem_cache *s, void *obj) local_unlock(&s->cpu_sheaves->lock); - empty = alloc_empty_sheaf(s, GFP_NOWAIT, SLAB_ALLOC_DEFAULT); + empty = alloc_empty_sheaf(s, gfp, alloc_flags); if (!empty) goto fail; if (!local_trylock(&s->cpu_sheaves->lock)) { - barn_put_empty_sheaf(barn, empty); + __free_empty_sheaf(s, empty, free_flags); goto fail; } pcs = this_cpu_ptr(s->cpu_sheaves); if (unlikely(pcs->rcu_free)) - barn_put_empty_sheaf(barn, empty); + __free_empty_sheaf(s, empty, free_flags); else pcs->rcu_free = empty; } @@ -6105,18 +6187,34 @@ bool __kfree_rcu_sheaf(struct kmem_cache *s, void *obj) * we flush before local_unlock to make sure a racing * flush_all_rcu_sheaves() doesn't miss this sheaf */ - if (rcu_sheaf) - call_rcu(&rcu_sheaf->rcu_head, rcu_free_sheaf); + if (rcu_sheaf) { + /* + * With !allow_spin, we might have interrupted call_rcu()'s + * IRQ-disabled critical section. If IRQs are not disabled, + * we know that's not the case. + */ + if (unlikely(!allow_spin && irqs_disabled())) { + struct deferred_percpu_work *dpw; + + dpw = this_cpu_ptr(&deferred_percpu_work); + if (llist_add(&rcu_sheaf->llnode, &dpw->rcu_sheaves)) + irq_work_queue(&dpw->work); + } else { + call_rcu(&rcu_sheaf->rcu_head, rcu_free_sheaf); + } + } local_unlock(&s->cpu_sheaves->lock); stat(s, FREE_RCU_SHEAF); - lock_map_release(&kfree_rcu_sheaf_map); + if (!IS_ENABLED(CONFIG_PREEMPT_RT)) + lock_map_release(&kfree_rcu_sheaf_map); return true; fail: stat(s, FREE_RCU_SHEAF_FAIL); - lock_map_release(&kfree_rcu_sheaf_map); + if (!IS_ENABLED(CONFIG_PREEMPT_RT)) + lock_map_release(&kfree_rcu_sheaf_map); return false; } @@ -6172,51 +6270,21 @@ static __always_inline bool can_free_to_pcs(struct slab *slab) } /* - * Bulk free objects to the percpu sheaves. - * Unlike free_to_pcs() this includes the calls to all necessary hooks - * and the fallback to freeing to slab pages. + * Try to free as many objects (already processed by free hooks) as possible to + * a single per-cpu sheaf. + * + * Returns how many objects were freed. Zero means failure and the caller should + * fall back to __kmem_cache_free_bulk(). */ -static void free_to_pcs_bulk(struct kmem_cache *s, size_t size, void **p) +static unsigned int __free_to_pcs_batch(struct kmem_cache *s, size_t size, void **p) { struct slub_percpu_sheaves *pcs; struct slab_sheaf *main, *empty; - bool init = slab_want_init_on_free(s); - unsigned int batch, i = 0; struct node_barn *barn; - void *remote_objects[PCS_BATCH_MAX]; - unsigned int remote_nr = 0; + unsigned int batch; - while (i < size) { - struct slab *slab = virt_to_slab(p[i]); - - memcg_slab_free_hook(s, slab, p + i, 1); - alloc_tagging_slab_free_hook(s, slab, p + i, 1); - - if (unlikely(!slab_free_hook(s, p[i], init, false))) { - p[i] = p[--size]; - continue; - } - - if (unlikely(!can_free_to_pcs(slab))) { - remote_objects[remote_nr] = p[i]; - p[i] = p[--size]; - if (++remote_nr >= PCS_BATCH_MAX) { - __kmem_cache_free_bulk(s, remote_nr, &remote_objects[0]); - stat_add(s, FREE_SLOWPATH, remote_nr); - remote_nr = 0; - } - continue; - } - - i++; - } - - if (!size) - goto flush_remote; - -next_batch: if (!local_trylock(&s->cpu_sheaves->lock)) - goto fallback; + return 0; pcs = this_cpu_ptr(s->cpu_sheaves); @@ -6262,60 +6330,95 @@ static void free_to_pcs_bulk(struct kmem_cache *s, size_t size, void **p) stat_add(s, FREE_FASTPATH, batch); - if (batch < size) { - p += batch; - size -= batch; - goto next_batch; - } - - if (remote_nr) - goto flush_remote; - - return; + return batch; no_empty: local_unlock(&s->cpu_sheaves->lock); - /* - * if we depleted all empty sheaves in the barn or there are too - * many full sheaves, free the rest to slab pages - */ -fallback: - __kmem_cache_free_bulk(s, size, p); - stat_add(s, FREE_SLOWPATH, size); + return 0; +} -flush_remote: +/* + * Bulk free objects to the percpu sheaves. + * Unlike free_to_pcs() this includes the calls to all necessary hooks + * and the fallback to freeing to slab pages. + */ +static void free_to_pcs_bulk(struct kmem_cache *s, size_t size, void **p) +{ + bool init = slab_want_init_on_free(s); + void **remote_objects = p; + unsigned int remote_nr = 0; + + /* + * Process the free hooks and separate out remote objects by + * partitioning the 'p' array in place: + * + * [0, remote_nr) - processed remote objects + * [remote_nr, i) - processed local objects + * [i, size) - unprocessed objects + */ + for (unsigned int i = 0; i < size;) { + struct slab *slab = virt_to_slab(p[i]); + + memcg_slab_free_hook(s, slab, p + i, 1); + alloc_tagging_slab_free_hook(s, slab, p + i, 1); + + if (unlikely(!slab_free_hook(s, p[i], init, false))) { + p[i] = p[--size]; + continue; + } + + if (unlikely(!can_free_to_pcs(slab))) { + if (i != remote_nr) + swap(remote_objects[remote_nr], p[i]); + remote_nr++; + } + + i++; + } + + p += remote_nr; + size -= remote_nr; + + while (size) { + unsigned int batch_freed = __free_to_pcs_batch(s, size, p); + + if (!batch_freed) { + __kmem_cache_free_bulk(s, size, p); + stat_add(s, FREE_SLOWPATH, size); + break; + } + + p += batch_freed; + size -= batch_freed; + } + + /* + * Processing remote objects last decreases the chances of cpu migration + * while freeing to sheaves and compromising object locality + */ if (remote_nr) { - __kmem_cache_free_bulk(s, remote_nr, &remote_objects[0]); + __kmem_cache_free_bulk(s, remote_nr, remote_objects); stat_add(s, FREE_SLOWPATH, remote_nr); } } -struct defer_free { - struct llist_head objects; - struct irq_work work; -}; - -static void free_deferred_objects(struct irq_work *work); - -static DEFINE_PER_CPU(struct defer_free, defer_free_objects) = { - .objects = LLIST_HEAD_INIT(objects), - .work = IRQ_WORK_INIT(free_deferred_objects), -}; - /* * In PREEMPT_RT irq_work runs in per-cpu kthread, so it's safe * to take sleeping spin_locks from __slab_free(). * In !PREEMPT_RT irq_work will run after local_unlock_irqrestore(). */ -static void free_deferred_objects(struct irq_work *work) +static void deferred_percpu_work_fn(struct irq_work *work) { - struct defer_free *df = container_of(work, struct defer_free, work); - struct llist_head *objs = &df->objects; + struct deferred_percpu_work *dpw; + struct llist_head *objs, *objs_by_rcu, *rcu_sheaves; struct llist_node *llnode, *pos, *t; + struct slab_sheaf *sheaf, *next; - if (llist_empty(objs)) - return; + dpw = container_of(work, struct deferred_percpu_work, work); + rcu_sheaves = &dpw->rcu_sheaves; + objs = &dpw->objects; + objs_by_rcu = &dpw->objects_by_rcu; llnode = llist_del_all(objs); llist_for_each_safe(pos, t, llnode) { @@ -6339,27 +6442,51 @@ static void free_deferred_objects(struct irq_work *work) __slab_free(s, slab, x, x, 1, _THIS_IP_); stat(s, FREE_SLOWPATH); } + + llnode = llist_del_all(objs_by_rcu); + llist_for_each_safe(pos, t, llnode) { + void *head = pos; + void *objp = kvmalloc_obj_start_addr(head); + + kvfree_call_rcu(head, objp); + } + + llnode = llist_del_all(rcu_sheaves); + llist_for_each_entry_safe(sheaf, next, llnode, llnode) + call_rcu(&sheaf->rcu_head, rcu_free_sheaf); } static void defer_free(struct kmem_cache *s, void *head) { - struct defer_free *df; + struct deferred_percpu_work *dpw; guard(preempt)(); head = kasan_reset_tag(head); - df = this_cpu_ptr(&defer_free_objects); - if (llist_add(head + s->offset, &df->objects)) - irq_work_queue(&df->work); + dpw = this_cpu_ptr(&deferred_percpu_work); + if (llist_add(head + s->offset, &dpw->objects)) + irq_work_queue(&dpw->work); } -void defer_free_barrier(void) +void defer_kfree_rcu(struct kvfree_rcu_head *head) +{ + struct deferred_percpu_work *dpw; + + guard(preempt)(); + + dpw = this_cpu_ptr(&deferred_percpu_work); + if (llist_add((struct llist_node *)head, &dpw->objects_by_rcu)) + irq_work_queue(&dpw->work); +} + +/* Must be called before flush_rcu_sheaves_on_cache() */ +void deferred_work_barrier(void) { int cpu; for_each_possible_cpu(cpu) - irq_work_sync(&per_cpu_ptr(&defer_free_objects, cpu)->work); + irq_work_sync(&per_cpu_ptr(&deferred_percpu_work, cpu)->work); } static __fastpath_inline @@ -6521,7 +6648,7 @@ static inline size_t slab_ksize(struct slab *slab) */ if (s->flags & (SLAB_TYPESAFE_BY_RCU | SLAB_STORE_USER)) return s->inuse; - else if (obj_exts_in_object(s, slab)) + else if (obj_exts_in_object(slab)) return s->inuse; /* * Else we can use all the padding etc for the allocation @@ -6618,43 +6745,21 @@ static void free_large_kmalloc(struct page *page, void *object) */ void kvfree_rcu_cb(struct rcu_head *head) { - void *obj = head; - struct page *page; - struct slab *slab; - struct kmem_cache *s; - void *slab_addr; + void *obj; + + obj = kvmalloc_obj_start_addr(head); if (is_vmalloc_addr(obj)) { - obj = (void *) PAGE_ALIGN_DOWN((unsigned long)obj); vfree(obj); - return; - } - - page = virt_to_page(obj); - slab = page_slab(page); - if (!slab) { - /* - * rcu_head offset can be only less than page size so no need to - * consider allocation order - */ - obj = (void *) PAGE_ALIGN_DOWN((unsigned long)obj); - free_large_kmalloc(page, obj); - return; - } - - s = slab->slab_cache; - slab_addr = slab_address(slab); - - if (is_kfence_address(obj)) { - obj = kfence_object_start(obj); } else { - unsigned int idx = __obj_to_index(s, slab_addr, obj); + struct page *page = virt_to_page(obj); + struct slab *slab = page_slab(page); - obj = slab_addr + s->size * idx; - obj = fixup_red_left(s, obj); + if (slab) + slab_free(slab->slab_cache, slab, obj, _RET_IP_); + else + free_large_kmalloc(page, obj); } - - slab_free(s, slab, obj, _RET_IP_); } /** @@ -7924,7 +8029,7 @@ static int calculate_sizes(struct kmem_cache_args *args, struct kmem_cache *s) aligned_size = ALIGN(size, s->align); #if defined(CONFIG_SLAB_OBJ_EXT) && defined(CONFIG_64BIT) if (slab_args_unmergeable(args, s->flags) && - (aligned_size - size >= sizeof(struct slabobj_ext))) + (aligned_size - size >= cache_obj_ext_size(s))) s->flags |= SLAB_OBJ_EXT_IN_OBJ; #endif size = aligned_size; @@ -8534,6 +8639,8 @@ void __init kmem_cache_init(void) boot_kmem_cache_node; int node; + slab_obj_ext_has_codetag_init(); + if (debug_guardpage_minorder()) slub_max_order = 0; @@ -8966,14 +9073,12 @@ static void process_slab(struct loc_track *t, struct kmem_cache *s, enum slab_stat_type { SL_ALL, /* All slabs */ SL_PARTIAL, /* Only partially allocated slabs */ - SL_CPU, /* Only slabs used for cpu caches */ SL_OBJECTS, /* Determine allocated objects not slabs */ SL_TOTAL /* Determine object capacity not slabs */ }; #define SO_ALL (1 << SL_ALL) #define SO_PARTIAL (1 << SL_PARTIAL) -#define SO_CPU (1 << SL_CPU) #define SO_OBJECTS (1 << SL_OBJECTS) #define SO_TOTAL (1 << SL_TOTAL) @@ -9162,7 +9267,7 @@ SLAB_ATTR_RO(partial); static ssize_t cpu_slabs_show(struct kmem_cache *s, char *buf) { - return show_slab_objects(s, buf, SO_CPU); + return sysfs_emit(buf, "0\n"); } SLAB_ATTR_RO(cpu_slabs); @@ -9664,6 +9769,11 @@ static int sysfs_slab_add(struct kmem_cache *s) s->kobj.kset = kset; err = kobject_init_and_add(&s->kobj, &slab_ktype, NULL, "%s", name); + /* + * Intentionally skip kobject_put(). See commit 2420baa8e046 + * ("mm/slab: Allow cache creation to proceed even if sysfs + * registration fails") + */ if (err) goto out; @@ -9729,28 +9839,20 @@ int sysfs_slab_alias(struct kmem_cache *s, const char *name) return 0; } -static int __init slab_sysfs_init(void) +static int __init slab_kset_init(void) { - struct kmem_cache *s; - int err; - - mutex_lock(&slab_mutex); - slab_kset = kset_create_and_add("slab", NULL, kernel_kobj); if (!slab_kset) { - mutex_unlock(&slab_mutex); pr_err("Cannot register slab subsystem.\n"); return -ENOMEM; } - slab_state = FULL; + return 0; +} - list_for_each_entry(s, &slab_caches, list) { - err = sysfs_slab_add(s); - if (err) - pr_err("SLUB: Unable to add boot slab %s to sysfs\n", - s->name); - } +static void __init slab_sysfs_process_aliases(void) +{ + int err; while (alias_list) { struct saved_alias *al = alias_list; @@ -9762,13 +9864,42 @@ static int __init slab_sysfs_init(void) al->name); kfree(al); } - - mutex_unlock(&slab_mutex); - return 0; } -late_initcall(slab_sysfs_init); #endif /* SLAB_SUPPORTS_SYSFS */ +#if defined(SLAB_SUPPORTS_SYSFS) || \ + (defined(CONFIG_SLUB_DEBUG) && defined(CONFIG_DEBUG_FS)) +static int __init slab_late_init(void) +{ + struct kmem_cache *s; + int err; + + mutex_lock(&slab_mutex); + + err = slab_kset_init(); + if (err) + goto out; + + slab_debugfs_root_init(); + slab_state = FULL; + + list_for_each_entry(s, &slab_caches, list) { + if (sysfs_slab_add(s)) + pr_err("SLUB: Unable to add boot slab %s to sysfs\n", + s->name); + + if (s->flags & SLAB_STORE_USER) + debugfs_slab_add(s); + } + + slab_sysfs_process_aliases(); +out: + mutex_unlock(&slab_mutex); + return err; +} +late_initcall(slab_late_init); +#endif + #if defined(CONFIG_SLUB_DEBUG) && defined(CONFIG_DEBUG_FS) static int slab_debugfs_show(struct seq_file *seq, void *v) { @@ -9959,23 +10090,16 @@ static void debugfs_slab_add(struct kmem_cache *s) void debugfs_slab_release(struct kmem_cache *s) { + if (unlikely(!slab_debugfs_root)) + return; + debugfs_lookup_and_remove(s->name, slab_debugfs_root); } -static int __init slab_debugfs_init(void) +static void __init slab_debugfs_root_init(void) { - struct kmem_cache *s; - slab_debugfs_root = debugfs_create_dir("slab", NULL); - - list_for_each_entry(s, &slab_caches, list) - if (s->flags & SLAB_STORE_USER) - debugfs_slab_add(s); - - return 0; - } -__initcall(slab_debugfs_init); #endif /* * The /proc/slabinfo ABI