mirror of
https://git.kernel.org/pub/scm/linux/kernel/git/torvalds/linux.git
synced 2026-08-31 08:15:07 -04:00
mm/zsmalloc: drop pool->lock from zs_free on 64-bit systems
With class_idx encoded in obj, zs_free() can locate the size_class without holding pool->lock on 64-bit systems. Page migration also takes class->lock and only rewrites the PFN field of obj, so: 1. read obj locklessly, 2. lock the size_class derived from obj's class_idx, 3. re-read obj under class->lock to get a stable PFN. This eliminates the rwlock read-side cacheline bouncing between zs_free() and migration/compaction on multi-core systems. Annotate handle_to_obj()/record_obj() with READ_ONCE()/WRITE_ONCE() to prevent load/store tearing on the lockless read path and silence KCSAN data race reports. When ZS_OBJ_CLASS_BITS == 0 (32-bit, or 64-bit with obj too narrow to hold class_idx), zs_free() keeps pool->lock. [akpm@linux-foundation.org: build fix] [akpm@linux-foundation.org: fix obj_to_class_idx() warning yet again] [baohua@kernel.org: update the comment about pool lock] Link: https://lore.kernel.org/20260725035733.53241-1-baohua@kernel.org Link: https://lore.kernel.org/20260626015003.2965881-3-haowenchao22@gmail.com Signed-off-by: Wenchao Hao <haowenchao@xiaomi.com> Reviewed-by: Nhat Pham <nphamcs@gmail.com> Reviewed-by: Barry Song <baohua@kernel.org> Cc: Joshua Hahn <joshua.hahnjy@gmail.com> Cc: Minchan Kim <minchan@kernel.org> Cc: Sergey Senozhatsky <senozhatsky@chromium.org> Cc: Xueyuan Chen <xueyuan.chen21@gmail.com> Cc: kernel test robot <lkp@intel.com> Signed-off-by: Andrew Morton <akpm@linux-foundation.org>
This commit is contained in:
committed by
Andrew Morton
parent
9909b088b1
commit
59e88952a8
@@ -21,6 +21,10 @@
|
||||
* pool->lock
|
||||
* class->lock
|
||||
* zspage->lock
|
||||
*
|
||||
* When ZS_OBJ_CLASS_BITS > 0, zs_free() skips pool->lock; it picks
|
||||
* the size_class from obj's encoded class_idx and serializes against
|
||||
* page migration via class->lock.
|
||||
*/
|
||||
|
||||
#include <linux/module.h>
|
||||
@@ -463,10 +467,13 @@ static void cache_free_zspage(struct zspage *zspage)
|
||||
kmem_cache_free(zspage_cachep, zspage);
|
||||
}
|
||||
|
||||
/* class->lock(which owns the handle) synchronizes races */
|
||||
/*
|
||||
* Pairs with READ_ONCE() in handle_to_obj(): zs_free() may read the
|
||||
* handle locklessly, so prevent store tearing here.
|
||||
*/
|
||||
static void record_obj(unsigned long handle, unsigned long obj)
|
||||
{
|
||||
*(unsigned long *)handle = obj;
|
||||
WRITE_ONCE(*(unsigned long *)handle, obj);
|
||||
}
|
||||
|
||||
static inline bool __maybe_unused is_first_zpdesc(struct zpdesc *zpdesc)
|
||||
@@ -816,7 +823,7 @@ static unsigned long location_to_obj(struct zpdesc *zpdesc, unsigned int obj_idx
|
||||
|
||||
static unsigned long handle_to_obj(unsigned long handle)
|
||||
{
|
||||
return *(unsigned long *)handle;
|
||||
return READ_ONCE(*(unsigned long *)handle);
|
||||
}
|
||||
|
||||
static inline bool obj_allocated(struct zpdesc *zpdesc, void *obj,
|
||||
@@ -1450,10 +1457,66 @@ static void obj_free(int class_size, unsigned long obj)
|
||||
mod_zspage_inuse(zspage, -1);
|
||||
}
|
||||
|
||||
#if (ZS_OBJ_CLASS_BITS > 0) || defined(CONFIG_COMPACTION)
|
||||
/* Folds to 0 when ZS_OBJ_CLASS_BITS == 0; no ifdef needed at callers. */
|
||||
static unsigned int obj_to_class_idx(unsigned long obj)
|
||||
{
|
||||
return (obj >> ZS_OBJ_IDX_BITS) & ZS_OBJ_CLASS_MASK;
|
||||
}
|
||||
#endif
|
||||
|
||||
/*
|
||||
* Resolve @handle to its zspage / size_class and acquire class->lock.
|
||||
*
|
||||
* When class_idx is encoded in obj (ZS_OBJ_CLASS_BITS > 0), it is
|
||||
* invariant under page migration, so the handle can be read locklessly
|
||||
* to pick the size_class. Once class->lock is held migration is
|
||||
* blocked and the handle is re-read to obtain a stable PFN.
|
||||
*
|
||||
* Otherwise (32-bit, or 64-bit fallback paths like UML where the
|
||||
* encoding is disabled), fall back to pool->lock for the lookup.
|
||||
*/
|
||||
#if ZS_OBJ_CLASS_BITS > 0
|
||||
static inline void obj_class_get_and_lock(struct zs_pool *pool, unsigned long handle,
|
||||
unsigned long *objp, struct zspage **zspagep,
|
||||
struct size_class **classp)
|
||||
__acquires(&(*classp)->lock)
|
||||
{
|
||||
struct zpdesc *f_zpdesc;
|
||||
unsigned long obj;
|
||||
|
||||
obj = handle_to_obj(handle);
|
||||
*classp = pool->size_class[obj_to_class_idx(obj)];
|
||||
spin_lock(&(*classp)->lock);
|
||||
/* Re-read under class->lock: PFN is now stable vs migration. */
|
||||
obj = handle_to_obj(handle);
|
||||
obj_to_zpdesc(obj, &f_zpdesc);
|
||||
*zspagep = get_zspage(f_zpdesc);
|
||||
*objp = obj;
|
||||
}
|
||||
#else
|
||||
static inline void obj_class_get_and_lock(struct zs_pool *pool, unsigned long handle,
|
||||
unsigned long *objp, struct zspage **zspagep,
|
||||
struct size_class **classp)
|
||||
__acquires(&(*classp)->lock)
|
||||
{
|
||||
struct zpdesc *f_zpdesc;
|
||||
unsigned long obj;
|
||||
|
||||
read_lock(&pool->lock);
|
||||
obj = handle_to_obj(handle);
|
||||
obj_to_zpdesc(obj, &f_zpdesc);
|
||||
*zspagep = get_zspage(f_zpdesc);
|
||||
*classp = zspage_class(pool, *zspagep);
|
||||
spin_lock(&(*classp)->lock);
|
||||
read_unlock(&pool->lock);
|
||||
*objp = obj;
|
||||
}
|
||||
#endif
|
||||
|
||||
void zs_free(struct zs_pool *pool, unsigned long handle)
|
||||
{
|
||||
struct zspage *zspage;
|
||||
struct zpdesc *f_zpdesc;
|
||||
unsigned long obj;
|
||||
struct size_class *class;
|
||||
int fullness;
|
||||
@@ -1461,17 +1524,7 @@ void zs_free(struct zs_pool *pool, unsigned long handle)
|
||||
if (IS_ERR_OR_NULL((void *)handle))
|
||||
return;
|
||||
|
||||
/*
|
||||
* The pool->lock protects the race with zpage's migration
|
||||
* so it's safe to get the page from handle.
|
||||
*/
|
||||
read_lock(&pool->lock);
|
||||
obj = handle_to_obj(handle);
|
||||
obj_to_zpdesc(obj, &f_zpdesc);
|
||||
zspage = get_zspage(f_zpdesc);
|
||||
class = zspage_class(pool, zspage);
|
||||
spin_lock(&class->lock);
|
||||
read_unlock(&pool->lock);
|
||||
obj_class_get_and_lock(pool, handle, &obj, &zspage, &class);
|
||||
|
||||
class_stat_sub(class, ZS_OBJS_INUSE, 1);
|
||||
obj_free(class->size, obj);
|
||||
@@ -1714,12 +1767,6 @@ static void lock_zspage(struct zspage *zspage)
|
||||
zspage_read_unlock(zspage);
|
||||
}
|
||||
|
||||
/* Folds to 0 when ZS_OBJ_CLASS_BITS == 0; no ifdef needed at callers. */
|
||||
static unsigned int obj_to_class_idx(unsigned long obj)
|
||||
{
|
||||
return (obj >> ZS_OBJ_IDX_BITS) & ZS_OBJ_CLASS_MASK;
|
||||
}
|
||||
|
||||
static void replace_sub_page(struct size_class *class, struct zspage *zspage,
|
||||
struct zpdesc *newzpdesc, struct zpdesc *oldzpdesc)
|
||||
{
|
||||
@@ -1785,8 +1832,8 @@ static int zs_page_migrate(struct page *newpage, struct page *page,
|
||||
pool = zspage->pool;
|
||||
|
||||
/*
|
||||
* The pool migrate_lock protects the race between zpage migration
|
||||
* and zs_free.
|
||||
* The pool migrate_lock protects against races between zpage migration
|
||||
* and zs_free(), but only when ZS_OBJ_CLASS_BITS does not apply.
|
||||
*/
|
||||
write_lock(&pool->lock);
|
||||
class = zspage_class(pool, zspage);
|
||||
@@ -1965,8 +2012,9 @@ static unsigned long __zs_compact(struct zs_pool *pool,
|
||||
unsigned long pages_freed = 0;
|
||||
|
||||
/*
|
||||
* protect the race between zpage migration and zs_free
|
||||
* as well as zpage allocation/free
|
||||
* Protect against races between zpage migration and zs_free()
|
||||
* (only when ZS_OBJ_CLASS_BITS does not apply), as well as
|
||||
* zpage allocation and free.
|
||||
*/
|
||||
write_lock(&pool->lock);
|
||||
spin_lock(&class->lock);
|
||||
|
||||
Reference in New Issue
Block a user