Merge branch 'bpf-add-memory-usage-for-arena-and-selftest'

Jiayuan Chen says:

====================
bpf: Add memory usage for arena and selftest

arena is the only map type whose map_mem_usage() still returns 0, so
bpftool and fdinfo always show 0 memlock for it no matter how many pages
it holds. This series adds the accounting, runs the arena selftests
serially so they don't OOM a small CI VM, and adds a test for it.

v1 -> v2:
- split the scratch_page -> arena refactor into a prep patch (no
  functional change)
- use WRITE_ONCE() for nr_pages to pair with the lockless reader
- reword the "run serially" commit message
- selftest: alloc prog returns 0 so the failure check is reached

v1: https://lore.kernel.org/bpf/20260716142746.8794-1-jiayuan.chen@linux.dev/
====================

Link: https://patch.msgid.link/20260717114117.350851-1-jiayuan.chen@linux.dev
Signed-off-by: Kumar Kartikeya Dwivedi <memxor@gmail.com>
This commit is contained in:
Kumar Kartikeya Dwivedi
2026-07-19 17:56:11 +02:00
11 changed files with 186 additions and 19 deletions

View File

@@ -55,8 +55,10 @@ struct bpf_arena {
struct vm_struct *kern_vm;
struct page *scratch_page;
struct range_tree rt;
/* protects rt */
/* protects rt and nr_pages */
rqspinlock_t spinlock;
/* number of pages currently populated in the arena */
u64 nr_pages;
struct list_head vma_list;
/* protects vma_list */
struct mutex lock;
@@ -143,14 +145,14 @@ static long compute_pgoff(struct bpf_arena *arena, long uaddr)
}
struct apply_range_data {
struct bpf_arena *arena;
struct page **pages;
struct page *scratch_page;
int i;
};
struct clear_range_data {
struct bpf_arena *arena;
struct llist_head *free_pages;
struct page *scratch_page;
};
static int apply_range_set_cb(pte_t *pte, unsigned long addr, void *data)
@@ -180,7 +182,7 @@ static int apply_range_set_cb(pte_t *pte, unsigned long addr, void *data)
if (pte_none(old))
continue;
if (WARN_ON_ONCE(pte_page(old) != d->scratch_page))
if (WARN_ON_ONCE(pte_page(old) != d->arena->scratch_page))
return -EBUSY;
ptep_get_and_clear(&init_mm, addr, pte);
flush_tlb_before_set(addr);
@@ -196,6 +198,7 @@ static int apply_range_set_cb(pte_t *pte, unsigned long addr, void *data)
set_pte_at(&init_mm, addr, pte, pteval);
#endif
d->i++;
WRITE_ONCE(d->arena->nr_pages, d->arena->nr_pages + 1);
return 0;
}
@@ -227,10 +230,11 @@ static int apply_range_clear_cb(pte_t *pte, unsigned long addr, void *data)
* scratches its PTE. A later bpf_arena_free_pages() over that range walks
* here. Without the skip, scratch_page would be freed.
*/
if (page == d->scratch_page)
if (page == d->arena->scratch_page)
return 0;
__llist_add(&page->pcp_llist, d->free_pages);
WRITE_ONCE(d->arena->nr_pages, d->arena->nr_pages - 1);
return 0;
}
@@ -413,7 +417,9 @@ static int arena_map_check_btf(struct bpf_map *map, const struct btf *btf,
static u64 arena_map_mem_usage(const struct bpf_map *map)
{
return 0;
struct bpf_arena *arena = container_of(map, struct bpf_arena, map);
return (u64)READ_ONCE(arena->nr_pages) << PAGE_SHIFT;
}
struct vma_list {
@@ -506,8 +512,7 @@ static vm_fault_t arena_vm_fault(struct vm_fault *vmf)
if (ret)
goto out_sigsegv_memcg;
struct apply_range_data data = { .pages = &page, .i = 0,
.scratch_page = arena->scratch_page };
struct apply_range_data data = { .arena = arena, .pages = &page, .i = 0 };
/* Account into memcg of the process that created bpf_arena */
ret = bpf_map_alloc_pages(map, NUMA_NO_NODE, 1, &page);
if (ret) {
@@ -696,8 +701,8 @@ static long arena_alloc_pages(struct bpf_arena *arena, long uaddr, long page_cnt
bpf_map_memcg_exit(old_memcg, new_memcg);
return 0;
}
data.arena = arena;
data.pages = pages;
data.scratch_page = arena->scratch_page;
if (raw_res_spin_lock_irqsave(&arena->spinlock, flags))
goto out_free_pages;
@@ -873,8 +878,8 @@ static void arena_free_pages(struct bpf_arena *arena, long uaddr, long page_cnt,
range_tree_set(&arena->rt, pgoff, page_cnt);
init_llist_head(&free_pages);
cdata.arena = arena;
cdata.free_pages = &free_pages;
cdata.scratch_page = arena->scratch_page;
/* clear ptes and collect struct pages */
apply_to_existing_page_range(&init_mm, kaddr, page_cnt << PAGE_SHIFT,
apply_range_clear_cb, &cdata);
@@ -981,8 +986,8 @@ static void arena_free_worker(struct work_struct *work)
bpf_map_memcg_enter(&arena->map, &old_memcg, &new_memcg);
init_llist_head(&free_pages);
cdata.arena = arena;
cdata.free_pages = &free_pages;
cdata.scratch_page = arena->scratch_page;
arena_vm_start = bpf_arena_get_kern_vm_start(arena);
user_vm_start = bpf_arena_get_user_vm_start(arena);

View File

@@ -222,7 +222,7 @@ static void test_store_release(struct arena_atomics *skel)
"store_release64_result");
}
void test_arena_atomics(void)
void serial_test_arena_atomics(void)
{
struct arena_atomics *skel;
int err;

View File

@@ -66,7 +66,7 @@ static void test_arena_direct_value_one_past_end(void)
close(map_fd);
}
void test_arena_direct_value(void)
void serial_test_arena_direct_value(void)
{
if (test__start_subtest("one_past_end"))
test_arena_direct_value_one_past_end();

View File

@@ -81,7 +81,7 @@ static void test_arena_htab_asm(void)
arena_htab_asm__destroy(skel);
}
void test_arena_htab(void)
void serial_test_arena_htab(void)
{
if (test__start_subtest("arena_htab_llvm"))
test_arena_htab_llvm();

View File

@@ -68,7 +68,7 @@ static void test_arena_list_add_del(int cnt, bool nonsleepable)
arena_list__destroy(skel);
}
void test_arena_list(void)
void serial_test_arena_list(void)
{
if (test__start_subtest("arena_list_1"))
test_arena_list_add_del(1, false);

View File

@@ -0,0 +1,122 @@
// SPDX-License-Identifier: GPL-2.0
#include <test_progs.h>
#include <sys/user.h>
#ifndef PAGE_SIZE /* on some archs it comes in sys/user.h */
#include <unistd.h>
#define PAGE_SIZE getpagesize()
#endif
#include "arena_mem_usage.skel.h"
/*
* arena_map_mem_usage() is surfaced to user space through the map's
* /proc/<pid>/fdinfo/<fd> "memlock:" line (the same value bpftool map show
* prints). Read it directly so the test has no external dependency.
*/
static long map_memlock(int map_fd)
{
char path[64], line[128];
long memlock = -1;
FILE *f;
snprintf(path, sizeof(path), "/proc/self/fdinfo/%d", map_fd);
f = fopen(path, "r");
if (!ASSERT_OK_PTR(f, "open_fdinfo"))
return -1;
while (fgets(line, sizeof(line), f)) {
if (sscanf(line, "memlock:\t%ld", &memlock) == 1)
break;
}
fclose(f);
ASSERT_NEQ(memlock, -1, "parse_memlock");
return memlock;
}
static int run(struct bpf_program *prog, const char *name)
{
LIBBPF_OPTS(bpf_test_run_opts, opts);
int err = bpf_prog_test_run_opts(bpf_program__fd(prog), &opts);
if (!ASSERT_OK(err, name))
return -1;
if (!ASSERT_OK(opts.retval, name))
return -1;
return 0;
}
void serial_test_arena_mem_usage(void)
{
struct arena_mem_usage *skel;
const long ps = PAGE_SIZE;
char *base;
size_t sz;
int fd, i;
skel = arena_mem_usage__open_and_load();
if (!ASSERT_OK_PTR(skel, "open_load"))
return;
fd = bpf_map__fd(skel->maps.arena);
/* Fresh arena: no data pages, and the scratch page is not counted. */
ASSERT_EQ(map_memlock(fd), 0, "initial");
/* BPF-side allocation of 17 pages. */
skel->bss->alloc_cnt = 17;
if (run(skel->progs.alloc, "alloc"))
goto out;
/*
* A NULL ptr means bpf_arena_alloc_pages() itself failed (e.g. the host
* is under memory pressure), not a miscount -- flag it distinctly so a
* red CI run is not mistaken for a counting bug.
*/
if (!ASSERT_OK_PTR(skel->bss->ptr, "arena_alloc_pages"))
goto out;
ASSERT_EQ(map_memlock(fd), 17 * ps, "after_alloc");
/* Free a single page (arena_free_pages page_cnt==1 path). */
skel->bss->free_byte_off = 0;
skel->bss->free_cnt = 1;
if (run(skel->progs.free_pages, "free_one"))
goto out;
ASSERT_EQ(map_memlock(fd), 16 * ps, "after_free_one");
/* Free ten pages in one call (bulk path); only the freed pages count. */
skel->bss->free_byte_off = 1 * ps;
skel->bss->free_cnt = 10;
if (run(skel->progs.free_pages, "free_bulk"))
goto out;
ASSERT_EQ(map_memlock(fd), 6 * ps, "after_free_bulk");
/* Free the remaining six -> arena empty again. */
skel->bss->free_byte_off = 11 * ps;
skel->bss->free_cnt = 6;
if (run(skel->progs.free_pages, "free_rest"))
goto out;
ASSERT_EQ(map_memlock(fd), 0, "after_free_rest");
/*
* User-space fault-in: touching unallocated arena pages allocates them
* through arena_vm_fault(). libbpf mmap()s the arena at map_extra during
* load, so bpf_map__initial_value() hands back that base.
*/
base = bpf_map__initial_value(skel->maps.arena, &sz);
if (!ASSERT_OK_PTR(base, "arena_base"))
goto out;
for (i = 0; i < 8; i++)
base[i * ps] = 1;
ASSERT_EQ(map_memlock(fd), 8 * ps, "after_faultin");
/*
* Free the faulted-in pages from BPF. They are mapped into the user vma
* (elevated refcount), so this also exercises the zap path.
*/
skel->bss->ptr = base;
skel->bss->free_byte_off = 0;
skel->bss->free_cnt = 8;
if (run(skel->progs.free_pages, "free_faulted"))
goto out;
ASSERT_EQ(map_memlock(fd), 0, "after_free_faulted");
out:
arena_mem_usage__destroy(skel);
}

View File

@@ -101,7 +101,7 @@ static void test_arena_spin_lock_size(int size)
return;
}
void test_arena_spin_lock(void)
void serial_test_arena_spin_lock(void)
{
repeat = 1000;
if (test__start_subtest("arena_spin_lock_1"))

View File

@@ -23,7 +23,7 @@ static void test_arena_str(void)
arena_strsearch__destroy(skel);
}
void test_arena_strsearch(void)
void serial_test_arena_strsearch(void)
{
if (test__start_subtest("arena_strsearch"))
test_arena_str();

View File

@@ -202,7 +202,7 @@ static void run_libarena_parallel_test(struct libarena *skel, struct bpf_program
run_libarena_parallel_fini(skel, name, prefixlen);
}
void test_libarena(void)
void serial_test_libarena(void)
{
struct arena_alloc_reserve_args args;
struct libarena *skel;

View File

@@ -85,7 +85,7 @@ static void run_test(void)
* Run the test depending on whether LLVM can compile arena ASAN
* programs.
*/
void test_libarena_asan(void)
void serial_test_libarena_asan(void)
{
#ifdef HAS_BPF_ARENA_ASAN
run_test();

View File

@@ -0,0 +1,40 @@
// SPDX-License-Identifier: GPL-2.0
#include <vmlinux.h>
#include <bpf/bpf_helpers.h>
#include "bpf_arena_common.h"
struct {
__uint(type, BPF_MAP_TYPE_ARENA);
__uint(map_flags, BPF_F_MMAPABLE);
__uint(max_entries, 1000); /* number of pages */
#ifdef __TARGET_ARCH_arm64
__ulong(map_extra, 0x1ull << 32); /* start of mmap() region */
#else
__ulong(map_extra, 0x1ull << 44); /* start of mmap() region */
#endif
} arena SEC(".maps");
void __arena *ptr;
int alloc_cnt; /* in: pages to allocate */
long free_byte_off; /* in: byte offset within ptr to start freeing */
int free_cnt; /* in: pages to free */
SEC("syscall")
int alloc(void *ctx)
{
ptr = bpf_arena_alloc_pages(&arena, NULL, alloc_cnt, NUMA_NO_NODE, 0);
/* Success/failure is checked from user space via skel->bss->ptr. */
return 0;
}
SEC("syscall")
int free_pages(void *ctx)
{
if (!ptr)
return 1;
bpf_arena_free_pages(&arena, (char __arena *)ptr + free_byte_off, free_cnt);
return 0;
}
char _license[] SEC("license") = "GPL";