xfrm: avoid lock inversion in nat keepalive work

nat_keepalive_work() walks the state table while xfrm_state_walk()
holds net->xfrm.xfrm_state_lock. Its callback then acquires x->lock,
which conflicts with the delete path taking the same locks in reverse
order via xfrm_state_delete() and __xfrm_state_delete(). This creates
an AB-BA deadlock that is reported by lockdep when a NAT keepalive
worker races with SA deletion.

Fix this by splitting the keepalive walk into two phases. First,
collect the candidate states while the walk holds xfrm_state_lock and
take a reference on each state. Then, after the walk completes, process
each collected state and acquire x->lock without nesting it under
xfrm_state_lock.

Fixes: f531d13bdf ("xfrm: support sending NAT keepalives in ESP in UDP states")
Cc: stable@vger.kernel.org
Reported-by: Vega <vega@nebusec.ai>
Assisted-by: Codex:gpt-5.4
Signed-off-by: Zihan Xi <xizh2024@lzu.edu.cn>
Signed-off-by: Ren Wei <enjou1224z@gmail.com>
Signed-off-by: Steffen Klassert <steffen.klassert@secunet.com>
This commit is contained in:
Zihan Xi
2026-07-21 23:25:42 +08:00
committed by Steffen Klassert
parent e1d7c5ac1c
commit 763fe700b7

View File

@@ -156,24 +156,51 @@ static void nat_keepalive_send(struct nat_keepalive *ka)
}
struct nat_keepalive_work_ctx {
struct list_head states;
time64_t next_run;
time64_t now;
};
static int nat_keepalive_work_single(struct xfrm_state *x, int count, void *ptr)
struct nat_keepalive_state {
struct list_head list;
struct xfrm_state *x;
};
static int nat_keepalive_work_collect(struct xfrm_state *x, int count, void *ptr)
{
struct nat_keepalive_work_ctx *ctx = ptr;
struct nat_keepalive_state *state;
if (!READ_ONCE(x->nat_keepalive_interval))
return 0;
state = kmalloc_obj(*state, GFP_ATOMIC);
if (!state)
return -ENOMEM;
xfrm_state_hold(x);
state->x = x;
list_add_tail(&state->list, &ctx->states);
return 0;
}
static void nat_keepalive_work_single(struct xfrm_state *x,
struct nat_keepalive_work_ctx *ctx)
{
bool send_keepalive = false;
struct nat_keepalive ka;
time64_t next_run;
time64_t next_run = 0;
u32 interval;
int delta;
spin_lock_bh(&x->lock);
if (x->km.state == XFRM_STATE_DEAD)
goto out;
interval = x->nat_keepalive_interval;
if (!interval)
return 0;
spin_lock(&x->lock);
goto out;
delta = (int)(ctx->now - x->lastused);
if (delta < interval) {
@@ -187,29 +214,41 @@ static int nat_keepalive_work_single(struct xfrm_state *x, int count, void *ptr)
send_keepalive = true;
}
spin_unlock(&x->lock);
out:
spin_unlock_bh(&x->lock);
if (send_keepalive)
nat_keepalive_send(&ka);
if (!ctx->next_run || next_run < ctx->next_run)
if (next_run && (!ctx->next_run || next_run < ctx->next_run))
ctx->next_run = next_run;
return 0;
}
static void nat_keepalive_work(struct work_struct *work)
{
struct nat_keepalive_state *state, *tmp;
struct nat_keepalive_work_ctx ctx;
struct xfrm_state_walk walk;
struct net *net;
int err;
INIT_LIST_HEAD(&ctx.states);
ctx.next_run = 0;
ctx.now = ktime_get_real_seconds();
net = container_of(work, struct net, xfrm.nat_keepalive_work.work);
xfrm_state_walk_init(&walk, IPPROTO_ESP, NULL);
xfrm_state_walk(net, &walk, nat_keepalive_work_single, &ctx);
err = xfrm_state_walk(net, &walk, nat_keepalive_work_collect, &ctx);
xfrm_state_walk_done(&walk, net);
list_for_each_entry_safe(state, tmp, &ctx.states, list) {
nat_keepalive_work_single(state->x, &ctx);
xfrm_state_put(state->x);
kfree(state);
}
if (err == -ENOMEM) {
schedule_delayed_work(&net->xfrm.nat_keepalive_work, 0);
return;
}
if (ctx.next_run)
schedule_delayed_work(&net->xfrm.nat_keepalive_work,
(ctx.next_run - ctx.now) * HZ);