summaryrefslogtreecommitdiff
diff options
context:
space:
mode:
authorZihan Xi <xizh2024@lzu.edu.cn>2026-07-21 23:25:42 +0800
committerGreg Kroah-Hartman <gregkh@linuxfoundation.org>2026-09-02 14:32:37 +0200
commit89ef3a2e1e4682ab82b0455ce113f8c39fb9e50d (patch)
treea8afbf201d92d0801962b8193bc2a2e1c4ee352a
parent7911e0236616487b89db6bd3ba3f408abd10eb23 (diff)
downloadlinux-stable-89ef3a2e1e4682ab82b0455ce113f8c39fb9e50d.tar.gz
linux-stable-89ef3a2e1e4682ab82b0455ce113f8c39fb9e50d.zip
xfrm: avoid lock inversion in nat keepalive work
commit 763fe700b7c58ad64fe5202c5638848244dd4127 upstream. nat_keepalive_work() walks the state table while xfrm_state_walk() holds net->xfrm.xfrm_state_lock. Its callback then acquires x->lock, which conflicts with the delete path taking the same locks in reverse order via xfrm_state_delete() and __xfrm_state_delete(). This creates an AB-BA deadlock that is reported by lockdep when a NAT keepalive worker races with SA deletion. Fix this by splitting the keepalive walk into two phases. First, collect the candidate states while the walk holds xfrm_state_lock and take a reference on each state. Then, after the walk completes, process each collected state and acquire x->lock without nesting it under xfrm_state_lock. Fixes: f531d13bdfe3 ("xfrm: support sending NAT keepalives in ESP in UDP states") Cc: stable@vger.kernel.org Reported-by: Vega <vega@nebusec.ai> Assisted-by: Codex:gpt-5.4 Signed-off-by: Zihan Xi <xizh2024@lzu.edu.cn> Signed-off-by: Ren Wei <enjou1224z@gmail.com> Signed-off-by: Steffen Klassert <steffen.klassert@secunet.com> Signed-off-by: Greg Kroah-Hartman <gregkh@linuxfoundation.org>
-rw-r--r--net/xfrm/xfrm_nat_keepalive.c57
1 files changed, 48 insertions, 9 deletions
diff --git a/net/xfrm/xfrm_nat_keepalive.c b/net/xfrm/xfrm_nat_keepalive.c
index eb1b6f67739e..8679c68c10a1 100644
--- a/net/xfrm/xfrm_nat_keepalive.c
+++ b/net/xfrm/xfrm_nat_keepalive.c
@@ -156,24 +156,51 @@ static void nat_keepalive_send(struct nat_keepalive *ka)
}
struct nat_keepalive_work_ctx {
+ struct list_head states;
time64_t next_run;
time64_t now;
};
-static int nat_keepalive_work_single(struct xfrm_state *x, int count, void *ptr)
+struct nat_keepalive_state {
+ struct list_head list;
+ struct xfrm_state *x;
+};
+
+static int nat_keepalive_work_collect(struct xfrm_state *x, int count, void *ptr)
{
struct nat_keepalive_work_ctx *ctx = ptr;
+ struct nat_keepalive_state *state;
+
+ if (!READ_ONCE(x->nat_keepalive_interval))
+ return 0;
+
+ state = kmalloc_obj(*state, GFP_ATOMIC);
+ if (!state)
+ return -ENOMEM;
+
+ xfrm_state_hold(x);
+ state->x = x;
+ list_add_tail(&state->list, &ctx->states);
+ return 0;
+}
+
+static void nat_keepalive_work_single(struct xfrm_state *x,
+ struct nat_keepalive_work_ctx *ctx)
+{
bool send_keepalive = false;
struct nat_keepalive ka;
- time64_t next_run;
+ time64_t next_run = 0;
u32 interval;
int delta;
+ spin_lock_bh(&x->lock);
+
+ if (x->km.state == XFRM_STATE_DEAD)
+ goto out;
+
interval = x->nat_keepalive_interval;
if (!interval)
- return 0;
-
- spin_lock(&x->lock);
+ goto out;
delta = (int)(ctx->now - x->lastused);
if (delta < interval) {
@@ -187,29 +214,41 @@ static int nat_keepalive_work_single(struct xfrm_state *x, int count, void *ptr)
send_keepalive = true;
}
- spin_unlock(&x->lock);
+out:
+ spin_unlock_bh(&x->lock);
if (send_keepalive)
nat_keepalive_send(&ka);
- if (!ctx->next_run || next_run < ctx->next_run)
+ if (next_run && (!ctx->next_run || next_run < ctx->next_run))
ctx->next_run = next_run;
- return 0;
}
static void nat_keepalive_work(struct work_struct *work)
{
+ struct nat_keepalive_state *state, *tmp;
struct nat_keepalive_work_ctx ctx;
struct xfrm_state_walk walk;
struct net *net;
+ int err;
+ INIT_LIST_HEAD(&ctx.states);
ctx.next_run = 0;
ctx.now = ktime_get_real_seconds();
net = container_of(work, struct net, xfrm.nat_keepalive_work.work);
xfrm_state_walk_init(&walk, IPPROTO_ESP, NULL);
- xfrm_state_walk(net, &walk, nat_keepalive_work_single, &ctx);
+ err = xfrm_state_walk(net, &walk, nat_keepalive_work_collect, &ctx);
xfrm_state_walk_done(&walk, net);
+ list_for_each_entry_safe(state, tmp, &ctx.states, list) {
+ nat_keepalive_work_single(state->x, &ctx);
+ xfrm_state_put(state->x);
+ kfree(state);
+ }
+ if (err == -ENOMEM) {
+ schedule_delayed_work(&net->xfrm.nat_keepalive_work, 0);
+ return;
+ }
if (ctx.next_run)
schedule_delayed_work(&net->xfrm.nat_keepalive_work,
(ctx.next_run - ctx.now) * HZ);