]> git.hungrycats.org Git - linux/commitdiff
xfrm: avoid lock inversion in nat keepalive work
authorZihan Xi <xizh2024@lzu.edu.cn>
Tue, 21 Jul 2026 15:25:42 +0000 (23:25 +0800)
committerGreg Kroah-Hartman <gregkh@linuxfoundation.org>
Wed, 2 Sep 2026 12:31:49 +0000 (14:31 +0200)
commit 763fe700b7c58ad64fe5202c5638848244dd4127 upstream.

nat_keepalive_work() walks the state table while xfrm_state_walk()
holds net->xfrm.xfrm_state_lock. Its callback then acquires x->lock,
which conflicts with the delete path taking the same locks in reverse
order via xfrm_state_delete() and __xfrm_state_delete(). This creates
an AB-BA deadlock that is reported by lockdep when a NAT keepalive
worker races with SA deletion.

Fix this by splitting the keepalive walk into two phases. First,
collect the candidate states while the walk holds xfrm_state_lock and
take a reference on each state. Then, after the walk completes, process
each collected state and acquire x->lock without nesting it under
xfrm_state_lock.

Fixes: f531d13bdfe3 ("xfrm: support sending NAT keepalives in ESP in UDP states")
Cc: stable@vger.kernel.org
Reported-by: Vega <vega@nebusec.ai>
Assisted-by: Codex:gpt-5.4
Signed-off-by: Zihan Xi <xizh2024@lzu.edu.cn>
Signed-off-by: Ren Wei <enjou1224z@gmail.com>
Signed-off-by: Steffen Klassert <steffen.klassert@secunet.com>
Signed-off-by: Greg Kroah-Hartman <gregkh@linuxfoundation.org>
net/xfrm/xfrm_nat_keepalive.c

index f50b1f48f2ed92c3c42974bdff23c00d68d04bc9..03e672d339dabbd6286c873e3b03a3d6d2f94135 100644 (file)
@@ -156,24 +156,51 @@ static void nat_keepalive_send(struct nat_keepalive *ka)
 }
 
 struct nat_keepalive_work_ctx {
+       struct list_head states;
        time64_t next_run;
        time64_t now;
 };
 
-static int nat_keepalive_work_single(struct xfrm_state *x, int count, void *ptr)
+struct nat_keepalive_state {
+       struct list_head list;
+       struct xfrm_state *x;
+};
+
+static int nat_keepalive_work_collect(struct xfrm_state *x, int count, void *ptr)
 {
        struct nat_keepalive_work_ctx *ctx = ptr;
+       struct nat_keepalive_state *state;
+
+       if (!READ_ONCE(x->nat_keepalive_interval))
+               return 0;
+
+       state = kmalloc_obj(*state, GFP_ATOMIC);
+       if (!state)
+               return -ENOMEM;
+
+       xfrm_state_hold(x);
+       state->x = x;
+       list_add_tail(&state->list, &ctx->states);
+       return 0;
+}
+
+static void nat_keepalive_work_single(struct xfrm_state *x,
+                                     struct nat_keepalive_work_ctx *ctx)
+{
        bool send_keepalive = false;
        struct nat_keepalive ka;
-       time64_t next_run;
+       time64_t next_run = 0;
        u32 interval;
        int delta;
 
+       spin_lock_bh(&x->lock);
+
+       if (x->km.state == XFRM_STATE_DEAD)
+               goto out;
+
        interval = x->nat_keepalive_interval;
        if (!interval)
-               return 0;
-
-       spin_lock(&x->lock);
+               goto out;
 
        delta = (int)(ctx->now - x->lastused);
        if (delta < interval) {
@@ -187,29 +214,41 @@ static int nat_keepalive_work_single(struct xfrm_state *x, int count, void *ptr)
                send_keepalive = true;
        }
 
-       spin_unlock(&x->lock);
+out:
+       spin_unlock_bh(&x->lock);
 
        if (send_keepalive)
                nat_keepalive_send(&ka);
 
-       if (!ctx->next_run || next_run < ctx->next_run)
+       if (next_run && (!ctx->next_run || next_run < ctx->next_run))
                ctx->next_run = next_run;
-       return 0;
 }
 
 static void nat_keepalive_work(struct work_struct *work)
 {
+       struct nat_keepalive_state *state, *tmp;
        struct nat_keepalive_work_ctx ctx;
        struct xfrm_state_walk walk;
        struct net *net;
+       int err;
 
+       INIT_LIST_HEAD(&ctx.states);
        ctx.next_run = 0;
        ctx.now = ktime_get_real_seconds();
 
        net = container_of(work, struct net, xfrm.nat_keepalive_work.work);
        xfrm_state_walk_init(&walk, IPPROTO_ESP, NULL);
-       xfrm_state_walk(net, &walk, nat_keepalive_work_single, &ctx);
+       err = xfrm_state_walk(net, &walk, nat_keepalive_work_collect, &ctx);
        xfrm_state_walk_done(&walk, net);
+       list_for_each_entry_safe(state, tmp, &ctx.states, list) {
+               nat_keepalive_work_single(state->x, &ctx);
+               xfrm_state_put(state->x);
+               kfree(state);
+       }
+       if (err == -ENOMEM) {
+               schedule_delayed_work(&net->xfrm.nat_keepalive_work, 0);
+               return;
+       }
        if (ctx.next_run)
                schedule_delayed_work(&net->xfrm.nat_keepalive_work,
                                      (ctx.next_run - ctx.now) * HZ);