]> git.hungrycats.org Git - linux/commitdiff
xfrm: bound nat keepalive state collection
authorZihan Xi <zihanx@nebusec.ai>
Mon, 17 Aug 2026 19:09:56 +0000 (19:09 +0000)
committerGreg Kroah-Hartman <gregkh@linuxfoundation.org>
Wed, 2 Sep 2026 12:31:49 +0000 (14:31 +0200)
commit 4e9442ce551ebd84b52ad649df721e2dc28af95a upstream.

The v1 nat keepalive fix allocates a GFP_ATOMIC object for every state
while collecting references for phase two. This makes the worker's
temporary memory use depend on the number of states and lets -ENOMEM abort
the scan.

Replace the allocated list with a fixed-size batch. When the batch is full,
return a private walk status so xfrm_state_walk() leaves a cursor; drain
the references after the walk releases xfrm_state_lock and resume from
the cursor. This bounds temporary memory use and avoids the allocation
failure path.

The v1 fix also moved nat_keepalive_send() out of the walk callback. Keep
the phase-two drain BH-disabled, as required by local_lock_nested_bh()
used by the keepalive sockets.

Fixes: 763fe700b7c5 ("xfrm: avoid lock inversion in nat keepalive work")
Cc: stable@vger.kernel.org
Cc: Eyal Birger <eyal.birger@gmail.com>
Reported-by: Vega <vega@nebusec.ai>
Assisted-by: Codex:gpt-5.4
Signed-off-by: Zihan Xi <zihanx@nebusec.ai>
Signed-off-by: Steffen Klassert <steffen.klassert@secunet.com>
Signed-off-by: Greg Kroah-Hartman <gregkh@linuxfoundation.org>
net/xfrm/xfrm_nat_keepalive.c

index 03e672d339dabbd6286c873e3b03a3d6d2f94135..22ca3dc1508b64170c03694a4d85220faa043e65 100644 (file)
@@ -155,32 +155,30 @@ static void nat_keepalive_send(struct nat_keepalive *ka)
        }
 }
 
+enum {
+       NAT_KEEPALIVE_BATCH_SIZE = 16,
+       NAT_KEEPALIVE_BATCH_FULL = 1,
+};
+
 struct nat_keepalive_work_ctx {
-       struct list_head states;
+       struct xfrm_state *batch[NAT_KEEPALIVE_BATCH_SIZE];
+       unsigned int nr;
        time64_t next_run;
        time64_t now;
 };
 
-struct nat_keepalive_state {
-       struct list_head list;
-       struct xfrm_state *x;
-};
-
 static int nat_keepalive_work_collect(struct xfrm_state *x, int count, void *ptr)
 {
        struct nat_keepalive_work_ctx *ctx = ptr;
-       struct nat_keepalive_state *state;
 
        if (!READ_ONCE(x->nat_keepalive_interval))
                return 0;
 
-       state = kmalloc_obj(*state, GFP_ATOMIC);
-       if (!state)
-               return -ENOMEM;
+       if (ctx->nr == ARRAY_SIZE(ctx->batch))
+               return NAT_KEEPALIVE_BATCH_FULL;
 
        xfrm_state_hold(x);
-       state->x = x;
-       list_add_tail(&state->list, &ctx->states);
+       ctx->batch[ctx->nr++] = x;
        return 0;
 }
 
@@ -226,29 +224,27 @@ out:
 
 static void nat_keepalive_work(struct work_struct *work)
 {
-       struct nat_keepalive_state *state, *tmp;
        struct nat_keepalive_work_ctx ctx;
        struct xfrm_state_walk walk;
        struct net *net;
-       int err;
+       int err, i;
 
-       INIT_LIST_HEAD(&ctx.states);
        ctx.next_run = 0;
        ctx.now = ktime_get_real_seconds();
 
        net = container_of(work, struct net, xfrm.nat_keepalive_work.work);
        xfrm_state_walk_init(&walk, IPPROTO_ESP, NULL);
-       err = xfrm_state_walk(net, &walk, nat_keepalive_work_collect, &ctx);
+       do {
+               ctx.nr = 0;
+               err = xfrm_state_walk(net, &walk, nat_keepalive_work_collect, &ctx);
+               local_bh_disable();
+               for (i = 0; i < ctx.nr; i++) {
+                       nat_keepalive_work_single(ctx.batch[i], &ctx);
+                       xfrm_state_put(ctx.batch[i]);
+               }
+               local_bh_enable();
+       } while (err == NAT_KEEPALIVE_BATCH_FULL);
        xfrm_state_walk_done(&walk, net);
-       list_for_each_entry_safe(state, tmp, &ctx.states, list) {
-               nat_keepalive_work_single(state->x, &ctx);
-               xfrm_state_put(state->x);
-               kfree(state);
-       }
-       if (err == -ENOMEM) {
-               schedule_delayed_work(&net->xfrm.nat_keepalive_work, 0);
-               return;
-       }
        if (ctx.next_run)
                schedule_delayed_work(&net->xfrm.nat_keepalive_work,
                                      (ctx.next_run - ctx.now) * HZ);