]> git.hungrycats.org Git - linux/commitdiff
nvme-tcp: fix usage of page_frag_cache
authorDmitry Bogdanov <d.bogdanov@yadro.com>
Thu, 27 Aug 2026 22:20:22 +0000 (15:20 -0700)
committerGreg Kroah-Hartman <gregkh@linuxfoundation.org>
Wed, 2 Sep 2026 12:31:46 +0000 (14:31 +0200)
commit 36ac05f7cfd59d90c597071304b14e98090d5dd1 upstream.

nvme uses page_frag_cache to preallocate PDU for each preallocated request
of block device. Block devices are created in parallel threads,
consequently page_frag_cache is used in not thread-safe manner.
That leads to incorrect refcounting of backstore pages and premature free.

That can be catched by !sendpage_ok inside network stack:

WARNING: CPU: 7 PID: 467 at ../net/core/skbuff.c:6931 skb_splice_from_iter+0xfa/0x310.
tcp_sendmsg_locked+0x782/0xce0
tcp_sendmsg+0x27/0x40
sock_sendmsg+0x8b/0xa0
nvme_tcp_try_send_cmd_pdu+0x149/0x2a0
Then random panic may occur.

Fix that by serializing the usage of page_frag_cache.

Fixes: 4e893ca81170 ("nvme_core: scan namespaces asynchronously")
Signed-off-by: Dmitry Bogdanov <d.bogdanov@yadro.com>
Signed-off-by: Daniel Wagner <wagi@kernel.org>
Signed-off-by: Keith Busch <kbusch@kernel.org>
[carlos.bilbao: adjust context in nvme_tcp_free_queue; branch predates
19bdb70c77d3 ("nvme-tcp: lockdep: use dynamic lockdep keys per socket
instance")]
Signed-off-by: Carlos Bilbao (Lambda) <carlos.bilbao@kernel.org>
Signed-off-by: Sasha Levin <sashal@kernel.org>
drivers/nvme/host/tcp.c

index afdbcff3d4821e8dad5ddcd826ff73883ef2c0cd..04c074ae8f7bb1fa8e03a812c7801e071eada53c 100644 (file)
@@ -145,6 +145,7 @@ struct nvme_tcp_queue {
 
        struct mutex            queue_lock;
        struct mutex            send_mutex;
+       struct mutex            pf_cache_lock;
        struct llist_head       req_list;
        struct list_head        send_list;
 
@@ -556,9 +557,11 @@ static int nvme_tcp_init_request(struct blk_mq_tag_set *set,
        struct nvme_tcp_queue *queue = &ctrl->queues[queue_idx];
        u8 hdgst = nvme_tcp_hdgst_len(queue);
 
+       mutex_lock(&queue->pf_cache_lock);
        req->pdu = page_frag_alloc(&queue->pf_cache,
                sizeof(struct nvme_tcp_cmd_pdu) + hdgst,
                GFP_KERNEL | __GFP_ZERO);
+       mutex_unlock(&queue->pf_cache_lock);
        if (!req->pdu)
                return -ENOMEM;
 
@@ -1423,9 +1426,11 @@ static int nvme_tcp_alloc_async_req(struct nvme_tcp_ctrl *ctrl)
        struct nvme_tcp_request *async = &ctrl->async_req;
        u8 hdgst = nvme_tcp_hdgst_len(queue);
 
+       mutex_lock(&queue->pf_cache_lock);
        async->pdu = page_frag_alloc(&queue->pf_cache,
                sizeof(struct nvme_tcp_cmd_pdu) + hdgst,
                GFP_KERNEL | __GFP_ZERO);
+       mutex_unlock(&queue->pf_cache_lock);
        if (!async->pdu)
                return -ENOMEM;
 
@@ -1453,6 +1458,7 @@ static void nvme_tcp_free_queue(struct nvme_ctrl *nctrl, int qid)
        kfree(queue->pdu);
        mutex_destroy(&queue->send_mutex);
        mutex_destroy(&queue->queue_lock);
+       mutex_destroy(&queue->pf_cache_lock);
 }
 
 static int nvme_tcp_init_connection(struct nvme_tcp_queue *queue)
@@ -1775,6 +1781,7 @@ static int nvme_tcp_alloc_queue(struct nvme_ctrl *nctrl, int qid,
        INIT_LIST_HEAD(&queue->send_list);
        mutex_init(&queue->send_mutex);
        INIT_WORK(&queue->io_work, nvme_tcp_io_work);
+       mutex_init(&queue->pf_cache_lock);
 
        if (qid > 0)
                queue->cmnd_capsule_len = nctrl->ioccsz * 16;
@@ -1906,6 +1913,7 @@ err_sock:
 err_destroy_mutex:
        mutex_destroy(&queue->send_mutex);
        mutex_destroy(&queue->queue_lock);
+       mutex_destroy(&queue->pf_cache_lock);
        return ret;
 }