[PATCH] nvmet-rdma: handle commands of established queue outside the CM handler

Shin'ichiro Kawasaki shinichiro.kawasaki at opensource.wdc.com
Tue Sep 22 05:44:47 PDT 2026


From: Shin'ichiro Kawasaki <shinichiro.kawasaki at wdc.com>

The commit 2a8513091d2f ("nvme-rdma: parallelize I/O queue allocation
and startup") modified the timing of command handling. The commands
queued before the establishment are kept in &queue->rsp_wait_list. After
the commit, the commands are executed in the rdma_cm handler, which
holds id_priv->handler_mutex. The command execution takes nvmet locks
such as nvmet_config_sem. This inverts the lock order in
nvmet_port_subsys_drop_link(), which calls rdma_destroy_id() through
nvmet_disable_port() under the nvmet_config_sem lock. This results in
the lockdep WARN below:

    WARNING: possible circular locking dependency detected
    rm/7346 is trying to acquire lock:
     (&id_priv->handler_mutex){+.+.}-{4:4}, at: rdma_destroy_id+0x17/0x20 [rdma_cm]
    but task is already holding lock:
     (nvmet_config_sem){++++}-{4:4}, at: nvmet_port_subsys_drop_link+0x52/0x340 [nvmet]

    -> #1 (nvmet_config_sem){++++}-{4:4}:
           down_read+0xa6/0x530
           nvmet_find_get_subsys+0x1f4/0x630 [nvmet]
           nvmet_ctrl_find_get+0xaa/0x320 [nvmet]
           nvmet_execute_io_connect+0x1e0/0x480 [nvmet]
           nvmet_rdma_execute_command+0x18e/0x480 [nvmet_rdma]
           nvmet_rdma_handle_command+0x428/0xb30 [nvmet_rdma]
           nvmet_rdma_cm_handler+0x8a8/0x30f0 [nvmet_rdma]
           cma_cm_event_handler+0xb0/0x3c0 [rdma_cm]
           cma_ib_handler+0x28a/0x9b0 [rdma_cm]
           cm_process_work+0x4b/0x3e0 [ib_cm]
           cm_work_handler+0x2485/0xac10 [ib_cm]

    -> #0 (&id_priv->handler_mutex){+.+.}-{4:4}:
           rdma_destroy_id+0x17/0x20 [rdma_cm]
           nvmet_rdma_disable_port+0x44/0x300 [nvmet_rdma]
           nvmet_rdma_remove_port+0x42/0x60 [nvmet_rdma]
           nvmet_disable_port+0x120/0x1f0 [nvmet]
           nvmet_port_subsys_drop_link+0x261/0x340 [nvmet]
           configfs_unlink+0x36d/0x790
           vfs_unlink+0x278/0xb40
           filename_unlinkat+0x31d/0x670
           __x64_sys_unlinkat+0xae/0x100

    other info that might help us debug this:

     Possible unsafe locking scenario:

           CPU0                    CPU1
           ----                    ----
      lock(nvmet_config_sem);
                                   lock(&id_priv->handler_mutex);
                                   lock(nvmet_config_sem);
      lock(&id_priv->handler_mutex);

     *** DEADLOCK ***

This WARN can be recreated by repeating the blktests test case nvme/004
or nvme/031 for rdma transport and rxe driver.

To avoid the lockdep WARN, handle the commands of established queue
outside the CM handler. Add a work item nvmet_rdma_establish_queue_work,
and schedule it from nvmet_rdma_queue_established() that holds
id_priv->handler_mutex. The work executes the commands, and changes the
queue state to LIVE. By doing this, the lock order is no longer
inverted.

Fixes: 2a8513091d2f ("nvme-rdma: parallelize I/O queue allocation and startup")
Signed-off-by: Shin'ichiro Kawasaki <shinichiro.kawasaki at wdc.com>
---
 drivers/nvme/target/rdma.c | 42 +++++++++++++++++++++++++++++++-------
 1 file changed, 35 insertions(+), 7 deletions(-)

diff --git a/drivers/nvme/target/rdma.c b/drivers/nvme/target/rdma.c
index ffa1e172c259..95c27eba816c 100644
--- a/drivers/nvme/target/rdma.c
+++ b/drivers/nvme/target/rdma.c
@@ -104,6 +104,7 @@ struct nvmet_rdma_queue {
 	struct nvmet_rdma_cmd	*cmds;
 
 	struct work_struct	release_work;
+	struct work_struct	establish_work;
 	struct list_head	rsp_wait_list;
 	struct list_head	rsp_wr_wait_list;
 	spinlock_t		rsp_wr_wait_lock;
@@ -170,6 +171,7 @@ static void nvmet_rdma_read_data_done(struct ib_cq *cq, struct ib_wc *wc);
 static void nvmet_rdma_write_data_done(struct ib_cq *cq, struct ib_wc *wc);
 static void nvmet_rdma_qp_event(struct ib_event *event, void *priv);
 static void nvmet_rdma_queue_disconnect(struct nvmet_rdma_queue *queue);
+static void nvmet_rdma_establish_queue_work(struct work_struct *work);
 static void nvmet_rdma_free_rsp(struct nvmet_rdma_device *ndev,
 				struct nvmet_rdma_rsp *r);
 static int nvmet_rdma_alloc_rsp(struct nvmet_rdma_device *ndev,
@@ -1380,6 +1382,7 @@ static void nvmet_rdma_free_queue(struct nvmet_rdma_queue *queue)
 {
 	pr_debug("freeing queue %d\n", queue->idx);
 
+	cancel_work_sync(&queue->establish_work);
 	nvmet_sq_destroy(&queue->nvme_sq);
 	nvmet_cq_put(&queue->nvme_cq);
 
@@ -1481,6 +1484,7 @@ nvmet_rdma_alloc_queue(struct nvmet_rdma_device *ndev,
 	 * inside a CM callback would trigger a deadlock. (great API design..)
 	 */
 	INIT_WORK(&queue->release_work, nvmet_rdma_release_queue_work);
+	INIT_WORK(&queue->establish_work, nvmet_rdma_establish_queue_work);
 	queue->dev = ndev;
 	queue->cm_id = cm_id;
 	queue->port = port->nport;
@@ -1658,18 +1662,20 @@ static int nvmet_rdma_queue_connect(struct rdma_cm_id *cm_id,
 	return ret;
 }
 
-static void nvmet_rdma_queue_established(struct nvmet_rdma_queue *queue)
+/*
+ * Execute the commands which were received before the queue became LIVE and
+ * then move the queue to the LIVE state.
+ */
+static void nvmet_rdma_establish_queue_work(struct work_struct *work)
 {
+	struct nvmet_rdma_queue *queue =
+		container_of(work, struct nvmet_rdma_queue, establish_work);
 	unsigned long flags;
 
 	spin_lock_irqsave(&queue->state_lock, flags);
-	if (queue->state != NVMET_RDMA_Q_CONNECTING) {
-		pr_warn("trying to establish a connected queue\n");
-		goto out_unlock;
-	}
-	queue->state = NVMET_RDMA_Q_LIVE;
 
-	while (!list_empty(&queue->rsp_wait_list)) {
+	while (queue->state == NVMET_RDMA_Q_CONNECTING &&
+	       !list_empty(&queue->rsp_wait_list)) {
 		struct nvmet_rdma_rsp *cmd;
 
 		cmd = list_first_entry(&queue->rsp_wait_list,
@@ -1681,6 +1687,28 @@ static void nvmet_rdma_queue_established(struct nvmet_rdma_queue *queue)
 		spin_lock_irqsave(&queue->state_lock, flags);
 	}
 
+	if (queue->state == NVMET_RDMA_Q_CONNECTING)
+		queue->state = NVMET_RDMA_Q_LIVE;
+
+	spin_unlock_irqrestore(&queue->state_lock, flags);
+}
+
+static void nvmet_rdma_queue_established(struct nvmet_rdma_queue *queue)
+{
+	unsigned long flags;
+
+	spin_lock_irqsave(&queue->state_lock, flags);
+	if (queue->state != NVMET_RDMA_Q_CONNECTING) {
+		pr_warn("trying to establish a connected queue\n");
+		goto out_unlock;
+	}
+
+	/*
+	 * Defer the command execution and the transition to the LIVE state to
+	 * nvmet_wq. This avoids acquiring nvmet locks here.
+	 */
+	queue_work(nvmet_wq, &queue->establish_work);
+
 out_unlock:
 	spin_unlock_irqrestore(&queue->state_lock, flags);
 }
-- 
2.54.0




More information about the Linux-nvme mailing list