[PATCH] nvmet-rdma: handle commands of established queue outside the CM handler
Shin'ichiro Kawasaki
shinichiro.kawasaki at opensource.wdc.com
Tue Sep 22 05:44:47 PDT 2026
From: Shin'ichiro Kawasaki <shinichiro.kawasaki at wdc.com>
The commit 2a8513091d2f ("nvme-rdma: parallelize I/O queue allocation
and startup") modified the timing of command handling. The commands
queued before the establishment are kept in &queue->rsp_wait_list. After
the commit, the commands are executed in the rdma_cm handler, which
holds id_priv->handler_mutex. The command execution takes nvmet locks
such as nvmet_config_sem. This inverts the lock order in
nvmet_port_subsys_drop_link(), which calls rdma_destroy_id() through
nvmet_disable_port() under the nvmet_config_sem lock. This results in
the lockdep WARN below:
WARNING: possible circular locking dependency detected
rm/7346 is trying to acquire lock:
(&id_priv->handler_mutex){+.+.}-{4:4}, at: rdma_destroy_id+0x17/0x20 [rdma_cm]
but task is already holding lock:
(nvmet_config_sem){++++}-{4:4}, at: nvmet_port_subsys_drop_link+0x52/0x340 [nvmet]
-> #1 (nvmet_config_sem){++++}-{4:4}:
down_read+0xa6/0x530
nvmet_find_get_subsys+0x1f4/0x630 [nvmet]
nvmet_ctrl_find_get+0xaa/0x320 [nvmet]
nvmet_execute_io_connect+0x1e0/0x480 [nvmet]
nvmet_rdma_execute_command+0x18e/0x480 [nvmet_rdma]
nvmet_rdma_handle_command+0x428/0xb30 [nvmet_rdma]
nvmet_rdma_cm_handler+0x8a8/0x30f0 [nvmet_rdma]
cma_cm_event_handler+0xb0/0x3c0 [rdma_cm]
cma_ib_handler+0x28a/0x9b0 [rdma_cm]
cm_process_work+0x4b/0x3e0 [ib_cm]
cm_work_handler+0x2485/0xac10 [ib_cm]
-> #0 (&id_priv->handler_mutex){+.+.}-{4:4}:
rdma_destroy_id+0x17/0x20 [rdma_cm]
nvmet_rdma_disable_port+0x44/0x300 [nvmet_rdma]
nvmet_rdma_remove_port+0x42/0x60 [nvmet_rdma]
nvmet_disable_port+0x120/0x1f0 [nvmet]
nvmet_port_subsys_drop_link+0x261/0x340 [nvmet]
configfs_unlink+0x36d/0x790
vfs_unlink+0x278/0xb40
filename_unlinkat+0x31d/0x670
__x64_sys_unlinkat+0xae/0x100
other info that might help us debug this:
Possible unsafe locking scenario:
CPU0 CPU1
---- ----
lock(nvmet_config_sem);
lock(&id_priv->handler_mutex);
lock(nvmet_config_sem);
lock(&id_priv->handler_mutex);
*** DEADLOCK ***
This WARN can be recreated by repeating the blktests test case nvme/004
or nvme/031 for rdma transport and rxe driver.
To avoid the lockdep WARN, handle the commands of established queue
outside the CM handler. Add a work item nvmet_rdma_establish_queue_work,
and schedule it from nvmet_rdma_queue_established() that holds
id_priv->handler_mutex. The work executes the commands, and changes the
queue state to LIVE. By doing this, the lock order is no longer
inverted.
Fixes: 2a8513091d2f ("nvme-rdma: parallelize I/O queue allocation and startup")
Signed-off-by: Shin'ichiro Kawasaki <shinichiro.kawasaki at wdc.com>
---
drivers/nvme/target/rdma.c | 42 +++++++++++++++++++++++++++++++-------
1 file changed, 35 insertions(+), 7 deletions(-)
diff --git a/drivers/nvme/target/rdma.c b/drivers/nvme/target/rdma.c
index ffa1e172c259..95c27eba816c 100644
--- a/drivers/nvme/target/rdma.c
+++ b/drivers/nvme/target/rdma.c
@@ -104,6 +104,7 @@ struct nvmet_rdma_queue {
struct nvmet_rdma_cmd *cmds;
struct work_struct release_work;
+ struct work_struct establish_work;
struct list_head rsp_wait_list;
struct list_head rsp_wr_wait_list;
spinlock_t rsp_wr_wait_lock;
@@ -170,6 +171,7 @@ static void nvmet_rdma_read_data_done(struct ib_cq *cq, struct ib_wc *wc);
static void nvmet_rdma_write_data_done(struct ib_cq *cq, struct ib_wc *wc);
static void nvmet_rdma_qp_event(struct ib_event *event, void *priv);
static void nvmet_rdma_queue_disconnect(struct nvmet_rdma_queue *queue);
+static void nvmet_rdma_establish_queue_work(struct work_struct *work);
static void nvmet_rdma_free_rsp(struct nvmet_rdma_device *ndev,
struct nvmet_rdma_rsp *r);
static int nvmet_rdma_alloc_rsp(struct nvmet_rdma_device *ndev,
@@ -1380,6 +1382,7 @@ static void nvmet_rdma_free_queue(struct nvmet_rdma_queue *queue)
{
pr_debug("freeing queue %d\n", queue->idx);
+ cancel_work_sync(&queue->establish_work);
nvmet_sq_destroy(&queue->nvme_sq);
nvmet_cq_put(&queue->nvme_cq);
@@ -1481,6 +1484,7 @@ nvmet_rdma_alloc_queue(struct nvmet_rdma_device *ndev,
* inside a CM callback would trigger a deadlock. (great API design..)
*/
INIT_WORK(&queue->release_work, nvmet_rdma_release_queue_work);
+ INIT_WORK(&queue->establish_work, nvmet_rdma_establish_queue_work);
queue->dev = ndev;
queue->cm_id = cm_id;
queue->port = port->nport;
@@ -1658,18 +1662,20 @@ static int nvmet_rdma_queue_connect(struct rdma_cm_id *cm_id,
return ret;
}
-static void nvmet_rdma_queue_established(struct nvmet_rdma_queue *queue)
+/*
+ * Execute the commands which were received before the queue became LIVE and
+ * then move the queue to the LIVE state.
+ */
+static void nvmet_rdma_establish_queue_work(struct work_struct *work)
{
+ struct nvmet_rdma_queue *queue =
+ container_of(work, struct nvmet_rdma_queue, establish_work);
unsigned long flags;
spin_lock_irqsave(&queue->state_lock, flags);
- if (queue->state != NVMET_RDMA_Q_CONNECTING) {
- pr_warn("trying to establish a connected queue\n");
- goto out_unlock;
- }
- queue->state = NVMET_RDMA_Q_LIVE;
- while (!list_empty(&queue->rsp_wait_list)) {
+ while (queue->state == NVMET_RDMA_Q_CONNECTING &&
+ !list_empty(&queue->rsp_wait_list)) {
struct nvmet_rdma_rsp *cmd;
cmd = list_first_entry(&queue->rsp_wait_list,
@@ -1681,6 +1687,28 @@ static void nvmet_rdma_queue_established(struct nvmet_rdma_queue *queue)
spin_lock_irqsave(&queue->state_lock, flags);
}
+ if (queue->state == NVMET_RDMA_Q_CONNECTING)
+ queue->state = NVMET_RDMA_Q_LIVE;
+
+ spin_unlock_irqrestore(&queue->state_lock, flags);
+}
+
+static void nvmet_rdma_queue_established(struct nvmet_rdma_queue *queue)
+{
+ unsigned long flags;
+
+ spin_lock_irqsave(&queue->state_lock, flags);
+ if (queue->state != NVMET_RDMA_Q_CONNECTING) {
+ pr_warn("trying to establish a connected queue\n");
+ goto out_unlock;
+ }
+
+ /*
+ * Defer the command execution and the transition to the LIVE state to
+ * nvmet_wq. This avoids acquiring nvmet locks here.
+ */
+ queue_work(nvmet_wq, &queue->establish_work);
+
out_unlock:
spin_unlock_irqrestore(&queue->state_lock, flags);
}
--
2.54.0
More information about the Linux-nvme
mailing list