[PATCH 4/6] nvme-mpath: support controller crd when failing over request

Sagi Grimberg <[email protected]>
Newsgroups org.infradead.lists.linux-nvme
Message-ID <[email protected]>
When failing over a request (due to a path based status) we should
repect controller crd returned in the nvme completion as much as
possible. Hence we want to delay the failover command execution by
the controller crdt.

We allocate a new nvme_mpath_failover_timer referencing the request
stolen bios in a staging list, and when the command retry delay expires,
and only then the bios are moved to the mpath head requeue list which is
immediately kicked to re-submit these bios. If we failed to allocate
a fot, we fallback to the existing behavior.

Given that we now have a new staging list for mpath devices, we drain
them when removing the device.

Signed-off-by: Sagi Grimberg <[email protected]>
---
 drivers/nvme/host/multipath.c | 96 +++++++++++++++++++++++++++++++++--
 drivers/nvme/host/nvme.h      |  1 +
 2 files changed, 93 insertions(+), 4 deletions(-)

diff --git a/drivers/nvme/host/multipath.c b/drivers/nvme/host/multipath.c
index b5501217303c..959dd1e05a2d 100644
--- a/drivers/nvme/host/multipath.c
+++ b/drivers/nvme/host/multipath.c
@@ -9,6 +9,13 @@
 #include <trace/events/block.h>
 #include "nvme.h"
 
+struct nvme_mpath_failover_timer {
+	struct list_head	entry;
+	struct nvme_ns_head	*head;
+	struct bio_list		bios;
+	struct timer_list	timer;
+};
+
 bool multipath = true;
 static bool multipath_always_on;
 
@@ -144,10 +151,48 @@ void nvme_mpath_start_freeze(struct nvme_subsystem *subsys)
 			blk_freeze_queue_start(h->disk->queue);
 }
 
+static void nvme_mpath_failover_timer_fn(struct timer_list *t)
+{
+	struct nvme_mpath_failover_timer *fot = timer_container_of(fot, t, timer);
+	struct nvme_ns_head *head = fot->head;
+	unsigned long flags;
+
+	spin_lock_irqsave(&head->requeue_lock, flags);
+	if (list_empty(&fot->entry)) {
+		spin_unlock_irqrestore(&head->requeue_lock, flags);
+		return;
+	}
+
+	list_del_init(&fot->entry);
+	if (fot->bios.head)
+		bio_list_merge(&head->requeue_list, &fot->bios);
+	spin_unlock_irqrestore(&head->requeue_lock, flags);
+	kblockd_schedule_work(&head->requeue_work);
+	kfree(fot);
+}
+
+static struct nvme_mpath_failover_timer *
+nvme_mpath_alloc_failover_timer(struct nvme_ns_head *head)
+{
+	struct nvme_mpath_failover_timer *fot;
+
+	fot = kzalloc(sizeof(*fot), GFP_ATOMIC);
+	if (!fot)
+		goto out;
+	fot->head = head;
+	bio_list_init(&fot->bios);
+	INIT_LIST_HEAD(&fot->entry);
+	timer_setup(&fot->timer, nvme_mpath_failover_timer_fn, 0);
+out:
+	return fot;
+}
+
 void nvme_failover_req(struct request *req)
 {
 	struct nvme_ns *ns = req->q->queuedata;
 	u16 status = nvme_req(req)->status & NVME_SCT_SC_MASK;
+	struct nvme_mpath_failover_timer *fot = NULL;
+	unsigned int delay;
 	unsigned long flags;
 	struct bio *bio;
 
@@ -167,13 +212,27 @@ void nvme_failover_req(struct request *req)
 	for (bio = req->bio; bio; bio = bio->bi_next)
 		bio_set_dev(bio, ns->head->disk->part0);
 
-	spin_lock_irqsave(&ns->head->requeue_lock, flags);
-	blk_steal_bios(&ns->head->requeue_list, req);
-	spin_unlock_irqrestore(&ns->head->requeue_lock, flags);
+	delay = nvme_crd_msecs(nvme_req(req));
+	if (delay) {
+		fot = nvme_mpath_alloc_failover_timer(ns->head);
+		if (fot) {
+			blk_steal_bios(&fot->bios, req);
+			spin_lock_irqsave(&ns->head->requeue_lock, flags);
+			list_add_tail(&fot->entry, &ns->head->fots);
+			spin_unlock_irqrestore(&ns->head->requeue_lock, flags);
+			mod_timer(&fot->timer, jiffies + msecs_to_jiffies(delay));
+		}
+	}
+	/* no CRD or timer allocation failed, fallback to immediate failover */
+	if (!fot) {
+		spin_lock_irqsave(&ns->head->requeue_lock, flags);
+		blk_steal_bios(&ns->head->requeue_list, req);
+		spin_unlock_irqrestore(&ns->head->requeue_lock, flags);
+		kblockd_schedule_work(&ns->head->requeue_work);
+	}
 
 	nvme_req(req)->status = 0;
 	nvme_end_req(req);
-	kblockd_schedule_work(&ns->head->requeue_work);
 }
 
 void nvme_mpath_start_request(struct request *rq)
@@ -689,6 +748,32 @@ static void nvme_requeue_work(struct work_struct *work)
 	}
 }
 
+static void nvme_mpath_drain_failover_timers(struct nvme_ns_head *head)
+{
+	struct nvme_mpath_failover_timer *fot;
+	unsigned long flags;
+
+	while (true) {
+		spin_lock_irqsave(&head->requeue_lock, flags);
+		fot = list_first_entry_or_null(&head->fots,
+			struct nvme_mpath_failover_timer, entry);
+		if (!fot) {
+			spin_unlock_irqrestore(&head->requeue_lock, flags);
+			return;
+		}
+		list_del_init(&fot->entry);
+		spin_unlock_irqrestore(&head->requeue_lock, flags);
+
+		timer_delete_sync(&fot->timer);
+
+		spin_lock_irqsave(&head->requeue_lock, flags);
+		if (fot->bios.head)
+			bio_list_merge(&head->requeue_list, &fot->bios);
+		spin_unlock_irqrestore(&head->requeue_lock, flags);
+		kfree(fot);
+	}
+}
+
 static void nvme_remove_head(struct nvme_ns_head *head)
 {
 	if (test_and_clear_bit(NVME_NSHEAD_DISK_LIVE, &head->flags)) {
@@ -696,6 +781,7 @@ static void nvme_remove_head(struct nvme_ns_head *head)
 		 * requeue I/O after NVME_NSHEAD_DISK_LIVE has been cleared
 		 * to allow multipath to fail all I/O.
 		 */
+		nvme_mpath_drain_failover_timers(head);
 		kblockd_schedule_work(&head->requeue_work);
 
 		if (test_and_clear_bit(NVME_NSHEAD_CDEV_LIVE, &head->flags))
@@ -731,6 +817,7 @@ int nvme_mpath_alloc_disk(struct nvme_ctrl *ctrl, struct nvme_ns_head *head)
 	mutex_init(&head->lock);
 	bio_list_init(&head->requeue_list);
 	spin_lock_init(&head->requeue_lock);
+	INIT_LIST_HEAD(&head->fots);
 	INIT_WORK(&head->requeue_work, nvme_requeue_work);
 	INIT_WORK(&head->partition_scan_work, nvme_partition_scan_work);
 	INIT_DELAYED_WORK(&head->remove_work, nvme_remove_head_work);
@@ -1424,6 +1511,7 @@ void nvme_mpath_put_disk(struct nvme_ns_head *head)
 	if (!head->disk)
 		return;
 	/* make sure all pending bios are cleaned up */
+	nvme_mpath_drain_failover_timers(head);
 	kblockd_schedule_work(&head->requeue_work);
 	flush_work(&head->requeue_work);
 	flush_work(&head->partition_scan_work);
diff --git a/drivers/nvme/host/nvme.h b/drivers/nvme/host/nvme.h
index 944462e912f6..e63f21b3110c 100644
--- a/drivers/nvme/host/nvme.h
+++ b/drivers/nvme/host/nvme.h
@@ -565,6 +565,7 @@ struct nvme_ns_head {
 	struct bio_list		requeue_list;
 	spinlock_t		requeue_lock;
 	struct work_struct	requeue_work;
+	struct list_head	fots;
 	struct work_struct	partition_scan_work;
 	struct mutex		lock;
 	unsigned long		flags;
-- 
2.43.0
lmpx.com only provides a reader for public news (NNTP) servers. It is not affiliated with the servers or forums shown here and is not responsible for the content of articles, which is written by their respective authors.