xref: /linux/drivers/nvme/host/multipath.c (revision 4a5e49ba0abb8b4328d6318c9aef0c0121f95507)
1 // SPDX-License-Identifier: GPL-2.0
2 /*
3  * Copyright (c) 2017-2018 Christoph Hellwig.
4  */
5 
6 #include <linux/backing-dev.h>
7 #include <linux/moduleparam.h>
8 #include <linux/vmalloc.h>
9 #include <trace/events/block.h>
10 #include "nvme.h"
11 
12 bool multipath = true;
13 static bool multipath_always_on;
14 
15 static int multipath_param_set(const char *val, const struct kernel_param *kp)
16 {
17 	int ret;
18 	bool *arg = kp->arg;
19 
20 	ret = param_set_bool(val, kp);
21 	if (ret)
22 		return ret;
23 
24 	if (multipath_always_on && !*arg) {
25 		pr_err("Can't disable multipath when multipath_always_on is configured.\n");
26 		*arg = true;
27 		return -EINVAL;
28 	}
29 
30 	return 0;
31 }
32 
33 static const struct kernel_param_ops multipath_param_ops = {
34 	.set = multipath_param_set,
35 	.get = param_get_bool,
36 };
37 
38 module_param_cb(multipath, &multipath_param_ops, &multipath, 0444);
39 MODULE_PARM_DESC(multipath,
40 	"turn on native support for multiple controllers per subsystem");
41 
42 static int multipath_always_on_set(const char *val,
43 		const struct kernel_param *kp)
44 {
45 	int ret;
46 	bool *arg = kp->arg;
47 
48 	ret = param_set_bool(val, kp);
49 	if (ret < 0)
50 		return ret;
51 
52 	if (*arg)
53 		multipath = true;
54 
55 	return 0;
56 }
57 
58 static const struct kernel_param_ops multipath_always_on_ops = {
59 	.set = multipath_always_on_set,
60 	.get = param_get_bool,
61 };
62 
63 module_param_cb(multipath_always_on, &multipath_always_on_ops,
64 		&multipath_always_on, 0444);
65 MODULE_PARM_DESC(multipath_always_on,
66 	"create multipath node always except for private namespace with non-unique nsid; note that this also implicitly enables native multipath support");
67 
68 static const char *nvme_iopolicy_names[] = {
69 	[NVME_IOPOLICY_NUMA]	= "numa",
70 	[NVME_IOPOLICY_RR]	= "round-robin",
71 	[NVME_IOPOLICY_QD]      = "queue-depth",
72 };
73 
74 static int iopolicy = NVME_IOPOLICY_NUMA;
75 
76 static int nvme_iopolicy_parse(const char *str)
77 {
78 	int i;
79 
80 	for (i = 0; i < ARRAY_SIZE(nvme_iopolicy_names); i++) {
81 		if (sysfs_streq(str, nvme_iopolicy_names[i]))
82 			return i;
83 	}
84 	return -EINVAL;
85 }
86 
87 static int nvme_set_iopolicy(const char *val, const struct kernel_param *kp)
88 {
89 	int policy;
90 
91 	if (!val)
92 		return -EINVAL;
93 
94 	policy = nvme_iopolicy_parse(val);
95 	if (policy < 0)
96 		return policy;
97 
98 	iopolicy = policy;
99 	return 0;
100 }
101 
102 static int nvme_get_iopolicy(char *buf, const struct kernel_param *kp)
103 {
104 	return sprintf(buf, "%s\n", nvme_iopolicy_names[iopolicy]);
105 }
106 
107 module_param_call(iopolicy, nvme_set_iopolicy, nvme_get_iopolicy,
108 	&iopolicy, 0644);
109 MODULE_PARM_DESC(iopolicy,
110 	"Default multipath I/O policy; 'numa' (default), 'round-robin' or 'queue-depth'");
111 
112 void nvme_mpath_default_iopolicy(struct nvme_subsystem *subsys)
113 {
114 	subsys->iopolicy = iopolicy;
115 }
116 
117 void nvme_mpath_unfreeze(struct nvme_subsystem *subsys)
118 {
119 	struct nvme_ns_head *h;
120 
121 	lockdep_assert_held(&subsys->lock);
122 	list_for_each_entry(h, &subsys->nsheads, entry)
123 		if (h->disk)
124 			blk_mq_unfreeze_queue_nomemrestore(h->disk->queue);
125 }
126 
127 void nvme_mpath_wait_freeze(struct nvme_subsystem *subsys)
128 {
129 	struct nvme_ns_head *h;
130 
131 	lockdep_assert_held(&subsys->lock);
132 	list_for_each_entry(h, &subsys->nsheads, entry)
133 		if (h->disk)
134 			blk_mq_freeze_queue_wait(h->disk->queue);
135 }
136 
137 void nvme_mpath_start_freeze(struct nvme_subsystem *subsys)
138 {
139 	struct nvme_ns_head *h;
140 
141 	lockdep_assert_held(&subsys->lock);
142 	list_for_each_entry(h, &subsys->nsheads, entry)
143 		if (h->disk)
144 			blk_freeze_queue_start(h->disk->queue);
145 }
146 
147 void nvme_failover_req(struct request *req)
148 {
149 	struct nvme_ns *ns = req->q->queuedata;
150 	u16 status = nvme_req(req)->status & NVME_SCT_SC_MASK;
151 	unsigned long flags;
152 	struct bio *bio;
153 
154 	nvme_mpath_clear_current_path(ns);
155 	atomic_long_inc(&ns->failover);
156 
157 	/*
158 	 * If we got back an ANA error, we know the controller is alive but not
159 	 * ready to serve this namespace.  Kick of a re-read of the ANA
160 	 * information page, and just try any other available path for now.
161 	 */
162 	if (nvme_is_ana_error(status) && ns->ctrl->ana_log_buf) {
163 		set_bit(NVME_NS_ANA_PENDING, &ns->flags);
164 		queue_work(nvme_wq, &ns->ctrl->ana_work);
165 	}
166 
167 	spin_lock_irqsave(&ns->head->requeue_lock, flags);
168 	for (bio = req->bio; bio; bio = bio->bi_next)
169 		bio_set_dev(bio, ns->head->disk->part0);
170 	blk_steal_bios(&ns->head->requeue_list, req);
171 	spin_unlock_irqrestore(&ns->head->requeue_lock, flags);
172 
173 	nvme_req(req)->status = 0;
174 	nvme_end_req(req);
175 	kblockd_schedule_work(&ns->head->requeue_work);
176 }
177 
178 void nvme_mpath_start_request(struct request *rq)
179 {
180 	struct nvme_ns *ns = rq->q->queuedata;
181 	struct gendisk *disk = ns->head->disk;
182 
183 	if ((READ_ONCE(ns->head->subsys->iopolicy) == NVME_IOPOLICY_QD) &&
184 	    !(nvme_req(rq)->flags & NVME_MPATH_CNT_ACTIVE)) {
185 		atomic_inc(&ns->ctrl->nr_active);
186 		nvme_req(rq)->flags |= NVME_MPATH_CNT_ACTIVE;
187 	}
188 
189 	if (!blk_queue_io_stat(disk->queue) ||
190 	    (nvme_req(rq)->flags & NVME_MPATH_IO_STATS))
191 		return;
192 	if (blk_rq_is_passthrough(rq) &&
193 	    !blk_rq_passthrough_stats(rq, disk->queue))
194 		return;
195 
196 	nvme_req(rq)->flags |= NVME_MPATH_IO_STATS;
197 	nvme_req(rq)->start_time = bdev_start_io_acct(disk->part0, req_op(rq),
198 						      jiffies);
199 }
200 EXPORT_SYMBOL_GPL(nvme_mpath_start_request);
201 
202 void nvme_mpath_end_request(struct request *rq)
203 {
204 	struct nvme_ns *ns = rq->q->queuedata;
205 
206 	if (nvme_req(rq)->flags & NVME_MPATH_CNT_ACTIVE)
207 		atomic_dec_if_positive(&ns->ctrl->nr_active);
208 
209 	if (!(nvme_req(rq)->flags & NVME_MPATH_IO_STATS))
210 		return;
211 	bdev_end_io_acct(ns->head->disk->part0, req_op(rq),
212 			 blk_rq_bytes(rq) >> SECTOR_SHIFT,
213 			 nvme_req(rq)->start_time);
214 }
215 
216 void nvme_kick_requeue_lists(struct nvme_ctrl *ctrl)
217 {
218 	struct nvme_ns *ns;
219 	int srcu_idx;
220 
221 	srcu_idx = srcu_read_lock(&ctrl->srcu);
222 	list_for_each_entry_srcu(ns, &ctrl->namespaces, list,
223 				 srcu_read_lock_held(&ctrl->srcu)) {
224 		if (!ns->head->disk)
225 			continue;
226 		kblockd_schedule_work(&ns->head->requeue_work);
227 		if (nvme_ctrl_state(ns->ctrl) == NVME_CTRL_LIVE)
228 			disk_uevent(ns->head->disk, KOBJ_CHANGE);
229 	}
230 	srcu_read_unlock(&ctrl->srcu, srcu_idx);
231 }
232 
233 static const char *nvme_ana_state_names[] = {
234 	[0]				= "invalid state",
235 	[NVME_ANA_OPTIMIZED]		= "optimized",
236 	[NVME_ANA_NONOPTIMIZED]		= "non-optimized",
237 	[NVME_ANA_INACCESSIBLE]		= "inaccessible",
238 	[NVME_ANA_PERSISTENT_LOSS]	= "persistent-loss",
239 	[NVME_ANA_CHANGE]		= "change",
240 };
241 
242 bool nvme_mpath_clear_current_path(struct nvme_ns *ns)
243 {
244 	struct nvme_ns_head *head = ns->head;
245 	bool changed = false;
246 	int node;
247 
248 	for_each_node(node) {
249 		if (ns == rcu_access_pointer(head->current_path[node])) {
250 			rcu_assign_pointer(head->current_path[node], NULL);
251 			changed = true;
252 		}
253 	}
254 	return changed;
255 }
256 
257 void nvme_mpath_clear_ctrl_paths(struct nvme_ctrl *ctrl)
258 {
259 	struct nvme_ns *ns;
260 	int srcu_idx;
261 
262 	srcu_idx = srcu_read_lock(&ctrl->srcu);
263 	list_for_each_entry_srcu(ns, &ctrl->namespaces, list,
264 				 srcu_read_lock_held(&ctrl->srcu)) {
265 		nvme_mpath_clear_current_path(ns);
266 		kblockd_schedule_work(&ns->head->requeue_work);
267 	}
268 	srcu_read_unlock(&ctrl->srcu, srcu_idx);
269 }
270 
271 void nvme_mpath_revalidate_paths(struct nvme_ns_head *head)
272 {
273 	sector_t capacity = get_capacity(head->disk);
274 	struct nvme_ns *ns;
275 	int node;
276 	int srcu_idx;
277 
278 	srcu_idx = srcu_read_lock(&head->srcu);
279 	list_for_each_entry_srcu(ns, &head->list, siblings,
280 				 srcu_read_lock_held(&head->srcu)) {
281 		if (capacity != get_capacity(ns->disk))
282 			clear_bit(NVME_NS_READY, &ns->flags);
283 	}
284 	srcu_read_unlock(&head->srcu, srcu_idx);
285 
286 	for_each_node(node)
287 		rcu_assign_pointer(head->current_path[node], NULL);
288 	kblockd_schedule_work(&head->requeue_work);
289 }
290 
291 #ifdef CONFIG_BLK_DEV_ZONED
292 int nvme_mpath_revalidate_zones(struct nvme_ns_head *head)
293 {
294 	struct gendisk *disk = head->disk;
295 	int ret;
296 
297 	if (!disk || !blk_queue_is_zoned(disk->queue) ||
298 	    !test_bit(NVME_NSHEAD_DISK_LIVE, &head->flags))
299 		return 0;
300 
301 	ret = blk_revalidate_disk_zones(disk);
302 	if (ret)
303 		dev_warn_ratelimited(disk_to_dev(disk),
304 				     "failed to revalidate zoned namespace head: %d\n",
305 				     ret);
306 	return ret;
307 }
308 #endif /* CONFIG_BLK_DEV_ZONED */
309 
310 static bool nvme_path_is_disabled(struct nvme_ns *ns)
311 {
312 	enum nvme_ctrl_state state = nvme_ctrl_state(ns->ctrl);
313 
314 	/*
315 	 * We don't treat NVME_CTRL_DELETING as a disabled path as I/O should
316 	 * still be able to complete assuming that the controller is connected.
317 	 * Otherwise it will fail immediately and return to the requeue list.
318 	 */
319 	if (state != NVME_CTRL_LIVE && state != NVME_CTRL_DELETING)
320 		return true;
321 	if (test_bit(NVME_NS_ANA_PENDING, &ns->flags) ||
322 	    !test_bit(NVME_NS_READY, &ns->flags))
323 		return true;
324 	return false;
325 }
326 
327 static struct nvme_ns *__nvme_find_path(struct nvme_ns_head *head, int node)
328 	__must_hold_shared(&head->srcu)
329 {
330 	int found_distance = INT_MAX, fallback_distance = INT_MAX, distance;
331 	struct nvme_ns *found = NULL, *fallback = NULL, *ns;
332 
333 	list_for_each_entry_srcu(ns, &head->list, siblings,
334 				 srcu_read_lock_held(&head->srcu)) {
335 		if (nvme_path_is_disabled(ns))
336 			continue;
337 
338 		if (ns->ctrl->numa_node != NUMA_NO_NODE &&
339 		    READ_ONCE(head->subsys->iopolicy) == NVME_IOPOLICY_NUMA)
340 			distance = node_distance(node, ns->ctrl->numa_node);
341 		else
342 			distance = LOCAL_DISTANCE;
343 
344 		switch (ns->ana_state) {
345 		case NVME_ANA_OPTIMIZED:
346 			if (distance < found_distance) {
347 				found_distance = distance;
348 				found = ns;
349 			}
350 			break;
351 		case NVME_ANA_NONOPTIMIZED:
352 			if (distance < fallback_distance) {
353 				fallback_distance = distance;
354 				fallback = ns;
355 			}
356 			break;
357 		default:
358 			break;
359 		}
360 	}
361 
362 	if (!found)
363 		found = fallback;
364 	if (found)
365 		rcu_assign_pointer(head->current_path[node], found);
366 	return found;
367 }
368 
369 static struct nvme_ns *nvme_next_ns(struct nvme_ns_head *head,
370 		struct nvme_ns *ns)
371 	__must_hold_shared(&head->srcu)
372 {
373 	ns = list_next_or_null_rcu(&head->list, &ns->siblings, struct nvme_ns,
374 			siblings);
375 	if (ns)
376 		return ns;
377 	return list_first_or_null_rcu(&head->list, struct nvme_ns, siblings);
378 }
379 
380 static struct nvme_ns *nvme_round_robin_path(struct nvme_ns_head *head)
381 	__must_hold_shared(&head->srcu)
382 {
383 	struct nvme_ns *ns, *found = NULL;
384 	int node = numa_node_id();
385 	struct nvme_ns *old = srcu_dereference(head->current_path[node],
386 					       &head->srcu);
387 
388 	if (unlikely(!old))
389 		return __nvme_find_path(head, node);
390 
391 	if (list_is_singular(&head->list)) {
392 		if (nvme_path_is_disabled(old))
393 			return NULL;
394 		return old;
395 	}
396 
397 	for (ns = nvme_next_ns(head, old);
398 	     ns && ns != old;
399 	     ns = nvme_next_ns(head, ns)) {
400 		if (nvme_path_is_disabled(ns))
401 			continue;
402 
403 		if (ns->ana_state == NVME_ANA_OPTIMIZED) {
404 			found = ns;
405 			goto out;
406 		}
407 		if (ns->ana_state == NVME_ANA_NONOPTIMIZED)
408 			found = ns;
409 	}
410 
411 	/*
412 	 * The loop above skips the current path for round-robin semantics.
413 	 * Fall back to the current path if either:
414 	 *  - no other optimized path found and current is optimized,
415 	 *  - no other usable path found and current is usable.
416 	 */
417 	if (!nvme_path_is_disabled(old) &&
418 	    (old->ana_state == NVME_ANA_OPTIMIZED ||
419 	     (!found && old->ana_state == NVME_ANA_NONOPTIMIZED)))
420 		return old;
421 
422 	if (!found)
423 		return NULL;
424 out:
425 	rcu_assign_pointer(head->current_path[node], found);
426 	return found;
427 }
428 
429 static struct nvme_ns *nvme_queue_depth_path(struct nvme_ns_head *head)
430 	__must_hold_shared(&head->srcu)
431 {
432 	struct nvme_ns *best_opt = NULL, *best_nonopt = NULL, *ns;
433 	unsigned int min_depth_opt = UINT_MAX, min_depth_nonopt = UINT_MAX;
434 	unsigned int depth;
435 
436 	list_for_each_entry_srcu(ns, &head->list, siblings,
437 				 srcu_read_lock_held(&head->srcu)) {
438 		if (nvme_path_is_disabled(ns))
439 			continue;
440 
441 		depth = atomic_read(&ns->ctrl->nr_active);
442 
443 		switch (ns->ana_state) {
444 		case NVME_ANA_OPTIMIZED:
445 			if (depth < min_depth_opt) {
446 				min_depth_opt = depth;
447 				best_opt = ns;
448 			}
449 			break;
450 		case NVME_ANA_NONOPTIMIZED:
451 			if (depth < min_depth_nonopt) {
452 				min_depth_nonopt = depth;
453 				best_nonopt = ns;
454 			}
455 			break;
456 		default:
457 			break;
458 		}
459 
460 		if (min_depth_opt == 0)
461 			return best_opt;
462 	}
463 
464 	return best_opt ? best_opt : best_nonopt;
465 }
466 
467 static inline bool nvme_path_is_optimized(struct nvme_ns *ns)
468 {
469 	return nvme_ctrl_state(ns->ctrl) == NVME_CTRL_LIVE &&
470 		ns->ana_state == NVME_ANA_OPTIMIZED;
471 }
472 
473 static struct nvme_ns *nvme_numa_path(struct nvme_ns_head *head)
474 	__must_hold_shared(&head->srcu)
475 {
476 	int node = numa_node_id();
477 	struct nvme_ns *ns;
478 
479 	ns = srcu_dereference(head->current_path[node], &head->srcu);
480 	if (unlikely(!ns))
481 		return __nvme_find_path(head, node);
482 	if (unlikely(!nvme_path_is_optimized(ns)))
483 		return __nvme_find_path(head, node);
484 	return ns;
485 }
486 
487 inline struct nvme_ns *nvme_find_path(struct nvme_ns_head *head)
488 {
489 	switch (READ_ONCE(head->subsys->iopolicy)) {
490 	case NVME_IOPOLICY_QD:
491 		return nvme_queue_depth_path(head);
492 	case NVME_IOPOLICY_RR:
493 		return nvme_round_robin_path(head);
494 	default:
495 		return nvme_numa_path(head);
496 	}
497 }
498 
499 static bool nvme_available_path(struct nvme_ns_head *head)
500 	__must_hold_shared(&head->srcu)
501 {
502 	struct nvme_ns *ns;
503 
504 	if (!test_bit(NVME_NSHEAD_DISK_LIVE, &head->flags))
505 		return false;
506 
507 	list_for_each_entry_srcu(ns, &head->list, siblings,
508 				 srcu_read_lock_held(&head->srcu)) {
509 		if (test_bit(NVME_CTRL_FAILFAST_EXPIRED, &ns->ctrl->flags))
510 			continue;
511 		switch (nvme_ctrl_state(ns->ctrl)) {
512 		case NVME_CTRL_LIVE:
513 		case NVME_CTRL_RESETTING:
514 		case NVME_CTRL_CONNECTING:
515 			return true;
516 		default:
517 			break;
518 		}
519 	}
520 
521 	/*
522 	 * If "head->delayed_removal_secs" is configured (i.e., non-zero), do
523 	 * not immediately fail I/O. Instead, requeue the I/O for the configured
524 	 * duration, anticipating that if there's a transient link failure then
525 	 * it may recover within this time window. This parameter is exported to
526 	 * userspace via sysfs, and its default value is zero. It is internally
527 	 * mapped to NVME_NSHEAD_QUEUE_IF_NO_PATH. When delayed_removal_secs is
528 	 * non-zero, this flag is set to true. When zero, the flag is cleared.
529 	 */
530 	return nvme_mpath_queue_if_no_path(head);
531 }
532 
533 static void nvme_ns_head_submit_bio(struct bio *bio)
534 {
535 	struct nvme_ns_head *head = bio->bi_bdev->bd_disk->private_data;
536 	struct device *dev = disk_to_dev(head->disk);
537 	struct nvme_ns *ns;
538 	int srcu_idx;
539 
540 	/*
541 	 * The namespace might be going away and the bio might be moved to a
542 	 * different queue via blk_steal_bios(), so we need to use the bio_split
543 	 * pool from the original queue to allocate the bvecs from.
544 	 */
545 	bio = bio_split_to_limits(bio);
546 	if (!bio)
547 		return;
548 
549 	srcu_idx = srcu_read_lock(&head->srcu);
550 	ns = nvme_find_path(head);
551 	if (likely(ns)) {
552 		bio_set_dev(bio, ns->disk->part0);
553 		/*
554 		 * Use BIO_REMAPPED to skip bio_check_eod() when this bio
555 		 * enters submit_bio_noacct() for the per-path device. The EOD
556 		 * check already passed on the multipath head.
557 		 */
558 		bio_set_flag(bio, BIO_REMAPPED);
559 		bio->bi_opf |= REQ_NVME_MPATH;
560 		trace_block_bio_remap(bio, disk_devt(ns->head->disk),
561 				      bio->bi_iter.bi_sector);
562 		submit_bio_noacct(bio);
563 	} else if (nvme_available_path(head)) {
564 		dev_warn_ratelimited(dev, "no usable path - requeuing I/O\n");
565 
566 		spin_lock_irq(&head->requeue_lock);
567 		bio_list_add(&head->requeue_list, bio);
568 		spin_unlock_irq(&head->requeue_lock);
569 		atomic_long_inc(&head->io_requeue_no_usable_path_count);
570 	} else {
571 		dev_warn_ratelimited(dev, "no available path - failing I/O\n");
572 
573 		bio_io_error(bio);
574 		atomic_long_inc(&head->io_fail_no_available_path_count);
575 	}
576 
577 	srcu_read_unlock(&head->srcu, srcu_idx);
578 }
579 
580 static int nvme_ns_head_open(struct gendisk *disk, blk_mode_t mode)
581 {
582 	if (!nvme_tryget_ns_head(disk->private_data))
583 		return -ENXIO;
584 	return 0;
585 }
586 
587 static void nvme_ns_head_release(struct gendisk *disk)
588 {
589 	nvme_put_ns_head(disk->private_data);
590 }
591 
592 static int nvme_ns_head_get_unique_id(struct gendisk *disk, u8 id[16],
593 		enum blk_unique_id type)
594 {
595 	struct nvme_ns_head *head = disk->private_data;
596 	struct nvme_ns *ns;
597 	int srcu_idx, ret = -EWOULDBLOCK;
598 
599 	srcu_idx = srcu_read_lock(&head->srcu);
600 	ns = nvme_find_path(head);
601 	if (ns)
602 		ret = nvme_ns_get_unique_id(ns, id, type);
603 	srcu_read_unlock(&head->srcu, srcu_idx);
604 	return ret;
605 }
606 
607 #ifdef CONFIG_BLK_DEV_ZONED
608 static int nvme_ns_head_report_zones(struct gendisk *disk, sector_t sector,
609 		unsigned int nr_zones, struct blk_report_zones_args *args)
610 {
611 	struct nvme_ns_head *head = disk->private_data;
612 	struct nvme_ns *ns;
613 	int srcu_idx, ret = -EWOULDBLOCK;
614 
615 	srcu_idx = srcu_read_lock(&head->srcu);
616 	ns = nvme_find_path(head);
617 	if (ns)
618 		ret = nvme_ns_report_zones(ns, sector, nr_zones, args);
619 	srcu_read_unlock(&head->srcu, srcu_idx);
620 	return ret;
621 }
622 #else
623 #define nvme_ns_head_report_zones	NULL
624 #endif /* CONFIG_BLK_DEV_ZONED */
625 
626 const struct block_device_operations nvme_ns_head_ops = {
627 	.owner		= THIS_MODULE,
628 	.submit_bio	= nvme_ns_head_submit_bio,
629 	.open		= nvme_ns_head_open,
630 	.release	= nvme_ns_head_release,
631 	.ioctl		= nvme_ns_head_ioctl,
632 	.compat_ioctl	= blkdev_compat_ptr_ioctl,
633 	.getgeo		= nvme_getgeo,
634 	.get_unique_id	= nvme_ns_head_get_unique_id,
635 	.report_zones	= nvme_ns_head_report_zones,
636 	.pr_ops		= &nvme_pr_ops,
637 };
638 
639 static const struct file_operations nvme_ns_head_chr_fops = {
640 	.owner		= THIS_MODULE,
641 	.unlocked_ioctl	= nvme_ns_head_chr_ioctl,
642 	.compat_ioctl	= compat_ptr_ioctl,
643 	.uring_cmd	= nvme_ns_head_chr_uring_cmd,
644 	.uring_cmd_iopoll = nvme_ns_chr_uring_cmd_iopoll,
645 };
646 
647 static void nvme_add_ns_head_cdev(struct nvme_ns_head *head)
648 {
649 	head->cdev_device.parent = &head->subsys->dev;
650 
651 	nvme_get_ns_head(head); /* Undone in nvme_cdev_rel() */
652 	if (nvme_cdev_add(&head->cdev, &head->cdev_device,
653 			&nvme_ns_head_chr_fops, THIS_MODULE,
654 			head->subsys->instance, head->instance)) {
655 		dev_err(disk_to_dev(head->disk),
656 			"Unable to create the ng%dn%d device\n",
657 			head->subsys->instance, head->instance);
658 		nvme_put_ns_head(head);
659 		return;
660 	}
661 	set_bit(NVME_NSHEAD_CDEV_LIVE, &head->flags);
662 }
663 
664 static void nvme_partition_scan_work(struct work_struct *work)
665 {
666 	struct nvme_ns_head *head =
667 		container_of(work, struct nvme_ns_head, partition_scan_work);
668 
669 	if (WARN_ON_ONCE(!test_and_clear_bit(GD_SUPPRESS_PART_SCAN,
670 					     &head->disk->state)))
671 		return;
672 
673 	mutex_lock(&head->disk->open_mutex);
674 	bdev_disk_changed(head->disk, false);
675 	mutex_unlock(&head->disk->open_mutex);
676 }
677 
678 static void nvme_requeue_work(struct work_struct *work)
679 {
680 	struct nvme_ns_head *head =
681 		container_of(work, struct nvme_ns_head, requeue_work);
682 	struct bio *bio, *next;
683 
684 	spin_lock_irq(&head->requeue_lock);
685 	next = bio_list_get(&head->requeue_list);
686 	spin_unlock_irq(&head->requeue_lock);
687 
688 	while ((bio = next) != NULL) {
689 		next = bio->bi_next;
690 		bio->bi_next = NULL;
691 
692 		submit_bio_noacct(bio);
693 	}
694 }
695 
696 static void nvme_remove_head(struct nvme_ns_head *head)
697 {
698 	if (test_and_clear_bit(NVME_NSHEAD_DISK_LIVE, &head->flags)) {
699 		/*
700 		 * Requeue I/O after NVME_NSHEAD_DISK_LIVE has been cleared
701 		 * to allow multipath to fail all I/O. First synchronize to
702 		 * add any bios to the requeue list.
703 		 */
704 		synchronize_srcu(&head->srcu);
705 		kblockd_schedule_work(&head->requeue_work);
706 
707 		if (test_and_clear_bit(NVME_NSHEAD_CDEV_LIVE, &head->flags))
708 			nvme_cdev_del(&head->cdev, &head->cdev_device);
709 		del_gendisk(head->disk);
710 	}
711 	nvme_put_ns_head(head);
712 }
713 
714 static void nvme_remove_head_work(struct work_struct *work)
715 {
716 	struct nvme_ns_head *head = container_of(to_delayed_work(work),
717 			struct nvme_ns_head, remove_work);
718 	bool remove = false;
719 
720 	mutex_lock(&head->subsys->lock);
721 	if (list_empty(&head->list)) {
722 		list_del_init(&head->entry);
723 		remove = true;
724 	}
725 	mutex_unlock(&head->subsys->lock);
726 	if (remove)
727 		nvme_remove_head(head);
728 
729 	module_put(THIS_MODULE);
730 }
731 
732 int nvme_mpath_alloc_disk(struct nvme_ctrl *ctrl, struct nvme_ns_head *head)
733 {
734 	struct queue_limits lim;
735 
736 	mutex_init(&head->lock);
737 	spin_lock_init(&head->requeue_lock);
738 	INIT_WORK(&head->requeue_work, nvme_requeue_work);
739 	INIT_WORK(&head->partition_scan_work, nvme_partition_scan_work);
740 	INIT_DELAYED_WORK(&head->remove_work, nvme_remove_head_work);
741 
742 	/*
743 	 * If "multipath_always_on" is enabled, a multipath node is added
744 	 * regardless of whether the disk is single/multi ported, and whether
745 	 * the namespace is shared or private. If "multipath_always_on" is not
746 	 * enabled, a multipath node is added only if the subsystem supports
747 	 * multiple controllers and the "multipath" option is configured. In
748 	 * either case, for private namespaces, we ensure that the NSID is
749 	 * unique.
750 	 */
751 	if (!multipath_always_on) {
752 		if (!(ctrl->subsys->cmic & NVME_CTRL_CMIC_MULTI_CTRL) ||
753 				!multipath)
754 			return 0;
755 	}
756 
757 	if (!nvme_is_unique_nsid(ctrl, head))
758 		return 0;
759 
760 	blk_set_stacking_limits(&lim);
761 	lim.dma_alignment = 3;
762 	lim.features |= BLK_FEAT_IO_STAT | BLK_FEAT_NOWAIT |
763 		BLK_FEAT_POLL | BLK_FEAT_ATOMIC_WRITES | BLK_FEAT_PCI_P2PDMA;
764 
765 	head->disk = blk_alloc_disk(&lim, ctrl->numa_node);
766 	if (IS_ERR(head->disk))
767 		return PTR_ERR(head->disk);
768 	head->disk->fops = &nvme_ns_head_ops;
769 	head->disk->private_data = head;
770 
771 	/*
772 	 * We need to suppress the partition scan from occuring within the
773 	 * controller's scan_work context. If a path error occurs here, the IO
774 	 * will wait until a path becomes available or all paths are torn down,
775 	 * but that action also occurs within scan_work, so it would deadlock.
776 	 * Defer the partition scan to a different context that does not block
777 	 * scan_work.
778 	 */
779 	set_bit(GD_SUPPRESS_PART_SCAN, &head->disk->state);
780 	sprintf(head->disk->disk_name, "nvme%dn%d",
781 			ctrl->subsys->instance, head->instance);
782 	nvme_get_ns_head(head);
783 	return 0;
784 }
785 
786 static void nvme_mpath_set_live(struct nvme_ns *ns)
787 {
788 	struct nvme_ns_head *head = ns->head;
789 	int rc;
790 
791 	if (!head->disk)
792 		return;
793 
794 	/*
795 	 * test_and_set_bit() is used because it is protecting against two nvme
796 	 * paths simultaneously calling device_add_disk() on the same namespace
797 	 * head.
798 	 */
799 	if (!test_and_set_bit(NVME_NSHEAD_DISK_LIVE, &head->flags)) {
800 		rc = device_add_disk(&head->subsys->dev, head->disk,
801 				     nvme_ns_attr_groups);
802 		if (rc) {
803 			clear_bit(NVME_NSHEAD_DISK_LIVE, &head->flags);
804 			return;
805 		}
806 		nvme_add_ns_head_cdev(head);
807 		queue_work(nvme_wq, &head->partition_scan_work);
808 	}
809 
810 	nvme_mpath_add_sysfs_link(ns->head);
811 
812 	mutex_lock(&head->lock);
813 	if (nvme_path_is_optimized(ns)) {
814 		int node, srcu_idx;
815 
816 		srcu_idx = srcu_read_lock(&head->srcu);
817 		for_each_online_node(node)
818 			__nvme_find_path(head, node);
819 		srcu_read_unlock(&head->srcu, srcu_idx);
820 	}
821 	mutex_unlock(&head->lock);
822 
823 	synchronize_srcu(&head->srcu);
824 	nvme_mpath_revalidate_zones(head);
825 	kblockd_schedule_work(&head->requeue_work);
826 }
827 
828 static int nvme_parse_ana_log(struct nvme_ctrl *ctrl, void *data,
829 		int (*cb)(struct nvme_ctrl *ctrl, struct nvme_ana_group_desc *,
830 			void *))
831 		__must_hold(&ctrl->ana_lock)
832 {
833 	void *base = ctrl->ana_log_buf;
834 	size_t offset = sizeof(struct nvme_ana_rsp_hdr);
835 	int error, i;
836 
837 	lockdep_assert_held(&ctrl->ana_lock);
838 
839 	for (i = 0; i < le16_to_cpu(ctrl->ana_log_buf->ngrps); i++) {
840 		struct nvme_ana_group_desc *desc = base + offset;
841 		u32 nr_nsids;
842 		size_t nsid_buf_size;
843 
844 		if (WARN_ON_ONCE(offset > ctrl->ana_log_size ||
845 				 sizeof(*desc) > ctrl->ana_log_size - offset))
846 			return -EINVAL;
847 
848 		nr_nsids = le32_to_cpu(desc->nnsids);
849 		nsid_buf_size = flex_array_size(desc, nsids, nr_nsids);
850 
851 		if (WARN_ON_ONCE(desc->grpid == 0))
852 			return -EINVAL;
853 		if (WARN_ON_ONCE(le32_to_cpu(desc->grpid) > ctrl->anagrpmax))
854 			return -EINVAL;
855 		if (WARN_ON_ONCE(desc->state == 0))
856 			return -EINVAL;
857 		if (WARN_ON_ONCE(desc->state > NVME_ANA_CHANGE))
858 			return -EINVAL;
859 
860 		offset += sizeof(*desc);
861 		if (WARN_ON_ONCE(nsid_buf_size > ctrl->ana_log_size - offset))
862 			return -EINVAL;
863 
864 		error = cb(ctrl, desc, data);
865 		if (error)
866 			return error;
867 
868 		offset += nsid_buf_size;
869 	}
870 
871 	return 0;
872 }
873 
874 static inline bool nvme_state_is_live(enum nvme_ana_state state)
875 {
876 	return state == NVME_ANA_OPTIMIZED || state == NVME_ANA_NONOPTIMIZED;
877 }
878 
879 static void nvme_update_ns_ana_state(struct nvme_ana_group_desc *desc,
880 		struct nvme_ns *ns)
881 {
882 	ns->ana_grpid = le32_to_cpu(desc->grpid);
883 	ns->ana_state = desc->state;
884 	clear_bit(NVME_NS_ANA_PENDING, &ns->flags);
885 	/*
886 	 * nvme_mpath_set_live() will trigger I/O to the multipath path device
887 	 * and in turn to this path device.  However we cannot accept this I/O
888 	 * if the controller is not live.  This may deadlock if called from
889 	 * nvme_mpath_init_identify() and the ctrl will never complete
890 	 * initialization, preventing I/O from completing.  For this case we
891 	 * will reprocess the ANA log page in nvme_mpath_update() once the
892 	 * controller is ready.
893 	 */
894 	if (nvme_state_is_live(ns->ana_state) &&
895 	    nvme_ctrl_state(ns->ctrl) == NVME_CTRL_LIVE)
896 		nvme_mpath_set_live(ns);
897 	else {
898 		/*
899 		 * Add sysfs link from multipath head gendisk node to path
900 		 * device gendisk node.
901 		 * If path's ana state is live (i.e. state is either optimized
902 		 * or non-optimized) while we alloc the ns then sysfs link would
903 		 * be created from nvme_mpath_set_live(). In that case we would
904 		 * not fallthrough this code path. However for the path's ana
905 		 * state other than live, we call nvme_mpath_set_live() only
906 		 * after ana state transitioned to the live state. But we still
907 		 * want to create the sysfs link from head node to a path device
908 		 * irrespctive of the path's ana state.
909 		 * If we reach through here then it means that path's ana state
910 		 * is not live but still create the sysfs link to this path from
911 		 * head node if head node of the path has already come alive.
912 		 */
913 		if (test_bit(NVME_NSHEAD_DISK_LIVE, &ns->head->flags))
914 			nvme_mpath_add_sysfs_link(ns->head);
915 	}
916 }
917 
918 static int nvme_update_ana_state(struct nvme_ctrl *ctrl,
919 		struct nvme_ana_group_desc *desc, void *data)
920 {
921 	u32 nr_nsids = le32_to_cpu(desc->nnsids), n = 0;
922 	unsigned *nr_change_groups = data;
923 	struct nvme_ns *ns;
924 	int srcu_idx;
925 
926 	dev_dbg(ctrl->device, "ANA group %d: %s.\n",
927 			le32_to_cpu(desc->grpid),
928 			nvme_ana_state_names[desc->state]);
929 
930 	if (desc->state == NVME_ANA_CHANGE)
931 		(*nr_change_groups)++;
932 
933 	if (!nr_nsids)
934 		return 0;
935 
936 	srcu_idx = srcu_read_lock(&ctrl->srcu);
937 	list_for_each_entry_srcu(ns, &ctrl->namespaces, list,
938 				 srcu_read_lock_held(&ctrl->srcu)) {
939 		unsigned nsid;
940 again:
941 		nsid = le32_to_cpu(desc->nsids[n]);
942 		if (ns->head->ns_id < nsid)
943 			continue;
944 		if (ns->head->ns_id == nsid)
945 			nvme_update_ns_ana_state(desc, ns);
946 		if (++n == nr_nsids)
947 			break;
948 		if (ns->head->ns_id > nsid)
949 			goto again;
950 	}
951 	srcu_read_unlock(&ctrl->srcu, srcu_idx);
952 	return 0;
953 }
954 
955 static int nvme_read_ana_log(struct nvme_ctrl *ctrl)
956 {
957 	u32 nr_change_groups = 0;
958 	int error;
959 
960 	mutex_lock(&ctrl->ana_lock);
961 	error = nvme_get_log(ctrl, NVME_NSID_ALL, NVME_LOG_ANA, 0, NVME_CSI_NVM,
962 			ctrl->ana_log_buf, ctrl->ana_log_size, 0);
963 	if (error) {
964 		dev_warn(ctrl->device, "Failed to get ANA log: %d\n", error);
965 		goto out_unlock;
966 	}
967 
968 	error = nvme_parse_ana_log(ctrl, &nr_change_groups,
969 			nvme_update_ana_state);
970 	if (error)
971 		goto out_unlock;
972 
973 	/*
974 	 * In theory we should have an ANATT timer per group as they might enter
975 	 * the change state at different times.  But that is a lot of overhead
976 	 * just to protect against a target that keeps entering new changes
977 	 * states while never finishing previous ones.  But we'll still
978 	 * eventually time out once all groups are in change state, so this
979 	 * isn't a big deal.
980 	 *
981 	 * We also double the ANATT value to provide some slack for transports
982 	 * or AEN processing overhead.
983 	 */
984 	if (nr_change_groups)
985 		mod_timer(&ctrl->anatt_timer, ctrl->anatt * HZ * 2 + jiffies);
986 	else
987 		timer_delete_sync(&ctrl->anatt_timer);
988 out_unlock:
989 	mutex_unlock(&ctrl->ana_lock);
990 	return error;
991 }
992 
993 static void nvme_ana_work(struct work_struct *work)
994 {
995 	struct nvme_ctrl *ctrl = container_of(work, struct nvme_ctrl, ana_work);
996 
997 	if (nvme_ctrl_state(ctrl) != NVME_CTRL_LIVE)
998 		return;
999 
1000 	nvme_read_ana_log(ctrl);
1001 }
1002 
1003 void nvme_mpath_update(struct nvme_ctrl *ctrl)
1004 {
1005 	u32 nr_change_groups = 0;
1006 
1007 	if (!ctrl->ana_log_buf)
1008 		return;
1009 
1010 	mutex_lock(&ctrl->ana_lock);
1011 	nvme_parse_ana_log(ctrl, &nr_change_groups, nvme_update_ana_state);
1012 	mutex_unlock(&ctrl->ana_lock);
1013 }
1014 
1015 static void nvme_anatt_timeout(struct timer_list *t)
1016 {
1017 	struct nvme_ctrl *ctrl = timer_container_of(ctrl, t, anatt_timer);
1018 
1019 	dev_info(ctrl->device, "ANATT timeout, resetting controller.\n");
1020 	nvme_reset_ctrl(ctrl);
1021 }
1022 
1023 void nvme_mpath_stop(struct nvme_ctrl *ctrl)
1024 {
1025 	if (!nvme_ctrl_use_ana(ctrl))
1026 		return;
1027 	timer_delete_sync(&ctrl->anatt_timer);
1028 	cancel_work_sync(&ctrl->ana_work);
1029 }
1030 
1031 #define SUBSYS_ATTR_RW(_name, _mode, _show, _store)  \
1032 	struct device_attribute subsys_attr_##_name =	\
1033 		__ATTR(_name, _mode, _show, _store)
1034 
1035 static ssize_t nvme_subsys_iopolicy_show(struct device *dev,
1036 		struct device_attribute *attr, char *buf)
1037 {
1038 	struct nvme_subsystem *subsys =
1039 		container_of(dev, struct nvme_subsystem, dev);
1040 
1041 	return sysfs_emit(buf, "%s\n",
1042 			  nvme_iopolicy_names[READ_ONCE(subsys->iopolicy)]);
1043 }
1044 
1045 static void nvme_subsys_iopolicy_update(struct nvme_subsystem *subsys,
1046 		int iopolicy)
1047 {
1048 	struct nvme_ctrl *ctrl;
1049 	int old_iopolicy = READ_ONCE(subsys->iopolicy);
1050 
1051 	if (old_iopolicy == iopolicy)
1052 		return;
1053 
1054 	WRITE_ONCE(subsys->iopolicy, iopolicy);
1055 
1056 	/* iopolicy changes clear the mpath by design */
1057 	mutex_lock(&nvme_subsystems_lock);
1058 	list_for_each_entry(ctrl, &subsys->ctrls, subsys_entry)
1059 		nvme_mpath_clear_ctrl_paths(ctrl);
1060 	mutex_unlock(&nvme_subsystems_lock);
1061 
1062 	pr_notice("subsysnqn %s iopolicy changed from %s to %s\n",
1063 			subsys->subnqn,
1064 			nvme_iopolicy_names[old_iopolicy],
1065 			nvme_iopolicy_names[iopolicy]);
1066 }
1067 
1068 static ssize_t nvme_subsys_iopolicy_store(struct device *dev,
1069 		struct device_attribute *attr, const char *buf, size_t count)
1070 {
1071 	struct nvme_subsystem *subsys =
1072 		container_of(dev, struct nvme_subsystem, dev);
1073 	int policy;
1074 
1075 	policy = nvme_iopolicy_parse(buf);
1076 	if (policy < 0)
1077 		return policy;
1078 
1079 	nvme_subsys_iopolicy_update(subsys, policy);
1080 	return count;
1081 }
1082 SUBSYS_ATTR_RW(iopolicy, S_IRUGO | S_IWUSR,
1083 		      nvme_subsys_iopolicy_show, nvme_subsys_iopolicy_store);
1084 
1085 static ssize_t ana_grpid_show(struct device *dev, struct device_attribute *attr,
1086 		char *buf)
1087 {
1088 	return sysfs_emit(buf, "%d\n", nvme_get_ns_from_dev(dev)->ana_grpid);
1089 }
1090 DEVICE_ATTR_RO(ana_grpid);
1091 
1092 static ssize_t ana_state_show(struct device *dev, struct device_attribute *attr,
1093 		char *buf)
1094 {
1095 	struct nvme_ns *ns = nvme_get_ns_from_dev(dev);
1096 
1097 	return sysfs_emit(buf, "%s\n", nvme_ana_state_names[ns->ana_state]);
1098 }
1099 DEVICE_ATTR_RO(ana_state);
1100 
1101 static ssize_t queue_depth_show(struct device *dev,
1102 		struct device_attribute *attr, char *buf)
1103 {
1104 	struct nvme_ns *ns = nvme_get_ns_from_dev(dev);
1105 
1106 	if (ns->head->subsys->iopolicy != NVME_IOPOLICY_QD)
1107 		return 0;
1108 
1109 	return sysfs_emit(buf, "%d\n", atomic_read(&ns->ctrl->nr_active));
1110 }
1111 DEVICE_ATTR_RO(queue_depth);
1112 
1113 static ssize_t numa_nodes_show(struct device *dev, struct device_attribute *attr,
1114 		char *buf)
1115 {
1116 	int node, srcu_idx;
1117 	nodemask_t numa_nodes;
1118 	struct nvme_ns *current_ns;
1119 	struct nvme_ns *ns = nvme_get_ns_from_dev(dev);
1120 	struct nvme_ns_head *head = ns->head;
1121 
1122 	if (head->subsys->iopolicy != NVME_IOPOLICY_NUMA)
1123 		return 0;
1124 
1125 	nodes_clear(numa_nodes);
1126 
1127 	srcu_idx = srcu_read_lock(&head->srcu);
1128 	for_each_node(node) {
1129 		current_ns = srcu_dereference(head->current_path[node],
1130 				&head->srcu);
1131 		if (ns == current_ns)
1132 			node_set(node, numa_nodes);
1133 	}
1134 	srcu_read_unlock(&head->srcu, srcu_idx);
1135 
1136 	return sysfs_emit(buf, "%*pbl\n", nodemask_pr_args(&numa_nodes));
1137 }
1138 DEVICE_ATTR_RO(numa_nodes);
1139 
1140 static ssize_t delayed_removal_secs_show(struct device *dev,
1141 		struct device_attribute *attr, char *buf)
1142 {
1143 	struct gendisk *disk = dev_to_disk(dev);
1144 	struct nvme_ns_head *head = disk->private_data;
1145 	int ret;
1146 
1147 	mutex_lock(&head->subsys->lock);
1148 	ret = sysfs_emit(buf, "%u\n", head->delayed_removal_secs);
1149 	mutex_unlock(&head->subsys->lock);
1150 	return ret;
1151 }
1152 
1153 static ssize_t delayed_removal_secs_store(struct device *dev,
1154 		struct device_attribute *attr, const char *buf, size_t count)
1155 {
1156 	struct gendisk *disk = dev_to_disk(dev);
1157 	struct nvme_ns_head *head = disk->private_data;
1158 	unsigned int sec;
1159 	int ret;
1160 
1161 	ret = kstrtouint(buf, 0, &sec);
1162 	if (ret < 0)
1163 		return ret;
1164 
1165 	mutex_lock(&head->subsys->lock);
1166 	head->delayed_removal_secs = sec;
1167 	if (sec)
1168 		set_bit(NVME_NSHEAD_QUEUE_IF_NO_PATH, &head->flags);
1169 	else
1170 		clear_bit(NVME_NSHEAD_QUEUE_IF_NO_PATH, &head->flags);
1171 	mutex_unlock(&head->subsys->lock);
1172 	/*
1173 	 * Ensure that update to NVME_NSHEAD_QUEUE_IF_NO_PATH is seen
1174 	 * by its reader.
1175 	 */
1176 	synchronize_srcu(&head->srcu);
1177 
1178 	return count;
1179 }
1180 
1181 DEVICE_ATTR_RW(delayed_removal_secs);
1182 
1183 static ssize_t multipath_failover_count_show(struct device *dev,
1184 		struct device_attribute *attr, char *buf)
1185 {
1186 	struct nvme_ns *ns = nvme_get_ns_from_dev(dev);
1187 
1188 	return sysfs_emit(buf, "%lu\n", atomic_long_read(&ns->failover));
1189 }
1190 
1191 static ssize_t multipath_failover_count_store(struct device *dev,
1192 		struct device_attribute *attr, const char *buf, size_t count)
1193 {
1194 	unsigned long failover;
1195 	int ret;
1196 	struct nvme_ns *ns = nvme_get_ns_from_dev(dev);
1197 
1198 	ret = kstrtoul(buf, 0, &failover);
1199 	if (ret)
1200 		return -EINVAL;
1201 
1202 	atomic_long_set(&ns->failover, failover);
1203 
1204 	return count;
1205 }
1206 
1207 DEVICE_ATTR_RW(multipath_failover_count);
1208 
1209 static ssize_t io_requeue_no_usable_path_count_show(struct device *dev,
1210 		struct device_attribute *attr, char *buf)
1211 {
1212 	struct gendisk *disk = dev_to_disk(dev);
1213 	struct nvme_ns_head *head = disk->private_data;
1214 
1215 	return sysfs_emit(buf, "%lu\n",
1216 		    atomic_long_read(&head->io_requeue_no_usable_path_count));
1217 }
1218 
1219 static ssize_t io_requeue_no_usable_path_count_store(struct device *dev,
1220 		struct device_attribute *attr, const char *buf, size_t count)
1221 {
1222 	int err;
1223 	unsigned long requeue_cnt;
1224 	struct gendisk *disk = dev_to_disk(dev);
1225 	struct nvme_ns_head *head = disk->private_data;
1226 
1227 	err = kstrtoul(buf, 0, &requeue_cnt);
1228 	if (err)
1229 		return -EINVAL;
1230 
1231 	atomic_long_set(&head->io_requeue_no_usable_path_count, requeue_cnt);
1232 
1233 	return count;
1234 }
1235 
1236 DEVICE_ATTR_RW(io_requeue_no_usable_path_count);
1237 
1238 static ssize_t io_fail_no_available_path_count_show(struct device *dev,
1239 		struct device_attribute *attr, char *buf)
1240 {
1241 	struct gendisk *disk = dev_to_disk(dev);
1242 	struct nvme_ns_head *head = disk->private_data;
1243 
1244 	return sysfs_emit(buf, "%lu\n",
1245 		    atomic_long_read(&head->io_fail_no_available_path_count));
1246 }
1247 
1248 static ssize_t io_fail_no_available_path_count_store(struct device *dev,
1249 		struct device_attribute *attr, const char *buf, size_t count)
1250 {
1251 	int err;
1252 	unsigned long fail_cnt;
1253 	struct gendisk *disk = dev_to_disk(dev);
1254 	struct nvme_ns_head *head = disk->private_data;
1255 
1256 	err = kstrtoul(buf, 0, &fail_cnt);
1257 	if (err)
1258 		return -EINVAL;
1259 
1260 	atomic_long_set(&head->io_fail_no_available_path_count, fail_cnt);
1261 
1262 	return count;
1263 }
1264 
1265 DEVICE_ATTR_RW(io_fail_no_available_path_count);
1266 
1267 static int nvme_lookup_ana_group_desc(struct nvme_ctrl *ctrl,
1268 		struct nvme_ana_group_desc *desc, void *data)
1269 {
1270 	struct nvme_ana_group_desc *dst = data;
1271 
1272 	if (desc->grpid != dst->grpid)
1273 		return 0;
1274 
1275 	*dst = *desc;
1276 	return -ENXIO; /* just break out of the loop */
1277 }
1278 
1279 void nvme_mpath_add_sysfs_link(struct nvme_ns_head *head)
1280 {
1281 	struct device *target;
1282 	int rc, srcu_idx;
1283 	struct nvme_ns *ns;
1284 	struct kobject *kobj;
1285 
1286 	/*
1287 	 * Ensure head disk node is already added otherwise we may get invalid
1288 	 * kobj for head disk node
1289 	 */
1290 	if (!test_bit(GD_ADDED, &head->disk->state))
1291 		return;
1292 
1293 	kobj = &disk_to_dev(head->disk)->kobj;
1294 
1295 	/*
1296 	 * loop through each ns chained through the head->list and create the
1297 	 * sysfs link from head node to the ns path node
1298 	 */
1299 	srcu_idx = srcu_read_lock(&head->srcu);
1300 
1301 	list_for_each_entry_srcu(ns, &head->list, siblings,
1302 				 srcu_read_lock_held(&head->srcu)) {
1303 		/*
1304 		 * Ensure that ns path disk node is already added otherwise we
1305 		 * may get invalid kobj name for target
1306 		 */
1307 		if (!test_bit(GD_ADDED, &ns->disk->state))
1308 			continue;
1309 
1310 		/*
1311 		 * Avoid creating link if it already exists for the given path.
1312 		 * When path ana state transitions from optimized to non-
1313 		 * optimized or vice-versa, the nvme_mpath_set_live() is
1314 		 * invoked which in truns call this function. Now if the sysfs
1315 		 * link already exists for the given path and we attempt to re-
1316 		 * create the link then sysfs code would warn about it loudly.
1317 		 * So we evaluate NVME_NS_SYSFS_ATTR_LINK flag here to ensure
1318 		 * that we're not creating duplicate link.
1319 		 * The test_and_set_bit() is used because it is protecting
1320 		 * against multiple nvme paths being simultaneously added.
1321 		 */
1322 		if (test_and_set_bit(NVME_NS_SYSFS_ATTR_LINK, &ns->flags))
1323 			continue;
1324 
1325 		target = disk_to_dev(ns->disk);
1326 		/*
1327 		 * Create sysfs link from head gendisk kobject @kobj to the
1328 		 * ns path gendisk kobject @target->kobj.
1329 		 */
1330 		rc = sysfs_add_link_to_group(kobj, nvme_ns_mpath_attr_group.name,
1331 				&target->kobj, dev_name(target));
1332 		if (unlikely(rc)) {
1333 			dev_err(disk_to_dev(ns->head->disk),
1334 					"failed to create link to %s\n",
1335 					dev_name(target));
1336 			clear_bit(NVME_NS_SYSFS_ATTR_LINK, &ns->flags);
1337 		}
1338 	}
1339 
1340 	srcu_read_unlock(&head->srcu, srcu_idx);
1341 }
1342 
1343 void nvme_mpath_remove_sysfs_link(struct nvme_ns *ns)
1344 {
1345 	struct device *target;
1346 	struct kobject *kobj;
1347 
1348 	if (!test_bit(NVME_NS_SYSFS_ATTR_LINK, &ns->flags))
1349 		return;
1350 
1351 	target = disk_to_dev(ns->disk);
1352 	kobj = &disk_to_dev(ns->head->disk)->kobj;
1353 	sysfs_remove_link_from_group(kobj, nvme_ns_mpath_attr_group.name,
1354 			dev_name(target));
1355 	clear_bit(NVME_NS_SYSFS_ATTR_LINK, &ns->flags);
1356 }
1357 
1358 void nvme_mpath_add_disk(struct nvme_ns *ns, __le32 anagrpid)
1359 {
1360 	if (nvme_ctrl_use_ana(ns->ctrl)) {
1361 		struct nvme_ana_group_desc desc = {
1362 			.grpid = anagrpid,
1363 			.state = 0,
1364 		};
1365 
1366 		mutex_lock(&ns->ctrl->ana_lock);
1367 		ns->ana_grpid = le32_to_cpu(anagrpid);
1368 		nvme_parse_ana_log(ns->ctrl, &desc, nvme_lookup_ana_group_desc);
1369 		mutex_unlock(&ns->ctrl->ana_lock);
1370 		if (desc.state) {
1371 			/* found the group desc: update */
1372 			nvme_update_ns_ana_state(&desc, ns);
1373 		} else {
1374 			/* group desc not found: trigger a re-read */
1375 			set_bit(NVME_NS_ANA_PENDING, &ns->flags);
1376 			queue_work(nvme_wq, &ns->ctrl->ana_work);
1377 		}
1378 	} else {
1379 		ns->ana_state = NVME_ANA_OPTIMIZED;
1380 		nvme_mpath_set_live(ns);
1381 	}
1382 
1383 }
1384 
1385 void nvme_mpath_remove_disk(struct nvme_ns_head *head)
1386 {
1387 	bool remove = false;
1388 
1389 	if (!head->disk)
1390 		return;
1391 
1392 	mutex_lock(&head->subsys->lock);
1393 	/*
1394 	 * We are called when all paths have been removed, and at that point
1395 	 * head->list is expected to be empty. However, nvme_ns_remove() and
1396 	 * nvme_init_ns_head() can run concurrently and so if head->delayed_
1397 	 * removal_secs is configured, it is possible that by the time we reach
1398 	 * this point, head->list may no longer be empty. Therefore, we recheck
1399 	 * head->list here. If it is no longer empty then we skip enqueuing the
1400 	 * delayed head removal work.
1401 	 */
1402 	if (!list_empty(&head->list))
1403 		goto out;
1404 
1405 	/*
1406 	 * Ensure that no one could remove this module while the head
1407 	 * remove work is pending.
1408 	 */
1409 	if (head->delayed_removal_secs && try_module_get(THIS_MODULE)) {
1410 		mod_delayed_work(nvme_wq, &head->remove_work,
1411 				head->delayed_removal_secs * HZ);
1412 	} else {
1413 		list_del_init(&head->entry);
1414 		remove = true;
1415 	}
1416 out:
1417 	mutex_unlock(&head->subsys->lock);
1418 	if (remove)
1419 		nvme_remove_head(head);
1420 }
1421 
1422 void nvme_mpath_put_disk(struct nvme_ns_head *head)
1423 {
1424 	if (!head->disk)
1425 		return;
1426 	/* make sure all pending bios are cleaned up */
1427 	kblockd_schedule_work(&head->requeue_work);
1428 	flush_work(&head->requeue_work);
1429 	flush_work(&head->partition_scan_work);
1430 	put_disk(head->disk);
1431 }
1432 
1433 void nvme_mpath_init_ctrl(struct nvme_ctrl *ctrl)
1434 {
1435 	mutex_init(&ctrl->ana_lock);
1436 	timer_setup(&ctrl->anatt_timer, nvme_anatt_timeout, 0);
1437 	INIT_WORK(&ctrl->ana_work, nvme_ana_work);
1438 }
1439 
1440 int nvme_mpath_init_identify(struct nvme_ctrl *ctrl, struct nvme_id_ctrl *id)
1441 {
1442 	size_t max_transfer_size = ctrl->max_hw_sectors << SECTOR_SHIFT;
1443 	size_t ana_log_size;
1444 	int error = 0;
1445 
1446 	/* check if multipath is enabled and we have the capability */
1447 	if (!multipath || !ctrl->subsys ||
1448 	    !(ctrl->subsys->cmic & NVME_CTRL_CMIC_ANA))
1449 		return 0;
1450 
1451 	/* initialize this in the identify path to cover controller resets */
1452 	atomic_set(&ctrl->nr_active, 0);
1453 
1454 	if (!ctrl->max_namespaces ||
1455 	    ctrl->max_namespaces > le32_to_cpu(id->nn)) {
1456 		dev_err(ctrl->device,
1457 			"Invalid MNAN value %u\n", ctrl->max_namespaces);
1458 		return -EINVAL;
1459 	}
1460 
1461 	ctrl->anacap = id->anacap;
1462 	ctrl->anatt = id->anatt;
1463 	ctrl->nanagrpid = le32_to_cpu(id->nanagrpid);
1464 	ctrl->anagrpmax = le32_to_cpu(id->anagrpmax);
1465 
1466 	ana_log_size = sizeof(struct nvme_ana_rsp_hdr) +
1467 		ctrl->nanagrpid * sizeof(struct nvme_ana_group_desc) +
1468 		ctrl->max_namespaces * sizeof(__le32);
1469 	if (ana_log_size > max_transfer_size) {
1470 		dev_err(ctrl->device,
1471 			"ANA log page size (%zd) larger than MDTS (%zd).\n",
1472 			ana_log_size, max_transfer_size);
1473 		dev_err(ctrl->device, "disabling ANA support.\n");
1474 		goto out_uninit;
1475 	}
1476 	if (ana_log_size > ctrl->ana_log_size) {
1477 		nvme_mpath_stop(ctrl);
1478 		nvme_mpath_uninit(ctrl);
1479 		ctrl->ana_log_buf = kvmalloc(ana_log_size, GFP_KERNEL);
1480 		if (!ctrl->ana_log_buf)
1481 			return -ENOMEM;
1482 	}
1483 	ctrl->ana_log_size = ana_log_size;
1484 	error = nvme_read_ana_log(ctrl);
1485 	if (error)
1486 		goto out_uninit;
1487 	return 0;
1488 
1489 out_uninit:
1490 	nvme_mpath_uninit(ctrl);
1491 	return error;
1492 }
1493 
1494 void nvme_mpath_uninit(struct nvme_ctrl *ctrl)
1495 {
1496 	kvfree(ctrl->ana_log_buf);
1497 	ctrl->ana_log_buf = NULL;
1498 	ctrl->ana_log_size = 0;
1499 }
1500