1 // SPDX-License-Identifier: GPL-2.0 2 /* 3 * Common code for the NVMe target. 4 * Copyright (c) 2015-2016 HGST, a Western Digital Company. 5 */ 6 #define pr_fmt(fmt) KBUILD_MODNAME ": " fmt 7 #include <linux/hex.h> 8 #include <linux/module.h> 9 #include <linux/random.h> 10 #include <linux/rculist.h> 11 #include <linux/pci-p2pdma.h> 12 #include <linux/scatterlist.h> 13 14 #include <generated/utsrelease.h> 15 16 #define CREATE_TRACE_POINTS 17 #include "trace.h" 18 19 #include "nvmet.h" 20 #include "debugfs.h" 21 22 struct kmem_cache *nvmet_bvec_cache; 23 struct workqueue_struct *buffered_io_wq; 24 struct workqueue_struct *zbd_wq; 25 static const struct nvmet_fabrics_ops *nvmet_transports[NVMF_TRTYPE_MAX]; 26 static DEFINE_IDA(cntlid_ida); 27 28 struct workqueue_struct *nvmet_wq; 29 EXPORT_SYMBOL_GPL(nvmet_wq); 30 struct workqueue_struct *nvmet_aen_wq; 31 EXPORT_SYMBOL_GPL(nvmet_aen_wq); 32 33 /* 34 * This read/write semaphore is used to synchronize access to configuration 35 * information on a target system that will result in discovery log page 36 * information change for at least one host. 37 * The full list of resources to protected by this semaphore is: 38 * 39 * - subsystems list 40 * - per-subsystem allowed hosts list 41 * - allow_any_host subsystem attribute 42 * - nvmet_genctr 43 * - the nvmet_transports array 44 * 45 * When updating any of those lists/structures write lock should be obtained, 46 * while when reading (populating discovery log page or checking host-subsystem 47 * link) read lock is obtained to allow concurrent reads. 48 */ 49 DECLARE_RWSEM(nvmet_config_sem); 50 51 u32 nvmet_ana_group_enabled[NVMET_MAX_ANAGRPS + 1]; 52 u64 nvmet_ana_chgcnt; 53 DECLARE_RWSEM(nvmet_ana_sem); 54 55 inline u16 errno_to_nvme_status(struct nvmet_req *req, int errno) 56 { 57 switch (errno) { 58 case 0: 59 return NVME_SC_SUCCESS; 60 case -ENOSPC: 61 req->error_loc = offsetof(struct nvme_rw_command, length); 62 return NVME_SC_CAP_EXCEEDED | NVME_STATUS_DNR; 63 case -EREMOTEIO: 64 req->error_loc = offsetof(struct nvme_rw_command, slba); 65 return NVME_SC_LBA_RANGE | NVME_STATUS_DNR; 66 case -EOPNOTSUPP: 67 req->error_loc = offsetof(struct nvme_common_command, opcode); 68 return NVME_SC_INVALID_OPCODE | NVME_STATUS_DNR; 69 case -ENODATA: 70 req->error_loc = offsetof(struct nvme_rw_command, nsid); 71 return NVME_SC_ACCESS_DENIED; 72 case -EIO: 73 fallthrough; 74 default: 75 req->error_loc = offsetof(struct nvme_common_command, opcode); 76 return NVME_SC_INTERNAL | NVME_STATUS_DNR; 77 } 78 } 79 80 u16 nvmet_report_invalid_opcode(struct nvmet_req *req) 81 { 82 pr_debug("unhandled cmd %d on qid %d\n", req->cmd->common.opcode, 83 req->sq->qid); 84 85 req->error_loc = offsetof(struct nvme_common_command, opcode); 86 return NVME_SC_INVALID_OPCODE | NVME_STATUS_DNR; 87 } 88 89 static struct nvmet_subsys *nvmet_find_get_subsys(struct nvmet_port *port, 90 const char *subsysnqn); 91 92 u16 nvmet_copy_to_sgl(struct nvmet_req *req, off_t off, const void *buf, 93 size_t len) 94 { 95 if (sg_pcopy_from_buffer(req->sg, req->sg_cnt, buf, len, off) != len) { 96 req->error_loc = offsetof(struct nvme_common_command, dptr); 97 return NVME_SC_SGL_INVALID_DATA | NVME_STATUS_DNR; 98 } 99 return 0; 100 } 101 102 u16 nvmet_copy_from_sgl(struct nvmet_req *req, off_t off, void *buf, size_t len) 103 { 104 if (sg_pcopy_to_buffer(req->sg, req->sg_cnt, buf, len, off) != len) { 105 req->error_loc = offsetof(struct nvme_common_command, dptr); 106 return NVME_SC_SGL_INVALID_DATA | NVME_STATUS_DNR; 107 } 108 return 0; 109 } 110 111 u16 nvmet_zero_sgl(struct nvmet_req *req, off_t off, size_t len) 112 { 113 if (sg_zero_buffer(req->sg, req->sg_cnt, len, off) != len) { 114 req->error_loc = offsetof(struct nvme_common_command, dptr); 115 return NVME_SC_SGL_INVALID_DATA | NVME_STATUS_DNR; 116 } 117 return 0; 118 } 119 120 static u32 nvmet_max_nsid(struct nvmet_subsys *subsys) 121 { 122 struct nvmet_ns *cur; 123 unsigned long idx; 124 u32 nsid = 0; 125 126 nvmet_for_each_enabled_ns(&subsys->namespaces, idx, cur) 127 nsid = cur->nsid; 128 129 return nsid; 130 } 131 132 static u32 nvmet_async_event_result(struct nvmet_async_event *aen) 133 { 134 return aen->event_type | (aen->event_info << 8) | (aen->log_page << 16); 135 } 136 137 static void nvmet_async_events_failall(struct nvmet_ctrl *ctrl) 138 { 139 struct nvmet_req *req; 140 141 mutex_lock(&ctrl->lock); 142 while (ctrl->nr_async_event_cmds) { 143 req = ctrl->async_event_cmds[--ctrl->nr_async_event_cmds]; 144 mutex_unlock(&ctrl->lock); 145 nvmet_req_complete(req, NVME_SC_INTERNAL | NVME_STATUS_DNR); 146 mutex_lock(&ctrl->lock); 147 } 148 mutex_unlock(&ctrl->lock); 149 } 150 151 static void nvmet_async_events_process(struct nvmet_ctrl *ctrl) 152 { 153 struct nvmet_async_event *aen; 154 struct nvmet_req *req; 155 156 mutex_lock(&ctrl->lock); 157 while (ctrl->nr_async_event_cmds && !list_empty(&ctrl->async_events)) { 158 aen = list_first_entry(&ctrl->async_events, 159 struct nvmet_async_event, entry); 160 req = ctrl->async_event_cmds[--ctrl->nr_async_event_cmds]; 161 nvmet_set_result(req, nvmet_async_event_result(aen)); 162 163 list_del(&aen->entry); 164 kfree(aen); 165 166 mutex_unlock(&ctrl->lock); 167 trace_nvmet_async_event(ctrl, req->cqe->result.u32); 168 nvmet_req_complete(req, 0); 169 mutex_lock(&ctrl->lock); 170 } 171 mutex_unlock(&ctrl->lock); 172 } 173 174 static void nvmet_async_events_free(struct nvmet_ctrl *ctrl) 175 { 176 struct nvmet_async_event *aen, *tmp; 177 178 mutex_lock(&ctrl->lock); 179 list_for_each_entry_safe(aen, tmp, &ctrl->async_events, entry) { 180 list_del(&aen->entry); 181 kfree(aen); 182 } 183 mutex_unlock(&ctrl->lock); 184 } 185 186 static void nvmet_async_event_work(struct work_struct *work) 187 { 188 struct nvmet_ctrl *ctrl = 189 container_of(work, struct nvmet_ctrl, async_event_work); 190 191 nvmet_async_events_process(ctrl); 192 } 193 194 void nvmet_add_async_event(struct nvmet_ctrl *ctrl, u8 event_type, 195 u8 event_info, u8 log_page) 196 { 197 struct nvmet_async_event *aen; 198 199 aen = kmalloc_obj(*aen); 200 if (!aen) 201 return; 202 203 aen->event_type = event_type; 204 aen->event_info = event_info; 205 aen->log_page = log_page; 206 207 mutex_lock(&ctrl->lock); 208 list_add_tail(&aen->entry, &ctrl->async_events); 209 mutex_unlock(&ctrl->lock); 210 211 queue_work(nvmet_aen_wq, &ctrl->async_event_work); 212 } 213 214 static void nvmet_add_to_changed_ns_log(struct nvmet_ctrl *ctrl, __le32 nsid) 215 { 216 u32 i; 217 218 mutex_lock(&ctrl->lock); 219 if (ctrl->nr_changed_ns > NVME_MAX_CHANGED_NAMESPACES) 220 goto out_unlock; 221 222 for (i = 0; i < ctrl->nr_changed_ns; i++) { 223 if (ctrl->changed_ns_list[i] == nsid) 224 goto out_unlock; 225 } 226 227 if (ctrl->nr_changed_ns == NVME_MAX_CHANGED_NAMESPACES) { 228 ctrl->changed_ns_list[0] = cpu_to_le32(0xffffffff); 229 ctrl->nr_changed_ns = U32_MAX; 230 goto out_unlock; 231 } 232 233 ctrl->changed_ns_list[ctrl->nr_changed_ns++] = nsid; 234 out_unlock: 235 mutex_unlock(&ctrl->lock); 236 } 237 238 void nvmet_ns_changed(struct nvmet_subsys *subsys, u32 nsid) 239 { 240 struct nvmet_ctrl *ctrl; 241 242 lockdep_assert_held(&subsys->lock); 243 244 list_for_each_entry(ctrl, &subsys->ctrls, subsys_entry) { 245 nvmet_add_to_changed_ns_log(ctrl, cpu_to_le32(nsid)); 246 if (nvmet_aen_bit_disabled(ctrl, NVME_AEN_BIT_NS_ATTR)) 247 continue; 248 nvmet_add_async_event(ctrl, NVME_AER_NOTICE, 249 NVME_AER_NOTICE_NS_CHANGED, 250 NVME_LOG_CHANGED_NS); 251 } 252 } 253 254 void nvmet_send_ana_event(struct nvmet_subsys *subsys, 255 struct nvmet_port *port) 256 { 257 struct nvmet_ctrl *ctrl; 258 259 mutex_lock(&subsys->lock); 260 list_for_each_entry(ctrl, &subsys->ctrls, subsys_entry) { 261 if (port && ctrl->port != port) 262 continue; 263 if (nvmet_aen_bit_disabled(ctrl, NVME_AEN_BIT_ANA_CHANGE)) 264 continue; 265 nvmet_add_async_event(ctrl, NVME_AER_NOTICE, 266 NVME_AER_NOTICE_ANA, NVME_LOG_ANA); 267 } 268 mutex_unlock(&subsys->lock); 269 } 270 271 void nvmet_port_send_ana_event(struct nvmet_port *port) 272 { 273 struct nvmet_subsys_link *p; 274 275 down_read(&nvmet_config_sem); 276 list_for_each_entry(p, &port->subsystems, entry) 277 nvmet_send_ana_event(p->subsys, port); 278 up_read(&nvmet_config_sem); 279 } 280 281 int nvmet_register_transport(const struct nvmet_fabrics_ops *ops) 282 { 283 int ret = 0; 284 285 down_write(&nvmet_config_sem); 286 if (nvmet_transports[ops->type]) 287 ret = -EINVAL; 288 else 289 nvmet_transports[ops->type] = ops; 290 up_write(&nvmet_config_sem); 291 292 return ret; 293 } 294 EXPORT_SYMBOL_GPL(nvmet_register_transport); 295 296 void nvmet_unregister_transport(const struct nvmet_fabrics_ops *ops) 297 { 298 down_write(&nvmet_config_sem); 299 nvmet_transports[ops->type] = NULL; 300 up_write(&nvmet_config_sem); 301 } 302 EXPORT_SYMBOL_GPL(nvmet_unregister_transport); 303 304 void nvmet_port_del_ctrls(struct nvmet_port *port, struct nvmet_subsys *subsys) 305 { 306 struct nvmet_ctrl *ctrl; 307 308 mutex_lock(&subsys->lock); 309 list_for_each_entry(ctrl, &subsys->ctrls, subsys_entry) { 310 if (ctrl->port == port) 311 ctrl->ops->delete_ctrl(ctrl); 312 } 313 mutex_unlock(&subsys->lock); 314 } 315 316 int nvmet_enable_port(struct nvmet_port *port) 317 { 318 const struct nvmet_fabrics_ops *ops; 319 int ret; 320 321 lockdep_assert_held(&nvmet_config_sem); 322 323 if (port->disc_addr.trtype == NVMF_TRTYPE_MAX) 324 return -EINVAL; 325 326 ops = nvmet_transports[port->disc_addr.trtype]; 327 if (!ops) { 328 up_write(&nvmet_config_sem); 329 request_module("nvmet-transport-%d", port->disc_addr.trtype); 330 down_write(&nvmet_config_sem); 331 ops = nvmet_transports[port->disc_addr.trtype]; 332 if (!ops) { 333 pr_err("transport type %d not supported\n", 334 port->disc_addr.trtype); 335 return -EINVAL; 336 } 337 } 338 339 if (!try_module_get(ops->owner)) 340 return -EINVAL; 341 342 /* 343 * If the user requested PI support and the transport isn't pi capable, 344 * don't enable the port. 345 */ 346 if (port->pi_enable && !(ops->flags & NVMF_METADATA_SUPPORTED)) { 347 pr_err("T10-PI is not supported by transport type %d\n", 348 port->disc_addr.trtype); 349 ret = -EINVAL; 350 goto out_put; 351 } 352 353 ret = ops->add_port(port); 354 if (ret) 355 goto out_put; 356 357 /* If the transport didn't set inline_data_size, then disable it. */ 358 if (port->inline_data_size < 0) 359 port->inline_data_size = 0; 360 361 /* 362 * If the transport didn't set the max_queue_size properly, then clamp 363 * it to the target limits. Also set default values in case the 364 * transport didn't set it at all. 365 */ 366 if (port->max_queue_size < 0) 367 port->max_queue_size = NVMET_MAX_QUEUE_SIZE; 368 else 369 port->max_queue_size = clamp_t(int, port->max_queue_size, 370 NVMET_MIN_QUEUE_SIZE, 371 NVMET_MAX_QUEUE_SIZE); 372 373 /* 374 * If the transport didn't set the mdts properly, then clamp it to the 375 * target limits. Also set default values in case the transport didn't 376 * set it at all. 377 */ 378 if (port->mdts < 0 || port->mdts > NVMET_MAX_MDTS) 379 port->mdts = 0; 380 381 port->enabled = true; 382 port->tr_ops = ops; 383 return 0; 384 385 out_put: 386 module_put(ops->owner); 387 return ret; 388 } 389 390 void nvmet_disable_port(struct nvmet_port *port) 391 { 392 const struct nvmet_fabrics_ops *ops; 393 394 lockdep_assert_held(&nvmet_config_sem); 395 396 port->enabled = false; 397 port->tr_ops = NULL; 398 399 ops = nvmet_transports[port->disc_addr.trtype]; 400 ops->remove_port(port); 401 module_put(ops->owner); 402 } 403 404 static void nvmet_keep_alive_timer(struct work_struct *work) 405 { 406 struct nvmet_ctrl *ctrl = container_of(to_delayed_work(work), 407 struct nvmet_ctrl, ka_work); 408 bool reset_tbkas = ctrl->reset_tbkas; 409 410 ctrl->reset_tbkas = false; 411 if (reset_tbkas) { 412 pr_debug("ctrl %d reschedule traffic based keep-alive timer\n", 413 ctrl->cntlid); 414 queue_delayed_work(nvmet_wq, &ctrl->ka_work, ctrl->kato * HZ); 415 return; 416 } 417 418 pr_err("ctrl %d keep-alive timer (%d seconds) expired!\n", 419 ctrl->cntlid, ctrl->kato); 420 421 nvmet_ctrl_fatal_error(ctrl); 422 } 423 424 void nvmet_start_keep_alive_timer(struct nvmet_ctrl *ctrl) 425 { 426 if (unlikely(ctrl->kato == 0)) 427 return; 428 429 pr_debug("ctrl %d start keep-alive timer for %d secs\n", 430 ctrl->cntlid, ctrl->kato); 431 432 queue_delayed_work(nvmet_wq, &ctrl->ka_work, ctrl->kato * HZ); 433 } 434 435 void nvmet_stop_keep_alive_timer(struct nvmet_ctrl *ctrl) 436 { 437 if (unlikely(ctrl->kato == 0)) 438 return; 439 440 pr_debug("ctrl %d stop keep-alive\n", ctrl->cntlid); 441 442 cancel_delayed_work_sync(&ctrl->ka_work); 443 } 444 445 u16 nvmet_req_find_ns(struct nvmet_req *req) 446 { 447 u32 nsid = le32_to_cpu(req->cmd->common.nsid); 448 struct nvmet_subsys *subsys = nvmet_req_subsys(req); 449 u16 status = NVME_SC_SUCCESS; 450 451 rcu_read_lock(); 452 req->ns = xa_load(&subsys->namespaces, nsid); 453 if (unlikely(!req->ns) || 454 !test_bit(NVMET_NS_IO_LIVE, &req->ns->flags) || 455 !percpu_ref_tryget_live_rcu(&req->ns->ref)) { 456 req->error_loc = offsetof(struct nvme_common_command, nsid); 457 if (!req->ns) { /* ns doesn't exist! */ 458 status = NVME_SC_INVALID_NS | NVME_STATUS_DNR; 459 goto unlock; 460 } 461 462 /* ns exists but it's disabled */ 463 req->ns = NULL; 464 status = NVME_SC_INTERNAL_PATH_ERROR; 465 } 466 unlock: 467 rcu_read_unlock(); 468 469 return status; 470 } 471 472 static void nvmet_destroy_namespace(struct percpu_ref *ref) 473 { 474 struct nvmet_ns *ns = container_of(ref, struct nvmet_ns, ref); 475 476 complete(&ns->disable_done); 477 } 478 479 void nvmet_put_namespace(struct nvmet_ns *ns) 480 { 481 percpu_ref_put(&ns->ref); 482 } 483 484 static void nvmet_ns_dev_disable(struct nvmet_ns *ns) 485 { 486 nvmet_bdev_ns_disable(ns); 487 nvmet_file_ns_disable(ns); 488 } 489 490 static int nvmet_p2pmem_ns_enable(struct nvmet_ns *ns) 491 { 492 int ret; 493 struct pci_dev *p2p_dev; 494 495 if (!ns->use_p2pmem) 496 return 0; 497 498 if (!ns->bdev) { 499 pr_err("peer-to-peer DMA is not supported by non-block device namespaces\n"); 500 return -EINVAL; 501 } 502 503 if (!blk_queue_pci_p2pdma(ns->bdev->bd_disk->queue)) { 504 pr_err("peer-to-peer DMA is not supported by the driver of %s\n", 505 ns->device_path); 506 return -EINVAL; 507 } 508 509 if (ns->p2p_dev) { 510 ret = pci_p2pdma_distance(ns->p2p_dev, nvmet_ns_dev(ns), true); 511 if (ret < 0) 512 return -EINVAL; 513 } else { 514 /* 515 * Right now we just check that there is p2pmem available so 516 * we can report an error to the user right away if there 517 * is not. We'll find the actual device to use once we 518 * setup the controller when the port's device is available. 519 */ 520 521 p2p_dev = pci_p2pmem_find(nvmet_ns_dev(ns)); 522 if (!p2p_dev) { 523 pr_err("no peer-to-peer memory is available for %s\n", 524 ns->device_path); 525 return -EINVAL; 526 } 527 528 pci_dev_put(p2p_dev); 529 } 530 531 return 0; 532 } 533 534 static void nvmet_p2pmem_ns_add_p2p(struct nvmet_ctrl *ctrl, 535 struct nvmet_ns *ns) 536 { 537 struct device *clients[2]; 538 struct pci_dev *p2p_dev; 539 int ret; 540 541 lockdep_assert_held(&ctrl->subsys->lock); 542 543 if (!ctrl->p2p_client || !ns->use_p2pmem) 544 return; 545 546 if (ns->p2p_dev) { 547 ret = pci_p2pdma_distance(ns->p2p_dev, ctrl->p2p_client, true); 548 if (ret < 0) 549 return; 550 551 p2p_dev = pci_dev_get(ns->p2p_dev); 552 } else { 553 clients[0] = ctrl->p2p_client; 554 clients[1] = nvmet_ns_dev(ns); 555 556 p2p_dev = pci_p2pmem_find_many(clients, ARRAY_SIZE(clients)); 557 if (!p2p_dev) { 558 pr_err("no peer-to-peer memory is available that's supported by %s and %s\n", 559 dev_name(ctrl->p2p_client), ns->device_path); 560 return; 561 } 562 } 563 564 ret = radix_tree_insert(&ctrl->p2p_ns_map, ns->nsid, p2p_dev); 565 if (ret < 0) 566 pci_dev_put(p2p_dev); 567 568 pr_info("using p2pmem on %s for nsid %u\n", pci_name(p2p_dev), 569 ns->nsid); 570 } 571 572 bool nvmet_ns_revalidate(struct nvmet_ns *ns) 573 { 574 loff_t oldsize = ns->size; 575 576 if (ns->bdev) 577 nvmet_bdev_ns_revalidate(ns); 578 else 579 nvmet_file_ns_revalidate(ns); 580 581 return oldsize != ns->size; 582 } 583 584 int nvmet_ns_enable(struct nvmet_ns *ns) 585 { 586 struct nvmet_subsys *subsys = ns->subsys; 587 struct nvmet_ctrl *ctrl; 588 int ret; 589 590 mutex_lock(&subsys->lock); 591 ret = 0; 592 593 if (nvmet_is_passthru_subsys(subsys)) { 594 pr_info("cannot enable both passthru and regular namespaces for a single subsystem"); 595 goto out_unlock; 596 } 597 598 if (ns->enabled) 599 goto out_unlock; 600 601 if (!ns->device_path) { 602 ret = -EINVAL; 603 goto out_unlock; 604 } 605 606 ret = nvmet_bdev_ns_enable(ns); 607 if (ret == -ENOTBLK) 608 ret = nvmet_file_ns_enable(ns); 609 if (ret) 610 goto out_unlock; 611 612 ret = nvmet_p2pmem_ns_enable(ns); 613 if (ret) 614 goto out_dev_disable; 615 616 list_for_each_entry(ctrl, &subsys->ctrls, subsys_entry) 617 nvmet_p2pmem_ns_add_p2p(ctrl, ns); 618 619 if (ns->pr.enable) { 620 ret = nvmet_pr_init_ns(ns); 621 if (ret) 622 goto out_dev_put; 623 } 624 625 ret = percpu_ref_init(&ns->ref, nvmet_destroy_namespace, 0, GFP_KERNEL); 626 if (ret) 627 goto out_pr_exit; 628 629 nvmet_ns_changed(subsys, ns->nsid); 630 ns->enabled = true; 631 xa_set_mark(&subsys->namespaces, ns->nsid, NVMET_NS_ENABLED); 632 nvmet_debugfs_ns_setup(ns); 633 set_bit(NVMET_NS_IO_LIVE, &ns->flags); 634 ret = 0; 635 out_unlock: 636 mutex_unlock(&subsys->lock); 637 return ret; 638 out_pr_exit: 639 if (ns->pr.enable) 640 nvmet_pr_exit_ns(ns); 641 out_dev_put: 642 list_for_each_entry(ctrl, &subsys->ctrls, subsys_entry) 643 pci_dev_put(radix_tree_delete(&ctrl->p2p_ns_map, ns->nsid)); 644 out_dev_disable: 645 nvmet_ns_dev_disable(ns); 646 goto out_unlock; 647 } 648 649 void nvmet_ns_disable(struct nvmet_ns *ns) 650 { 651 struct nvmet_subsys *subsys = ns->subsys; 652 struct nvmet_ctrl *ctrl; 653 654 if (!test_and_clear_bit(NVMET_NS_IO_LIVE, &ns->flags)) 655 return; 656 657 mutex_lock(&subsys->lock); 658 659 xa_clear_mark(&subsys->namespaces, ns->nsid, NVMET_NS_ENABLED); 660 nvmet_debugfs_ns_free(ns); 661 662 list_for_each_entry(ctrl, &subsys->ctrls, subsys_entry) 663 pci_dev_put(radix_tree_delete(&ctrl->p2p_ns_map, ns->nsid)); 664 665 mutex_unlock(&subsys->lock); 666 667 /* 668 * Now that we removed the namespaces from the lookup list, we 669 * can kill the per_cpu ref and wait for any remaining references 670 * to be dropped, as well as a RCU grace period for anyone only 671 * using the namespace under rcu_read_lock(). Note that we can't 672 * use call_rcu here as we need to ensure the namespaces have 673 * been fully destroyed before unloading the module. 674 */ 675 percpu_ref_kill(&ns->ref); 676 synchronize_rcu(); 677 wait_for_completion(&ns->disable_done); 678 percpu_ref_exit(&ns->ref); 679 680 if (ns->pr.enable) 681 nvmet_pr_exit_ns(ns); 682 683 mutex_lock(&subsys->lock); 684 nvmet_ns_changed(subsys, ns->nsid); 685 nvmet_ns_dev_disable(ns); 686 ns->enabled = false; 687 mutex_unlock(&subsys->lock); 688 } 689 690 void nvmet_ns_free(struct nvmet_ns *ns) 691 { 692 struct nvmet_subsys *subsys = ns->subsys; 693 694 nvmet_ns_disable(ns); 695 696 mutex_lock(&subsys->lock); 697 698 xa_erase(&subsys->namespaces, ns->nsid); 699 if (ns->nsid == subsys->max_nsid) 700 subsys->max_nsid = nvmet_max_nsid(subsys); 701 702 subsys->nr_namespaces--; 703 mutex_unlock(&subsys->lock); 704 705 down_write(&nvmet_ana_sem); 706 nvmet_ana_group_enabled[ns->anagrpid]--; 707 up_write(&nvmet_ana_sem); 708 709 kfree(ns->device_path); 710 kfree(ns); 711 } 712 713 struct nvmet_ns *nvmet_ns_alloc(struct nvmet_subsys *subsys, u32 nsid) 714 { 715 struct nvmet_ns *ns; 716 717 mutex_lock(&subsys->lock); 718 719 if (subsys->nr_namespaces == NVMET_MAX_NAMESPACES) 720 goto out_unlock; 721 722 ns = kzalloc_obj(*ns); 723 if (!ns) 724 goto out_unlock; 725 726 init_completion(&ns->disable_done); 727 728 ns->nsid = nsid; 729 ns->subsys = subsys; 730 731 if (ns->nsid > subsys->max_nsid) 732 subsys->max_nsid = nsid; 733 734 if (xa_insert(&subsys->namespaces, ns->nsid, ns, GFP_KERNEL)) 735 goto out_exit; 736 737 subsys->nr_namespaces++; 738 739 mutex_unlock(&subsys->lock); 740 741 down_write(&nvmet_ana_sem); 742 ns->anagrpid = NVMET_DEFAULT_ANA_GRPID; 743 nvmet_ana_group_enabled[ns->anagrpid]++; 744 up_write(&nvmet_ana_sem); 745 746 uuid_gen(&ns->uuid); 747 ns->buffered_io = false; 748 ns->csi = NVME_CSI_NVM; 749 750 return ns; 751 out_exit: 752 subsys->max_nsid = nvmet_max_nsid(subsys); 753 kfree(ns); 754 out_unlock: 755 mutex_unlock(&subsys->lock); 756 return NULL; 757 } 758 759 static void nvmet_update_sq_head(struct nvmet_req *req) 760 { 761 if (req->sq->size) { 762 u32 old_sqhd, new_sqhd; 763 764 old_sqhd = READ_ONCE(req->sq->sqhd); 765 do { 766 new_sqhd = (old_sqhd + 1) % req->sq->size; 767 } while (!try_cmpxchg(&req->sq->sqhd, &old_sqhd, new_sqhd)); 768 } 769 req->cqe->sq_head = cpu_to_le16(req->sq->sqhd & 0x0000FFFF); 770 } 771 772 static void nvmet_set_error(struct nvmet_req *req, u16 status) 773 { 774 struct nvmet_ctrl *ctrl = req->sq->ctrl; 775 struct nvme_error_slot *new_error_slot; 776 unsigned long flags; 777 778 req->cqe->status = cpu_to_le16(status << 1); 779 780 if (!ctrl || req->error_loc == NVMET_NO_ERROR_LOC) 781 return; 782 783 spin_lock_irqsave(&ctrl->error_lock, flags); 784 ctrl->err_counter++; 785 new_error_slot = 786 &ctrl->slots[ctrl->err_counter % NVMET_ERROR_LOG_SLOTS]; 787 788 new_error_slot->error_count = cpu_to_le64(ctrl->err_counter); 789 new_error_slot->sqid = cpu_to_le16(req->sq->qid); 790 new_error_slot->cmdid = cpu_to_le16(req->cmd->common.command_id); 791 new_error_slot->status_field = cpu_to_le16(status << 1); 792 new_error_slot->param_error_location = cpu_to_le16(req->error_loc); 793 new_error_slot->lba = cpu_to_le64(req->error_slba); 794 new_error_slot->nsid = req->cmd->common.nsid; 795 spin_unlock_irqrestore(&ctrl->error_lock, flags); 796 797 /* set the more bit for this request */ 798 req->cqe->status |= cpu_to_le16(1 << 14); 799 } 800 801 static void __nvmet_req_complete(struct nvmet_req *req, u16 status) 802 { 803 struct nvmet_ns *ns = req->ns; 804 struct nvmet_pr_per_ctrl_ref *pc_ref = req->pc_ref; 805 806 if (!req->sq->sqhd_disabled) 807 nvmet_update_sq_head(req); 808 req->cqe->sq_id = cpu_to_le16(req->sq->qid); 809 req->cqe->command_id = req->cmd->common.command_id; 810 811 if (unlikely(status)) 812 nvmet_set_error(req, status); 813 814 trace_nvmet_req_complete(req); 815 816 req->ops->queue_response(req); 817 818 if (pc_ref) 819 nvmet_pr_put_ns_pc_ref(pc_ref); 820 if (ns) 821 nvmet_put_namespace(ns); 822 } 823 824 void nvmet_req_complete(struct nvmet_req *req, u16 status) 825 { 826 struct nvmet_sq *sq = req->sq; 827 828 __nvmet_req_complete(req, status); 829 percpu_ref_put(&sq->ref); 830 } 831 EXPORT_SYMBOL_GPL(nvmet_req_complete); 832 833 void nvmet_cq_init(struct nvmet_cq *cq) 834 { 835 refcount_set(&cq->ref, 1); 836 } 837 EXPORT_SYMBOL_GPL(nvmet_cq_init); 838 839 bool nvmet_cq_get(struct nvmet_cq *cq) 840 { 841 return refcount_inc_not_zero(&cq->ref); 842 } 843 EXPORT_SYMBOL_GPL(nvmet_cq_get); 844 845 void nvmet_cq_put(struct nvmet_cq *cq) 846 { 847 if (refcount_dec_and_test(&cq->ref)) 848 nvmet_cq_destroy(cq); 849 } 850 EXPORT_SYMBOL_GPL(nvmet_cq_put); 851 852 void nvmet_cq_setup(struct nvmet_ctrl *ctrl, struct nvmet_cq *cq, 853 u16 qid, u16 size) 854 { 855 cq->qid = qid; 856 cq->size = size; 857 858 ctrl->cqs[qid] = cq; 859 } 860 861 void nvmet_cq_destroy(struct nvmet_cq *cq) 862 { 863 struct nvmet_ctrl *ctrl = cq->ctrl; 864 865 if (ctrl) { 866 ctrl->cqs[cq->qid] = NULL; 867 nvmet_ctrl_put(cq->ctrl); 868 cq->ctrl = NULL; 869 } 870 } 871 872 void nvmet_sq_setup(struct nvmet_ctrl *ctrl, struct nvmet_sq *sq, 873 u16 qid, u16 size) 874 { 875 sq->sqhd = 0; 876 sq->qid = qid; 877 sq->size = size; 878 879 ctrl->sqs[qid] = sq; 880 } 881 882 static void nvmet_confirm_sq(struct percpu_ref *ref) 883 { 884 struct nvmet_sq *sq = container_of(ref, struct nvmet_sq, ref); 885 886 complete(&sq->confirm_done); 887 } 888 889 u16 nvmet_check_cqid(struct nvmet_ctrl *ctrl, u16 cqid, bool create) 890 { 891 if (!ctrl->cqs) 892 return NVME_SC_INTERNAL | NVME_STATUS_DNR; 893 894 if (cqid > ctrl->max_qid) 895 return NVME_SC_QID_INVALID | NVME_STATUS_DNR; 896 897 if ((create && ctrl->cqs[cqid]) || (!create && !ctrl->cqs[cqid])) 898 return NVME_SC_QID_INVALID | NVME_STATUS_DNR; 899 900 return NVME_SC_SUCCESS; 901 } 902 903 u16 nvmet_check_io_cqid(struct nvmet_ctrl *ctrl, u16 cqid, bool create) 904 { 905 if (!cqid) 906 return NVME_SC_QID_INVALID | NVME_STATUS_DNR; 907 return nvmet_check_cqid(ctrl, cqid, create); 908 } 909 910 bool nvmet_cq_in_use(struct nvmet_cq *cq) 911 { 912 return refcount_read(&cq->ref) > 1; 913 } 914 EXPORT_SYMBOL_GPL(nvmet_cq_in_use); 915 916 u16 nvmet_cq_create(struct nvmet_ctrl *ctrl, struct nvmet_cq *cq, 917 u16 qid, u16 size) 918 { 919 u16 status; 920 921 status = nvmet_check_cqid(ctrl, qid, true); 922 if (status != NVME_SC_SUCCESS) 923 return status; 924 925 if (!kref_get_unless_zero(&ctrl->ref)) 926 return NVME_SC_INTERNAL | NVME_STATUS_DNR; 927 cq->ctrl = ctrl; 928 929 nvmet_cq_init(cq); 930 nvmet_cq_setup(ctrl, cq, qid, size); 931 932 return NVME_SC_SUCCESS; 933 } 934 EXPORT_SYMBOL_GPL(nvmet_cq_create); 935 936 u16 nvmet_check_sqid(struct nvmet_ctrl *ctrl, u16 sqid, 937 bool create) 938 { 939 if (!ctrl->sqs) 940 return NVME_SC_INTERNAL | NVME_STATUS_DNR; 941 942 if (sqid > ctrl->max_qid) 943 return NVME_SC_QID_INVALID | NVME_STATUS_DNR; 944 945 if ((create && ctrl->sqs[sqid]) || 946 (!create && !ctrl->sqs[sqid])) 947 return NVME_SC_QID_INVALID | NVME_STATUS_DNR; 948 949 return NVME_SC_SUCCESS; 950 } 951 952 u16 nvmet_sq_create(struct nvmet_ctrl *ctrl, struct nvmet_sq *sq, 953 struct nvmet_cq *cq, u16 sqid, u16 size) 954 { 955 u16 status; 956 int ret; 957 958 if (!kref_get_unless_zero(&ctrl->ref)) 959 return NVME_SC_INTERNAL | NVME_STATUS_DNR; 960 961 status = nvmet_check_sqid(ctrl, sqid, true); 962 if (status != NVME_SC_SUCCESS) 963 goto ctrl_put; 964 965 ret = nvmet_sq_init(sq, cq); 966 if (ret) { 967 status = NVME_SC_INTERNAL | NVME_STATUS_DNR; 968 goto ctrl_put; 969 } 970 971 nvmet_sq_setup(ctrl, sq, sqid, size); 972 sq->ctrl = ctrl; 973 974 return NVME_SC_SUCCESS; 975 976 ctrl_put: 977 nvmet_ctrl_put(ctrl); 978 return status; 979 } 980 EXPORT_SYMBOL_GPL(nvmet_sq_create); 981 982 void nvmet_sq_destroy(struct nvmet_sq *sq) 983 { 984 struct nvmet_ctrl *ctrl = sq->ctrl; 985 986 /* 987 * If this is the admin queue, complete all AERs so that our 988 * queue doesn't have outstanding requests on it. 989 */ 990 if (ctrl && ctrl->sqs && ctrl->sqs[0] == sq) 991 nvmet_async_events_failall(ctrl); 992 percpu_ref_kill_and_confirm(&sq->ref, nvmet_confirm_sq); 993 wait_for_completion(&sq->confirm_done); 994 wait_for_completion(&sq->free_done); 995 percpu_ref_exit(&sq->ref); 996 nvmet_auth_sq_destroy(sq); 997 nvmet_cq_put(sq->cq); 998 999 /* 1000 * we must reference the ctrl again after waiting for inflight IO 1001 * to complete. Because admin connect may have sneaked in after we 1002 * store sq->ctrl locally, but before we killed the percpu_ref. the 1003 * admin connect allocates and assigns sq->ctrl, which now needs a 1004 * final ref put, as this ctrl is going away. 1005 */ 1006 ctrl = sq->ctrl; 1007 1008 if (ctrl) { 1009 /* 1010 * The teardown flow may take some time, and the host may not 1011 * send us keep-alive during this period, hence reset the 1012 * traffic based keep-alive timer so we don't trigger a 1013 * controller teardown as a result of a keep-alive expiration. 1014 */ 1015 ctrl->reset_tbkas = true; 1016 sq->ctrl->sqs[sq->qid] = NULL; 1017 nvmet_ctrl_put(ctrl); 1018 sq->ctrl = NULL; /* allows reusing the queue later */ 1019 } 1020 } 1021 EXPORT_SYMBOL_GPL(nvmet_sq_destroy); 1022 1023 static void nvmet_sq_free(struct percpu_ref *ref) 1024 { 1025 struct nvmet_sq *sq = container_of(ref, struct nvmet_sq, ref); 1026 1027 complete(&sq->free_done); 1028 } 1029 1030 int nvmet_sq_init(struct nvmet_sq *sq, struct nvmet_cq *cq) 1031 { 1032 int ret; 1033 1034 if (!nvmet_cq_get(cq)) 1035 return -EINVAL; 1036 1037 ret = percpu_ref_init(&sq->ref, nvmet_sq_free, 0, GFP_KERNEL); 1038 if (ret) { 1039 pr_err("percpu_ref init failed!\n"); 1040 nvmet_cq_put(cq); 1041 return ret; 1042 } 1043 init_completion(&sq->free_done); 1044 init_completion(&sq->confirm_done); 1045 nvmet_auth_sq_init(sq); 1046 sq->cq = cq; 1047 1048 return 0; 1049 } 1050 EXPORT_SYMBOL_GPL(nvmet_sq_init); 1051 1052 static inline u16 nvmet_check_ana_state(struct nvmet_port *port, 1053 struct nvmet_ns *ns) 1054 { 1055 enum nvme_ana_state state = port->ana_state[ns->anagrpid]; 1056 1057 if (unlikely(state == NVME_ANA_INACCESSIBLE)) 1058 return NVME_SC_ANA_INACCESSIBLE; 1059 if (unlikely(state == NVME_ANA_PERSISTENT_LOSS)) 1060 return NVME_SC_ANA_PERSISTENT_LOSS; 1061 if (unlikely(state == NVME_ANA_CHANGE)) 1062 return NVME_SC_ANA_TRANSITION; 1063 return 0; 1064 } 1065 1066 static inline u16 nvmet_io_cmd_check_access(struct nvmet_req *req) 1067 { 1068 if (unlikely(req->ns->readonly)) { 1069 switch (req->cmd->common.opcode) { 1070 case nvme_cmd_read: 1071 case nvme_cmd_flush: 1072 break; 1073 default: 1074 return NVME_SC_NS_WRITE_PROTECTED; 1075 } 1076 } 1077 1078 return 0; 1079 } 1080 1081 static u32 nvmet_io_cmd_transfer_len(struct nvmet_req *req) 1082 { 1083 struct nvme_command *cmd = req->cmd; 1084 u32 metadata_len = 0; 1085 1086 if (nvme_is_fabrics(cmd)) 1087 return nvmet_fabrics_io_cmd_data_len(req); 1088 1089 if (!req->ns) 1090 return 0; 1091 1092 switch (req->cmd->common.opcode) { 1093 case nvme_cmd_read: 1094 case nvme_cmd_write: 1095 case nvme_cmd_zone_append: 1096 if (req->sq->ctrl->pi_support && nvmet_ns_has_pi(req->ns)) 1097 metadata_len = nvmet_rw_metadata_len(req); 1098 return nvmet_rw_data_len(req) + metadata_len; 1099 case nvme_cmd_dsm: 1100 return nvmet_dsm_len(req); 1101 case nvme_cmd_zone_mgmt_recv: 1102 return (le32_to_cpu(req->cmd->zmr.numd) + 1) << 2; 1103 default: 1104 return 0; 1105 } 1106 } 1107 1108 static u16 nvmet_parse_io_cmd(struct nvmet_req *req) 1109 { 1110 struct nvme_command *cmd = req->cmd; 1111 u16 ret; 1112 1113 if (nvme_is_fabrics(cmd)) 1114 return nvmet_parse_fabrics_io_cmd(req); 1115 1116 if (unlikely(!nvmet_check_auth_status(req))) 1117 return NVME_SC_AUTH_REQUIRED | NVME_STATUS_DNR; 1118 1119 ret = nvmet_check_ctrl_status(req); 1120 if (unlikely(ret)) 1121 return ret; 1122 1123 if (nvmet_is_passthru_req(req)) 1124 return nvmet_parse_passthru_io_cmd(req); 1125 1126 ret = nvmet_req_find_ns(req); 1127 if (unlikely(ret)) 1128 return ret; 1129 1130 ret = nvmet_check_ana_state(req->port, req->ns); 1131 if (unlikely(ret)) { 1132 req->error_loc = offsetof(struct nvme_common_command, nsid); 1133 return ret; 1134 } 1135 ret = nvmet_io_cmd_check_access(req); 1136 if (unlikely(ret)) { 1137 req->error_loc = offsetof(struct nvme_common_command, nsid); 1138 return ret; 1139 } 1140 1141 if (req->ns->pr.enable) { 1142 ret = nvmet_parse_pr_cmd(req); 1143 if (!ret) 1144 return ret; 1145 } 1146 1147 switch (req->ns->csi) { 1148 case NVME_CSI_NVM: 1149 if (req->ns->file) 1150 ret = nvmet_file_parse_io_cmd(req); 1151 else 1152 ret = nvmet_bdev_parse_io_cmd(req); 1153 break; 1154 case NVME_CSI_ZNS: 1155 if (IS_ENABLED(CONFIG_BLK_DEV_ZONED)) 1156 ret = nvmet_bdev_zns_parse_io_cmd(req); 1157 else 1158 ret = NVME_SC_INVALID_IO_CMD_SET; 1159 break; 1160 default: 1161 ret = NVME_SC_INVALID_IO_CMD_SET; 1162 } 1163 if (ret) 1164 return ret; 1165 1166 if (req->ns->pr.enable) { 1167 ret = nvmet_pr_check_cmd_access(req); 1168 if (ret) 1169 return ret; 1170 1171 ret = nvmet_pr_get_ns_pc_ref(req); 1172 } 1173 return ret; 1174 } 1175 1176 bool nvmet_req_init(struct nvmet_req *req, struct nvmet_sq *sq, 1177 const struct nvmet_fabrics_ops *ops) 1178 { 1179 u8 flags = req->cmd->common.flags; 1180 u16 status; 1181 1182 req->cq = sq->cq; 1183 req->sq = sq; 1184 req->ops = ops; 1185 req->sg = NULL; 1186 req->metadata_sg = NULL; 1187 req->sg_cnt = 0; 1188 req->metadata_sg_cnt = 0; 1189 req->transfer_len = 0; 1190 req->metadata_len = 0; 1191 req->cqe->result.u64 = 0; 1192 req->cqe->status = 0; 1193 req->cqe->sq_head = 0; 1194 req->ns = NULL; 1195 req->error_loc = NVMET_NO_ERROR_LOC; 1196 req->error_slba = 0; 1197 req->pc_ref = NULL; 1198 1199 /* no support for fused commands yet */ 1200 if (unlikely(flags & (NVME_CMD_FUSE_FIRST | NVME_CMD_FUSE_SECOND))) { 1201 req->error_loc = offsetof(struct nvme_common_command, flags); 1202 status = NVME_SC_INVALID_FIELD | NVME_STATUS_DNR; 1203 goto fail; 1204 } 1205 1206 /* 1207 * For fabrics, PSDT field shall describe metadata pointer (MPTR) that 1208 * contains an address of a single contiguous physical buffer that is 1209 * byte aligned. For PCI controllers, this is optional so not enforced. 1210 */ 1211 if (unlikely((flags & NVME_CMD_SGL_ALL) != NVME_CMD_SGL_METABUF)) { 1212 if (!req->sq->ctrl || !nvmet_is_pci_ctrl(req->sq->ctrl)) { 1213 req->error_loc = 1214 offsetof(struct nvme_common_command, flags); 1215 status = NVME_SC_INVALID_FIELD | NVME_STATUS_DNR; 1216 goto fail; 1217 } 1218 } 1219 1220 if (unlikely(!req->sq->ctrl)) 1221 /* will return an error for any non-connect command: */ 1222 status = nvmet_parse_connect_cmd(req); 1223 else if (likely(req->sq->qid != 0)) 1224 status = nvmet_parse_io_cmd(req); 1225 else 1226 status = nvmet_parse_admin_cmd(req); 1227 1228 if (status) 1229 goto fail; 1230 1231 trace_nvmet_req_init(req, req->cmd); 1232 1233 if (unlikely(!percpu_ref_tryget_live(&sq->ref))) { 1234 status = NVME_SC_INVALID_FIELD | NVME_STATUS_DNR; 1235 goto fail; 1236 } 1237 1238 if (sq->ctrl) 1239 sq->ctrl->reset_tbkas = true; 1240 1241 return true; 1242 1243 fail: 1244 __nvmet_req_complete(req, status); 1245 return false; 1246 } 1247 EXPORT_SYMBOL_GPL(nvmet_req_init); 1248 1249 void nvmet_req_uninit(struct nvmet_req *req) 1250 { 1251 percpu_ref_put(&req->sq->ref); 1252 if (req->pc_ref) 1253 nvmet_pr_put_ns_pc_ref(req->pc_ref); 1254 if (req->ns) 1255 nvmet_put_namespace(req->ns); 1256 } 1257 EXPORT_SYMBOL_GPL(nvmet_req_uninit); 1258 1259 size_t nvmet_req_transfer_len(struct nvmet_req *req) 1260 { 1261 if (likely(req->sq->qid != 0)) 1262 return nvmet_io_cmd_transfer_len(req); 1263 if (unlikely(!req->sq->ctrl)) 1264 return nvmet_connect_cmd_data_len(req); 1265 return nvmet_admin_cmd_data_len(req); 1266 } 1267 EXPORT_SYMBOL_GPL(nvmet_req_transfer_len); 1268 1269 bool nvmet_check_transfer_len(struct nvmet_req *req, size_t len) 1270 { 1271 if (unlikely(len != req->transfer_len)) { 1272 u16 status; 1273 1274 req->error_loc = offsetof(struct nvme_common_command, dptr); 1275 if (req->cmd->common.flags & NVME_CMD_SGL_ALL) 1276 status = NVME_SC_SGL_INVALID_DATA; 1277 else 1278 status = NVME_SC_INVALID_FIELD; 1279 nvmet_req_complete(req, status | NVME_STATUS_DNR); 1280 return false; 1281 } 1282 1283 return true; 1284 } 1285 EXPORT_SYMBOL_GPL(nvmet_check_transfer_len); 1286 1287 bool nvmet_check_data_len_lte(struct nvmet_req *req, size_t data_len) 1288 { 1289 if (unlikely(data_len > req->transfer_len)) { 1290 u16 status; 1291 1292 req->error_loc = offsetof(struct nvme_common_command, dptr); 1293 if (req->cmd->common.flags & NVME_CMD_SGL_ALL) 1294 status = NVME_SC_SGL_INVALID_DATA; 1295 else 1296 status = NVME_SC_INVALID_FIELD; 1297 nvmet_req_complete(req, status | NVME_STATUS_DNR); 1298 return false; 1299 } 1300 1301 return true; 1302 } 1303 1304 static unsigned int nvmet_data_transfer_len(struct nvmet_req *req) 1305 { 1306 return req->transfer_len - req->metadata_len; 1307 } 1308 1309 static int nvmet_req_alloc_p2pmem_sgls(struct pci_dev *p2p_dev, 1310 struct nvmet_req *req) 1311 { 1312 req->sg = pci_p2pmem_alloc_sgl(p2p_dev, &req->sg_cnt, 1313 nvmet_data_transfer_len(req)); 1314 if (!req->sg) 1315 goto out_err; 1316 1317 if (req->metadata_len) { 1318 req->metadata_sg = pci_p2pmem_alloc_sgl(p2p_dev, 1319 &req->metadata_sg_cnt, req->metadata_len); 1320 if (!req->metadata_sg) 1321 goto out_free_sg; 1322 } 1323 1324 req->p2p_dev = p2p_dev; 1325 1326 return 0; 1327 out_free_sg: 1328 pci_p2pmem_free_sgl(req->p2p_dev, req->sg); 1329 out_err: 1330 return -ENOMEM; 1331 } 1332 1333 static struct pci_dev *nvmet_req_find_p2p_dev(struct nvmet_req *req) 1334 { 1335 if (!IS_ENABLED(CONFIG_PCI_P2PDMA) || 1336 !req->sq->ctrl || !req->sq->qid || !req->ns) 1337 return NULL; 1338 return radix_tree_lookup(&req->sq->ctrl->p2p_ns_map, req->ns->nsid); 1339 } 1340 1341 int nvmet_req_alloc_sgls(struct nvmet_req *req) 1342 { 1343 struct pci_dev *p2p_dev = nvmet_req_find_p2p_dev(req); 1344 1345 if (p2p_dev && !nvmet_req_alloc_p2pmem_sgls(p2p_dev, req)) 1346 return 0; 1347 1348 req->sg = sgl_alloc(nvmet_data_transfer_len(req), GFP_KERNEL, 1349 &req->sg_cnt); 1350 if (unlikely(!req->sg)) 1351 goto out; 1352 1353 if (req->metadata_len) { 1354 req->metadata_sg = sgl_alloc(req->metadata_len, GFP_KERNEL, 1355 &req->metadata_sg_cnt); 1356 if (unlikely(!req->metadata_sg)) 1357 goto out_free; 1358 } 1359 1360 return 0; 1361 out_free: 1362 sgl_free(req->sg); 1363 out: 1364 return -ENOMEM; 1365 } 1366 EXPORT_SYMBOL_GPL(nvmet_req_alloc_sgls); 1367 1368 void nvmet_req_free_sgls(struct nvmet_req *req) 1369 { 1370 if (req->p2p_dev) { 1371 pci_p2pmem_free_sgl(req->p2p_dev, req->sg); 1372 if (req->metadata_sg) 1373 pci_p2pmem_free_sgl(req->p2p_dev, req->metadata_sg); 1374 req->p2p_dev = NULL; 1375 } else { 1376 sgl_free(req->sg); 1377 if (req->metadata_sg) 1378 sgl_free(req->metadata_sg); 1379 } 1380 1381 req->sg = NULL; 1382 req->metadata_sg = NULL; 1383 req->sg_cnt = 0; 1384 req->metadata_sg_cnt = 0; 1385 } 1386 EXPORT_SYMBOL_GPL(nvmet_req_free_sgls); 1387 1388 static inline bool nvmet_css_supported(u8 cc_css) 1389 { 1390 switch (cc_css << NVME_CC_CSS_SHIFT) { 1391 case NVME_CC_CSS_NVM: 1392 case NVME_CC_CSS_CSI: 1393 return true; 1394 default: 1395 return false; 1396 } 1397 } 1398 1399 static void nvmet_start_ctrl(struct nvmet_ctrl *ctrl) 1400 { 1401 lockdep_assert_held(&ctrl->lock); 1402 1403 /* 1404 * Only I/O controllers should verify iosqes,iocqes. 1405 * Strictly speaking, the spec says a discovery controller 1406 * should verify iosqes,iocqes are zeroed, however that 1407 * would break backwards compatibility, so don't enforce it. 1408 */ 1409 if (!nvmet_is_disc_subsys(ctrl->subsys) && 1410 (nvmet_cc_iosqes(ctrl->cc) != NVME_NVM_IOSQES || 1411 nvmet_cc_iocqes(ctrl->cc) != NVME_NVM_IOCQES)) { 1412 ctrl->csts = NVME_CSTS_CFS; 1413 return; 1414 } 1415 1416 if (nvmet_cc_mps(ctrl->cc) != 0 || 1417 nvmet_cc_ams(ctrl->cc) != 0 || 1418 !nvmet_css_supported(nvmet_cc_css(ctrl->cc))) { 1419 ctrl->csts = NVME_CSTS_CFS; 1420 return; 1421 } 1422 1423 ctrl->csts = NVME_CSTS_RDY; 1424 1425 /* 1426 * Controllers that are not yet enabled should not really enforce the 1427 * keep alive timeout, but we still want to track a timeout and cleanup 1428 * in case a host died before it enabled the controller. Hence, simply 1429 * reset the keep alive timer when the controller is enabled. 1430 */ 1431 if (ctrl->kato) 1432 mod_delayed_work(nvmet_wq, &ctrl->ka_work, ctrl->kato * HZ); 1433 } 1434 1435 static void nvmet_clear_ctrl(struct nvmet_ctrl *ctrl) 1436 { 1437 lockdep_assert_held(&ctrl->lock); 1438 1439 /* XXX: tear down queues? */ 1440 ctrl->csts &= ~NVME_CSTS_RDY; 1441 ctrl->cc = 0; 1442 } 1443 1444 void nvmet_update_cc(struct nvmet_ctrl *ctrl, u32 new) 1445 { 1446 u32 old; 1447 1448 mutex_lock(&ctrl->lock); 1449 old = ctrl->cc; 1450 ctrl->cc = new; 1451 1452 if (nvmet_cc_en(new) && !nvmet_cc_en(old)) 1453 nvmet_start_ctrl(ctrl); 1454 if (!nvmet_cc_en(new) && nvmet_cc_en(old)) 1455 nvmet_clear_ctrl(ctrl); 1456 if (nvmet_cc_shn(new) && !nvmet_cc_shn(old)) { 1457 nvmet_clear_ctrl(ctrl); 1458 ctrl->csts |= NVME_CSTS_SHST_CMPLT; 1459 } 1460 if (!nvmet_cc_shn(new) && nvmet_cc_shn(old)) 1461 ctrl->csts &= ~NVME_CSTS_SHST_CMPLT; 1462 mutex_unlock(&ctrl->lock); 1463 } 1464 EXPORT_SYMBOL_GPL(nvmet_update_cc); 1465 1466 static void nvmet_init_cap(struct nvmet_ctrl *ctrl) 1467 { 1468 /* command sets supported: NVMe command set: */ 1469 ctrl->cap = (1ULL << 37); 1470 /* Controller supports one or more I/O Command Sets */ 1471 ctrl->cap |= (1ULL << 43); 1472 /* CC.EN timeout in 500msec units: */ 1473 ctrl->cap |= (15ULL << 24); 1474 /* maximum queue entries supported: */ 1475 if (ctrl->ops->get_max_queue_size) 1476 ctrl->cap |= min_t(u16, ctrl->ops->get_max_queue_size(ctrl), 1477 ctrl->port->max_queue_size) - 1; 1478 else 1479 ctrl->cap |= ctrl->port->max_queue_size - 1; 1480 1481 if (nvmet_is_passthru_subsys(ctrl->subsys)) 1482 nvmet_passthrough_override_cap(ctrl); 1483 } 1484 1485 struct nvmet_ctrl *nvmet_ctrl_find_get(const char *subsysnqn, 1486 const char *hostnqn, u16 cntlid, 1487 struct nvmet_req *req) 1488 { 1489 struct nvmet_ctrl *ctrl = NULL; 1490 struct nvmet_subsys *subsys; 1491 1492 subsys = nvmet_find_get_subsys(req->port, subsysnqn); 1493 if (!subsys) { 1494 pr_warn("connect request for invalid subsystem %s!\n", 1495 subsysnqn); 1496 req->cqe->result.u32 = IPO_IATTR_CONNECT_DATA(subsysnqn); 1497 goto out; 1498 } 1499 1500 mutex_lock(&subsys->lock); 1501 list_for_each_entry(ctrl, &subsys->ctrls, subsys_entry) { 1502 if (ctrl->cntlid == cntlid) { 1503 if (strncmp(hostnqn, ctrl->hostnqn, NVMF_NQN_SIZE)) { 1504 pr_warn("hostnqn mismatch.\n"); 1505 continue; 1506 } 1507 if (!kref_get_unless_zero(&ctrl->ref)) 1508 continue; 1509 1510 /* ctrl found */ 1511 goto found; 1512 } 1513 } 1514 1515 ctrl = NULL; /* ctrl not found */ 1516 pr_warn("could not find controller %d for subsys %s / host %s\n", 1517 cntlid, subsysnqn, hostnqn); 1518 req->cqe->result.u32 = IPO_IATTR_CONNECT_DATA(cntlid); 1519 1520 found: 1521 mutex_unlock(&subsys->lock); 1522 nvmet_subsys_put(subsys); 1523 out: 1524 return ctrl; 1525 } 1526 1527 u16 nvmet_check_ctrl_status(struct nvmet_req *req) 1528 { 1529 if (unlikely(!(req->sq->ctrl->cc & NVME_CC_ENABLE))) { 1530 pr_err("got cmd %d while CC.EN == 0 on qid = %d\n", 1531 req->cmd->common.opcode, req->sq->qid); 1532 return NVME_SC_CMD_SEQ_ERROR | NVME_STATUS_DNR; 1533 } 1534 1535 if (unlikely(!(req->sq->ctrl->csts & NVME_CSTS_RDY))) { 1536 pr_err("got cmd %d while CSTS.RDY == 0 on qid = %d\n", 1537 req->cmd->common.opcode, req->sq->qid); 1538 return NVME_SC_CMD_SEQ_ERROR | NVME_STATUS_DNR; 1539 } 1540 1541 if (unlikely(!nvmet_check_auth_status(req))) { 1542 pr_warn("qid %d not authenticated\n", req->sq->qid); 1543 return NVME_SC_AUTH_REQUIRED | NVME_STATUS_DNR; 1544 } 1545 return 0; 1546 } 1547 1548 bool nvmet_host_allowed(struct nvmet_subsys *subsys, const char *hostnqn) 1549 { 1550 struct nvmet_host_link *p; 1551 1552 lockdep_assert_held(&nvmet_config_sem); 1553 1554 if (subsys->allow_any_host) 1555 return true; 1556 1557 if (nvmet_is_disc_subsys(subsys)) /* allow all access to disc subsys */ 1558 return true; 1559 1560 list_for_each_entry(p, &subsys->hosts, entry) { 1561 if (!strcmp(nvmet_host_name(p->host), hostnqn)) 1562 return true; 1563 } 1564 1565 return false; 1566 } 1567 1568 static void nvmet_setup_p2p_ns_map(struct nvmet_ctrl *ctrl, 1569 struct device *p2p_client) 1570 { 1571 struct nvmet_ns *ns; 1572 unsigned long idx; 1573 1574 lockdep_assert_held(&ctrl->subsys->lock); 1575 1576 if (!p2p_client) 1577 return; 1578 1579 ctrl->p2p_client = get_device(p2p_client); 1580 1581 nvmet_for_each_enabled_ns(&ctrl->subsys->namespaces, idx, ns) 1582 nvmet_p2pmem_ns_add_p2p(ctrl, ns); 1583 } 1584 1585 static void nvmet_release_p2p_ns_map(struct nvmet_ctrl *ctrl) 1586 { 1587 struct radix_tree_iter iter; 1588 void __rcu **slot; 1589 1590 lockdep_assert_held(&ctrl->subsys->lock); 1591 1592 radix_tree_for_each_slot(slot, &ctrl->p2p_ns_map, &iter, 0) 1593 pci_dev_put(radix_tree_deref_slot(slot)); 1594 1595 put_device(ctrl->p2p_client); 1596 } 1597 1598 static void nvmet_fatal_error_handler(struct work_struct *work) 1599 { 1600 struct nvmet_ctrl *ctrl = 1601 container_of(work, struct nvmet_ctrl, fatal_err_work); 1602 1603 pr_err("ctrl %d fatal error occurred!\n", ctrl->cntlid); 1604 ctrl->ops->delete_ctrl(ctrl); 1605 } 1606 1607 struct nvmet_ctrl *nvmet_alloc_ctrl(struct nvmet_alloc_ctrl_args *args) 1608 { 1609 struct nvmet_subsys *subsys; 1610 struct nvmet_ctrl *ctrl; 1611 u32 kato = args->kato; 1612 u8 dhchap_status; 1613 int ret; 1614 1615 args->status = NVME_SC_CONNECT_INVALID_PARAM | NVME_STATUS_DNR; 1616 subsys = nvmet_find_get_subsys(args->port, args->subsysnqn); 1617 if (!subsys) { 1618 pr_warn("connect request for invalid subsystem %s!\n", 1619 args->subsysnqn); 1620 args->result = IPO_IATTR_CONNECT_DATA(subsysnqn); 1621 args->error_loc = offsetof(struct nvme_common_command, dptr); 1622 return NULL; 1623 } 1624 1625 down_read(&nvmet_config_sem); 1626 if (!nvmet_host_allowed(subsys, args->hostnqn)) { 1627 pr_info("connect by host %s for subsystem %s not allowed\n", 1628 args->hostnqn, args->subsysnqn); 1629 args->result = IPO_IATTR_CONNECT_DATA(hostnqn); 1630 up_read(&nvmet_config_sem); 1631 args->status = NVME_SC_CONNECT_INVALID_HOST | NVME_STATUS_DNR; 1632 args->error_loc = offsetof(struct nvme_common_command, dptr); 1633 goto out_put_subsystem; 1634 } 1635 up_read(&nvmet_config_sem); 1636 1637 args->status = NVME_SC_INTERNAL; 1638 ctrl = kzalloc_obj(*ctrl); 1639 if (!ctrl) 1640 goto out_put_subsystem; 1641 mutex_init(&ctrl->lock); 1642 1643 ctrl->port = args->port; 1644 ctrl->ops = args->ops; 1645 1646 #ifdef CONFIG_NVME_TARGET_PASSTHRU 1647 /* By default, set loop targets to clear IDS by default */ 1648 if (ctrl->port->disc_addr.trtype == NVMF_TRTYPE_LOOP) 1649 subsys->clear_ids = 1; 1650 #endif 1651 1652 INIT_WORK(&ctrl->async_event_work, nvmet_async_event_work); 1653 INIT_LIST_HEAD(&ctrl->async_events); 1654 INIT_RADIX_TREE(&ctrl->p2p_ns_map, GFP_KERNEL); 1655 INIT_WORK(&ctrl->fatal_err_work, nvmet_fatal_error_handler); 1656 INIT_DELAYED_WORK(&ctrl->ka_work, nvmet_keep_alive_timer); 1657 1658 memcpy(ctrl->hostnqn, args->hostnqn, NVMF_NQN_SIZE); 1659 if (args->hostid) 1660 uuid_copy(&ctrl->hostid, args->hostid); 1661 1662 kref_init(&ctrl->ref); 1663 ctrl->subsys = subsys; 1664 ctrl->pi_support = ctrl->port->pi_enable && ctrl->subsys->pi_support; 1665 nvmet_init_cap(ctrl); 1666 WRITE_ONCE(ctrl->aen_enabled, NVMET_AEN_CFG_OPTIONAL); 1667 1668 ctrl->changed_ns_list = kmalloc_array(NVME_MAX_CHANGED_NAMESPACES, 1669 sizeof(__le32), GFP_KERNEL); 1670 if (!ctrl->changed_ns_list) 1671 goto out_free_ctrl; 1672 1673 /* 1674 * Discovery controllers may use some arbitrary high value 1675 * in order to cleanup stale discovery sessions 1676 */ 1677 if (nvmet_is_disc_subsys(ctrl->subsys) && !kato) 1678 kato = NVMET_DISC_KATO_MS; 1679 1680 /* keep-alive timeout in seconds */ 1681 ctrl->kato = DIV_ROUND_UP(kato, 1000); 1682 1683 ctrl->err_counter = 0; 1684 spin_lock_init(&ctrl->error_lock); 1685 1686 down_read(&nvmet_config_sem); 1687 mutex_lock(&subsys->lock); 1688 1689 ctrl->max_qid = subsys->max_qid; 1690 1691 ctrl->sqs = kzalloc_objs(struct nvmet_sq *, ctrl->max_qid + 1); 1692 if (!ctrl->sqs) 1693 goto out_free_changed_ns_list; 1694 1695 ctrl->cqs = kzalloc_objs(struct nvmet_cq *, ctrl->max_qid + 1); 1696 if (!ctrl->cqs) 1697 goto out_free_sqs; 1698 1699 ret = ida_alloc_range(&cntlid_ida, 1700 subsys->cntlid_min, subsys->cntlid_max, 1701 GFP_KERNEL); 1702 if (ret < 0) { 1703 args->status = NVME_SC_CONNECT_CTRL_BUSY | NVME_STATUS_DNR; 1704 goto out_free_cqs; 1705 } 1706 ctrl->cntlid = ret; 1707 1708 ret = nvmet_ctrl_init_pr(ctrl); 1709 if (ret) 1710 goto init_pr_fail; 1711 list_add_tail(&ctrl->subsys_entry, &subsys->ctrls); 1712 nvmet_setup_p2p_ns_map(ctrl, args->p2p_client); 1713 nvmet_debugfs_ctrl_setup(ctrl); 1714 mutex_unlock(&subsys->lock); 1715 up_read(&nvmet_config_sem); 1716 1717 nvmet_start_keep_alive_timer(ctrl); 1718 1719 dhchap_status = nvmet_setup_auth(ctrl, args->sq, false); 1720 if (dhchap_status) { 1721 pr_err("Failed to setup authentication, dhchap status %u\n", 1722 dhchap_status); 1723 nvmet_ctrl_put(ctrl); 1724 if (dhchap_status == NVME_AUTH_DHCHAP_FAILURE_FAILED) 1725 args->status = 1726 NVME_SC_CONNECT_INVALID_HOST | NVME_STATUS_DNR; 1727 else 1728 args->status = NVME_SC_INTERNAL; 1729 return NULL; 1730 } 1731 1732 args->status = NVME_SC_SUCCESS; 1733 1734 pr_info("Created %s controller %d for subsystem %s for NQN %s%s%s%s.\n", 1735 nvmet_is_disc_subsys(ctrl->subsys) ? "discovery" : "nvm", 1736 ctrl->cntlid, ctrl->subsys->subsysnqn, ctrl->hostnqn, 1737 ctrl->pi_support ? " T10-PI is enabled" : "", 1738 nvmet_has_auth(ctrl, args->sq) ? " with DH-HMAC-CHAP" : "", 1739 nvmet_queue_tls_keyid(args->sq) ? ", TLS" : ""); 1740 1741 return ctrl; 1742 1743 init_pr_fail: 1744 ida_free(&cntlid_ida, ctrl->cntlid); 1745 out_free_cqs: 1746 kfree(ctrl->cqs); 1747 out_free_sqs: 1748 kfree(ctrl->sqs); 1749 out_free_changed_ns_list: 1750 mutex_unlock(&subsys->lock); 1751 up_read(&nvmet_config_sem); 1752 kfree(ctrl->changed_ns_list); 1753 out_free_ctrl: 1754 kfree(ctrl); 1755 out_put_subsystem: 1756 nvmet_subsys_put(subsys); 1757 return NULL; 1758 } 1759 EXPORT_SYMBOL_GPL(nvmet_alloc_ctrl); 1760 1761 static void nvmet_ctrl_free(struct kref *ref) 1762 { 1763 struct nvmet_ctrl *ctrl = container_of(ref, struct nvmet_ctrl, ref); 1764 struct nvmet_subsys *subsys = ctrl->subsys; 1765 1766 mutex_lock(&subsys->lock); 1767 nvmet_ctrl_destroy_pr(ctrl); 1768 nvmet_release_p2p_ns_map(ctrl); 1769 list_del(&ctrl->subsys_entry); 1770 mutex_unlock(&subsys->lock); 1771 1772 nvmet_stop_keep_alive_timer(ctrl); 1773 1774 cancel_work_sync(&ctrl->async_event_work); 1775 cancel_work_sync(&ctrl->fatal_err_work); 1776 1777 nvmet_destroy_auth(ctrl); 1778 1779 nvmet_debugfs_ctrl_free(ctrl); 1780 1781 ida_free(&cntlid_ida, ctrl->cntlid); 1782 1783 nvmet_async_events_free(ctrl); 1784 kfree(ctrl->sqs); 1785 kfree(ctrl->cqs); 1786 kfree(ctrl->changed_ns_list); 1787 kfree(ctrl); 1788 1789 nvmet_subsys_put(subsys); 1790 } 1791 1792 void nvmet_ctrl_put(struct nvmet_ctrl *ctrl) 1793 { 1794 kref_put(&ctrl->ref, nvmet_ctrl_free); 1795 } 1796 EXPORT_SYMBOL_GPL(nvmet_ctrl_put); 1797 1798 void nvmet_ctrl_fatal_error(struct nvmet_ctrl *ctrl) 1799 { 1800 mutex_lock(&ctrl->lock); 1801 if (!(ctrl->csts & NVME_CSTS_CFS)) { 1802 ctrl->csts |= NVME_CSTS_CFS; 1803 queue_work(nvmet_wq, &ctrl->fatal_err_work); 1804 } 1805 mutex_unlock(&ctrl->lock); 1806 } 1807 EXPORT_SYMBOL_GPL(nvmet_ctrl_fatal_error); 1808 1809 ssize_t nvmet_ctrl_host_traddr(struct nvmet_ctrl *ctrl, 1810 char *traddr, size_t traddr_len) 1811 { 1812 if (!ctrl->ops->host_traddr) 1813 return -EOPNOTSUPP; 1814 return ctrl->ops->host_traddr(ctrl, traddr, traddr_len); 1815 } 1816 1817 static struct nvmet_subsys *nvmet_find_get_subsys(struct nvmet_port *port, 1818 const char *subsysnqn) 1819 { 1820 struct nvmet_subsys_link *p; 1821 1822 if (!port) 1823 return NULL; 1824 1825 if (!strcmp(NVME_DISC_SUBSYS_NAME, subsysnqn)) { 1826 if (!kref_get_unless_zero(&nvmet_disc_subsys->ref)) 1827 return NULL; 1828 return nvmet_disc_subsys; 1829 } 1830 1831 down_read(&nvmet_config_sem); 1832 if (!strncmp(nvmet_disc_subsys->subsysnqn, subsysnqn, 1833 NVMF_NQN_SIZE)) { 1834 if (kref_get_unless_zero(&nvmet_disc_subsys->ref)) { 1835 up_read(&nvmet_config_sem); 1836 return nvmet_disc_subsys; 1837 } 1838 } 1839 list_for_each_entry(p, &port->subsystems, entry) { 1840 if (!strncmp(p->subsys->subsysnqn, subsysnqn, 1841 NVMF_NQN_SIZE)) { 1842 if (!kref_get_unless_zero(&p->subsys->ref)) 1843 break; 1844 up_read(&nvmet_config_sem); 1845 return p->subsys; 1846 } 1847 } 1848 up_read(&nvmet_config_sem); 1849 return NULL; 1850 } 1851 1852 struct nvmet_subsys *nvmet_subsys_alloc(const char *subsysnqn, 1853 enum nvme_subsys_type type) 1854 { 1855 struct nvmet_subsys *subsys; 1856 char serial[NVMET_SN_MAX_SIZE / 2]; 1857 int ret; 1858 1859 subsys = kzalloc_obj(*subsys); 1860 if (!subsys) 1861 return ERR_PTR(-ENOMEM); 1862 1863 subsys->ver = NVMET_DEFAULT_VS; 1864 /* generate a random serial number as our controllers are ephemeral: */ 1865 get_random_bytes(&serial, sizeof(serial)); 1866 bin2hex(subsys->serial, &serial, sizeof(serial)); 1867 1868 subsys->model_number = kstrdup(NVMET_DEFAULT_CTRL_MODEL, GFP_KERNEL); 1869 if (!subsys->model_number) { 1870 ret = -ENOMEM; 1871 goto free_subsys; 1872 } 1873 1874 subsys->ieee_oui = 0; 1875 1876 subsys->firmware_rev = kstrndup(UTS_RELEASE, NVMET_FR_MAX_SIZE, GFP_KERNEL); 1877 if (!subsys->firmware_rev) { 1878 ret = -ENOMEM; 1879 goto free_mn; 1880 } 1881 1882 switch (type) { 1883 case NVME_NQN_NVME: 1884 subsys->max_qid = NVMET_NR_QUEUES; 1885 break; 1886 case NVME_NQN_DISC: 1887 case NVME_NQN_CURR: 1888 subsys->max_qid = 0; 1889 break; 1890 default: 1891 pr_err("%s: Unknown Subsystem type - %d\n", __func__, type); 1892 ret = -EINVAL; 1893 goto free_fr; 1894 } 1895 subsys->type = type; 1896 subsys->subsysnqn = kstrndup(subsysnqn, NVMF_NQN_SIZE, 1897 GFP_KERNEL); 1898 if (!subsys->subsysnqn) { 1899 ret = -ENOMEM; 1900 goto free_fr; 1901 } 1902 subsys->cntlid_min = NVME_CNTLID_MIN; 1903 subsys->cntlid_max = NVME_CNTLID_MAX; 1904 kref_init(&subsys->ref); 1905 1906 mutex_init(&subsys->lock); 1907 xa_init(&subsys->namespaces); 1908 INIT_LIST_HEAD(&subsys->ctrls); 1909 INIT_LIST_HEAD(&subsys->hosts); 1910 1911 ret = nvmet_debugfs_subsys_setup(subsys); 1912 if (ret) 1913 goto free_subsysnqn; 1914 1915 return subsys; 1916 1917 free_subsysnqn: 1918 kfree(subsys->subsysnqn); 1919 free_fr: 1920 kfree(subsys->firmware_rev); 1921 free_mn: 1922 kfree(subsys->model_number); 1923 free_subsys: 1924 kfree(subsys); 1925 return ERR_PTR(ret); 1926 } 1927 1928 static void nvmet_subsys_free(struct kref *ref) 1929 { 1930 struct nvmet_subsys *subsys = 1931 container_of(ref, struct nvmet_subsys, ref); 1932 1933 WARN_ON_ONCE(!list_empty(&subsys->ctrls)); 1934 WARN_ON_ONCE(!list_empty(&subsys->hosts)); 1935 WARN_ON_ONCE(!xa_empty(&subsys->namespaces)); 1936 1937 nvmet_debugfs_subsys_free(subsys); 1938 1939 xa_destroy(&subsys->namespaces); 1940 nvmet_passthru_subsys_free(subsys); 1941 1942 kfree(subsys->subsysnqn); 1943 kfree(subsys->model_number); 1944 kfree(subsys->firmware_rev); 1945 kfree(subsys); 1946 } 1947 1948 void nvmet_subsys_del_ctrls(struct nvmet_subsys *subsys) 1949 { 1950 struct nvmet_ctrl *ctrl; 1951 1952 mutex_lock(&subsys->lock); 1953 list_for_each_entry(ctrl, &subsys->ctrls, subsys_entry) 1954 ctrl->ops->delete_ctrl(ctrl); 1955 mutex_unlock(&subsys->lock); 1956 } 1957 1958 void nvmet_subsys_put(struct nvmet_subsys *subsys) 1959 { 1960 kref_put(&subsys->ref, nvmet_subsys_free); 1961 } 1962 1963 static int __init nvmet_init(void) 1964 { 1965 int error = -ENOMEM; 1966 1967 nvmet_ana_group_enabled[NVMET_DEFAULT_ANA_GRPID] = 1; 1968 1969 nvmet_bvec_cache = kmem_cache_create("nvmet-bvec", 1970 NVMET_MAX_MPOOL_BVEC * sizeof(struct bio_vec), 0, 1971 SLAB_HWCACHE_ALIGN, NULL); 1972 if (!nvmet_bvec_cache) 1973 return -ENOMEM; 1974 1975 zbd_wq = alloc_workqueue("nvmet-zbd-wq", WQ_MEM_RECLAIM | WQ_PERCPU, 1976 0); 1977 if (!zbd_wq) 1978 goto out_destroy_bvec_cache; 1979 1980 buffered_io_wq = alloc_workqueue("nvmet-buffered-io-wq", 1981 WQ_MEM_RECLAIM | WQ_PERCPU, 0); 1982 if (!buffered_io_wq) 1983 goto out_free_zbd_work_queue; 1984 1985 nvmet_wq = alloc_workqueue("nvmet-wq", 1986 WQ_MEM_RECLAIM | WQ_UNBOUND | WQ_SYSFS, 0); 1987 if (!nvmet_wq) 1988 goto out_free_buffered_work_queue; 1989 1990 nvmet_aen_wq = alloc_workqueue("nvmet-aen-wq", 1991 WQ_MEM_RECLAIM | WQ_UNBOUND, 0); 1992 if (!nvmet_aen_wq) 1993 goto out_free_nvmet_work_queue; 1994 1995 error = nvmet_init_debugfs(); 1996 if (error) 1997 goto out_free_nvmet_aen_work_queue; 1998 1999 error = nvmet_init_discovery(); 2000 if (error) 2001 goto out_exit_debugfs; 2002 2003 error = nvmet_init_configfs(); 2004 if (error) 2005 goto out_exit_discovery; 2006 2007 return 0; 2008 2009 out_exit_discovery: 2010 nvmet_exit_discovery(); 2011 out_exit_debugfs: 2012 nvmet_exit_debugfs(); 2013 out_free_nvmet_aen_work_queue: 2014 destroy_workqueue(nvmet_aen_wq); 2015 out_free_nvmet_work_queue: 2016 destroy_workqueue(nvmet_wq); 2017 out_free_buffered_work_queue: 2018 destroy_workqueue(buffered_io_wq); 2019 out_free_zbd_work_queue: 2020 destroy_workqueue(zbd_wq); 2021 out_destroy_bvec_cache: 2022 kmem_cache_destroy(nvmet_bvec_cache); 2023 return error; 2024 } 2025 2026 static void __exit nvmet_exit(void) 2027 { 2028 nvmet_exit_configfs(); 2029 nvmet_exit_discovery(); 2030 nvmet_exit_debugfs(); 2031 ida_destroy(&cntlid_ida); 2032 destroy_workqueue(nvmet_aen_wq); 2033 destroy_workqueue(nvmet_wq); 2034 destroy_workqueue(buffered_io_wq); 2035 destroy_workqueue(zbd_wq); 2036 kmem_cache_destroy(nvmet_bvec_cache); 2037 2038 BUILD_BUG_ON(sizeof(struct nvmf_disc_rsp_page_entry) != 1024); 2039 BUILD_BUG_ON(sizeof(struct nvmf_disc_rsp_page_hdr) != 1024); 2040 } 2041 2042 module_init(nvmet_init); 2043 module_exit(nvmet_exit); 2044 2045 MODULE_DESCRIPTION("NVMe target core framework"); 2046 MODULE_LICENSE("GPL v2"); 2047