1 // SPDX-License-Identifier: CDDL-1.0 2 /* 3 * This file and its contents are supplied under the terms of the 4 * Common Development and Distribution License ("CDDL"), version 1.0. 5 * You may only use this file in accordance with the terms of version 6 * 1.0 of the CDDL. 7 * 8 * A full copy of the text of the CDDL should have accompanied this 9 * source. A copy of the CDDL is also available via the Internet at 10 * https://opensource.org/license/CDDL-1.0. 11 */ 12 /* 13 * Copyright (c) 2005, 2010, Oracle and/or its affiliates. All rights reserved. 14 * Copyright (c) 2011, 2024 by Delphix. All rights reserved. 15 * Copyright 2015 Nexenta Systems, Inc. All rights reserved. 16 * Copyright (c) 2014 Spectra Logic Corporation, All rights reserved. 17 * Copyright 2013 Saso Kiselkov. All rights reserved. 18 * Copyright (c) 2017 Datto Inc. 19 * Copyright (c) 2017, Intel Corporation. 20 * Copyright (c) 2019, loli10K <ezomori.nozomu@gmail.com>. All rights reserved. 21 * Copyright (c) 2023, 2024, 2025, Klara, Inc. 22 */ 23 24 #include <sys/zfs_context.h> 25 #include <sys/zfs_chksum.h> 26 #include <sys/spa_impl.h> 27 #include <sys/zio.h> 28 #include <sys/zio_checksum.h> 29 #include <sys/zio_compress.h> 30 #include <sys/dmu.h> 31 #include <sys/dmu_tx.h> 32 #include <sys/zap.h> 33 #include <sys/zil.h> 34 #include <sys/vdev_impl.h> 35 #include <sys/vdev_initialize.h> 36 #include <sys/vdev_trim.h> 37 #include <sys/vdev_file.h> 38 #include <sys/vdev_raidz.h> 39 #include <sys/metaslab.h> 40 #include <sys/uberblock_impl.h> 41 #include <sys/txg.h> 42 #include <sys/avl.h> 43 #include <sys/unique.h> 44 #include <sys/dsl_pool.h> 45 #include <sys/dsl_dir.h> 46 #include <sys/dsl_prop.h> 47 #include <sys/fm/util.h> 48 #include <sys/dsl_scan.h> 49 #include <sys/fs/zfs.h> 50 #include <sys/metaslab_impl.h> 51 #include <sys/arc.h> 52 #include <sys/brt.h> 53 #include <sys/ddt.h> 54 #include <sys/kstat.h> 55 #include "zfs_prop.h" 56 #include <sys/btree.h> 57 #include <sys/zfeature.h> 58 #include <sys/qat.h> 59 #include <sys/zstd/zstd.h> 60 61 /* 62 * SPA locking 63 * 64 * There are three basic locks for managing spa_t structures: 65 * 66 * spa_namespace_lock (global mutex) 67 * 68 * This lock must be acquired to do any of the following: 69 * 70 * - Lookup a spa_t by name 71 * - Add or remove a spa_t from the namespace 72 * - Increase spa_refcount from non-zero 73 * - Check if spa_refcount is zero 74 * - Rename a spa_t 75 * - add/remove/attach/detach devices 76 * - Held for the duration of create/destroy 77 * - Held at the start and end of import and export 78 * 79 * It does not need to handle recursion. A create or destroy may 80 * reference objects (files or zvols) in other pools, but by 81 * definition they must have an existing reference, and will never need 82 * to lookup a spa_t by name. 83 * 84 * spa_refcount (per-spa zfs_refcount_t protected by mutex) 85 * 86 * This reference count keep track of any active users of the spa_t. The 87 * spa_t cannot be destroyed or freed while this is non-zero. Internally, 88 * the refcount is never really 'zero' - opening a pool implicitly keeps 89 * some references in the DMU. Internally we check against spa_minref, but 90 * present the image of a zero/non-zero value to consumers. 91 * 92 * spa_config_lock[] (per-spa array of rwlocks) 93 * 94 * This protects the spa_t from config changes, and must be held in 95 * the following circumstances: 96 * 97 * - RW_READER to perform I/O to the spa 98 * - RW_WRITER to change the vdev config 99 * 100 * The locking order is fairly straightforward: 101 * 102 * spa_namespace_lock -> spa_refcount 103 * 104 * The namespace lock must be acquired to increase the refcount from 0 105 * or to check if it is zero. 106 * 107 * spa_refcount -> spa_config_lock[] 108 * 109 * There must be at least one valid reference on the spa_t to acquire 110 * the config lock. 111 * 112 * spa_namespace_lock -> spa_config_lock[] 113 * 114 * The namespace lock must always be taken before the config lock. 115 * 116 * 117 * The spa_namespace_lock can be acquired directly and is globally visible. 118 * 119 * The namespace is manipulated using the following functions, all of which 120 * require the spa_namespace_lock to be held. 121 * 122 * spa_lookup() Lookup a spa_t by name. 123 * 124 * spa_add() Create a new spa_t in the namespace. 125 * 126 * spa_remove() Remove a spa_t from the namespace. This also 127 * frees up any memory associated with the spa_t. 128 * 129 * spa_next() Returns the next spa_t in the system, or the 130 * first if NULL is passed. 131 * 132 * spa_evict_all() Shutdown and remove all spa_t structures in 133 * the system. 134 * 135 * spa_guid_exists() Determine whether a pool/device guid exists. 136 * 137 * The spa_refcount is manipulated using the following functions: 138 * 139 * spa_open_ref() Adds a reference to the given spa_t. Must be 140 * called with spa_namespace_lock held if the 141 * refcount is currently zero. 142 * 143 * spa_close() Remove a reference from the spa_t. This will 144 * not free the spa_t or remove it from the 145 * namespace. No locking is required. 146 * 147 * spa_refcount_zero() Returns true if the refcount is currently 148 * zero. Must be called with spa_namespace_lock 149 * held. 150 * 151 * The spa_config_lock[] is an array of rwlocks, ordered as follows: 152 * SCL_CONFIG > SCL_STATE > SCL_ALLOC > SCL_ZIO > SCL_FREE > SCL_VDEV. 153 * spa_config_lock[] is manipulated with spa_config_{enter,exit,held}(). 154 * 155 * To read the configuration, it suffices to hold one of these locks as reader. 156 * To modify the configuration, you must hold all locks as writer. To modify 157 * vdev state without altering the vdev tree's topology (e.g. online/offline), 158 * you must hold SCL_STATE and SCL_ZIO as writer. 159 * 160 * We use these distinct config locks to avoid recursive lock entry. 161 * For example, spa_sync() (which holds SCL_CONFIG as reader) induces 162 * block allocations (SCL_ALLOC), which may require reading space maps 163 * from disk (dmu_read() -> zio_read() -> SCL_ZIO). 164 * 165 * The spa config locks cannot be normal rwlocks because we need the 166 * ability to hand off ownership. For example, SCL_ZIO is acquired 167 * by the issuing thread and later released by an interrupt thread. 168 * They do, however, obey the usual write-wanted semantics to prevent 169 * writer (i.e. system administrator) starvation. 170 * 171 * The lock acquisition rules are as follows: 172 * 173 * SCL_CONFIG 174 * Protects changes to the vdev tree topology, such as vdev 175 * add/remove/attach/detach. Protects the dirty config list 176 * (spa_config_dirty_list) and the set of spares and l2arc devices. 177 * 178 * SCL_STATE 179 * Protects changes to pool state and vdev state, such as vdev 180 * online/offline/fault/degrade/clear. Protects the dirty state list 181 * (spa_state_dirty_list) and global pool state (spa_state). 182 * 183 * SCL_ALLOC 184 * Protects changes to metaslab groups and classes. 185 * Held as reader by metaslab_alloc() and metaslab_claim(). 186 * 187 * SCL_ZIO 188 * Held by bp-level zios (those which have no io_vd upon entry) 189 * to prevent changes to the vdev tree. The bp-level zio implicitly 190 * protects all of its vdev child zios, which do not hold SCL_ZIO. 191 * 192 * SCL_FREE 193 * Protects changes to metaslab groups and classes. 194 * Held as reader by metaslab_free(). SCL_FREE is distinct from 195 * SCL_ALLOC, and lower than SCL_ZIO, so that we can safely free 196 * blocks in zio_done() while another i/o that holds either 197 * SCL_ALLOC or SCL_ZIO is waiting for this i/o to complete. 198 * 199 * SCL_VDEV 200 * Held as reader to prevent changes to the vdev tree during trivial 201 * inquiries such as bp_get_dsize(). SCL_VDEV is distinct from the 202 * other locks, and lower than all of them, to ensure that it's safe 203 * to acquire regardless of caller context. 204 * 205 * In addition, the following rules apply: 206 * 207 * (a) spa_props_lock protects pool properties, spa_config and spa_config_list. 208 * The lock ordering is SCL_CONFIG > spa_props_lock. 209 * 210 * (b) I/O operations on leaf vdevs. For any zio operation that takes 211 * an explicit vdev_t argument -- such as zio_ioctl(), zio_read_phys(), 212 * or zio_write_phys() -- the caller must ensure that the config cannot 213 * cannot change in the interim, and that the vdev cannot be reopened. 214 * SCL_STATE as reader suffices for both. 215 * 216 * The vdev configuration is protected by spa_vdev_enter() / spa_vdev_exit(). 217 * 218 * spa_vdev_enter() Acquire the namespace lock and the config lock 219 * for writing. 220 * 221 * spa_vdev_exit() Release the config lock, wait for all I/O 222 * to complete, sync the updated configs to the 223 * cache, and release the namespace lock. 224 * 225 * vdev state is protected by spa_vdev_state_enter() / spa_vdev_state_exit(). 226 * Like spa_vdev_enter/exit, these are convenience wrappers -- the actual 227 * locking is, always, based on spa_namespace_lock and spa_config_lock[]. 228 */ 229 230 static avl_tree_t spa_namespace_avl; 231 static kmutex_t spa_namespace_lock; 232 static kcondvar_t spa_namespace_cv; 233 234 static const int spa_max_replication_override = SPA_DVAS_PER_BP; 235 236 static kmutex_t spa_spare_lock; 237 static avl_tree_t spa_spare_avl; 238 static kmutex_t spa_l2cache_lock; 239 static avl_tree_t spa_l2cache_avl; 240 241 spa_mode_t spa_mode_global = SPA_MODE_UNINIT; 242 243 #ifdef ZFS_DEBUG 244 /* 245 * Everything except dprintf, set_error, indirect_remap, and raidz_reconstruct 246 * is on by default in debug builds. 247 */ 248 int zfs_flags = ~(ZFS_DEBUG_DPRINTF | ZFS_DEBUG_SET_ERROR | 249 ZFS_DEBUG_INDIRECT_REMAP | ZFS_DEBUG_RAIDZ_RECONSTRUCT); 250 #else 251 int zfs_flags = 0; 252 #endif 253 254 /* 255 * zfs_recover can be set to nonzero to attempt to recover from 256 * otherwise-fatal errors, typically caused by on-disk corruption. When 257 * set, calls to zfs_panic_recover() will turn into warning messages. 258 * This should only be used as a last resort, as it typically results 259 * in leaked space, or worse. 260 */ 261 int zfs_recover = B_FALSE; 262 263 /* 264 * If destroy encounters an EIO while reading metadata (e.g. indirect 265 * blocks), space referenced by the missing metadata can not be freed. 266 * Normally this causes the background destroy to become "stalled", as 267 * it is unable to make forward progress. While in this stalled state, 268 * all remaining space to free from the error-encountering filesystem is 269 * "temporarily leaked". Set this flag to cause it to ignore the EIO, 270 * permanently leak the space from indirect blocks that can not be read, 271 * and continue to free everything else that it can. 272 * 273 * The default, "stalling" behavior is useful if the storage partially 274 * fails (i.e. some but not all i/os fail), and then later recovers. In 275 * this case, we will be able to continue pool operations while it is 276 * partially failed, and when it recovers, we can continue to free the 277 * space, with no leaks. However, note that this case is actually 278 * fairly rare. 279 * 280 * Typically pools either (a) fail completely (but perhaps temporarily, 281 * e.g. a top-level vdev going offline), or (b) have localized, 282 * permanent errors (e.g. disk returns the wrong data due to bit flip or 283 * firmware bug). In case (a), this setting does not matter because the 284 * pool will be suspended and the sync thread will not be able to make 285 * forward progress regardless. In case (b), because the error is 286 * permanent, the best we can do is leak the minimum amount of space, 287 * which is what setting this flag will do. Therefore, it is reasonable 288 * for this flag to normally be set, but we chose the more conservative 289 * approach of not setting it, so that there is no possibility of 290 * leaking space in the "partial temporary" failure case. 291 */ 292 int zfs_free_leak_on_eio = B_FALSE; 293 294 /* 295 * Expiration time in milliseconds. This value has two meanings. First it is 296 * used to determine when the spa_deadman() logic should fire. By default the 297 * spa_deadman() will fire if spa_sync() has not completed in 600 seconds. 298 * Secondly, the value determines if an I/O is considered "hung". Any I/O that 299 * has not completed in zfs_deadman_synctime_ms is considered "hung" resulting 300 * in one of three behaviors controlled by zfs_deadman_failmode. 301 */ 302 uint64_t zfs_deadman_synctime_ms = 600000UL; /* 10 min. */ 303 304 /* 305 * This value controls the maximum amount of time zio_wait() will block for an 306 * outstanding IO. By default this is 300 seconds at which point the "hung" 307 * behavior will be applied as described for zfs_deadman_synctime_ms. 308 */ 309 uint64_t zfs_deadman_ziotime_ms = 300000UL; /* 5 min. */ 310 311 /* 312 * Check time in milliseconds. This defines the frequency at which we check 313 * for hung I/O. 314 */ 315 uint64_t zfs_deadman_checktime_ms = 60000UL; /* 1 min. */ 316 317 /* 318 * By default the deadman is enabled. 319 */ 320 int zfs_deadman_enabled = B_TRUE; 321 322 /* 323 * Controls the behavior of the deadman when it detects a "hung" I/O. 324 * Valid values are zfs_deadman_failmode=<wait|continue|panic>. 325 * 326 * wait - Wait for the "hung" I/O (default) 327 * continue - Attempt to recover from a "hung" I/O 328 * panic - Panic the system 329 */ 330 const char *zfs_deadman_failmode = "wait"; 331 332 /* 333 * The worst case is single-sector max-parity RAID-Z blocks, in which 334 * case the space requirement is exactly (VDEV_RAIDZ_MAXPARITY + 1) 335 * times the size; so just assume that. Add to this the fact that 336 * we can have up to 3 DVAs per bp, and one more factor of 2 because 337 * the block may be dittoed with up to 3 DVAs by ddt_sync(). All together, 338 * the worst case is: 339 * (VDEV_RAIDZ_MAXPARITY + 1) * SPA_DVAS_PER_BP * 2 == 24 340 */ 341 uint_t spa_asize_inflation = 24; 342 343 /* 344 * Normally, we don't allow the last 3.2% (1/(2^spa_slop_shift)) of space in 345 * the pool to be consumed (bounded by spa_max_slop). This ensures that we 346 * don't run the pool completely out of space, due to unaccounted changes (e.g. 347 * to the MOS). It also limits the worst-case time to allocate space. If we 348 * have less than this amount of free space, most ZPL operations (e.g. write, 349 * create) will return ENOSPC. The ZIL metaslabs (spa_embedded_log_class) are 350 * also part of this 3.2% of space which can't be consumed by normal writes; 351 * the slop space "proper" (spa_get_slop_space()) is decreased by the embedded 352 * log space. 353 * 354 * Certain operations (e.g. file removal, most administrative actions) can 355 * use half the slop space. They will only return ENOSPC if less than half 356 * the slop space is free. Typically, once the pool has less than the slop 357 * space free, the user will use these operations to free up space in the pool. 358 * These are the operations that call dsl_pool_adjustedsize() with the netfree 359 * argument set to TRUE. 360 * 361 * Operations that are almost guaranteed to free up space in the absence of 362 * a pool checkpoint can use up to three quarters of the slop space 363 * (e.g zfs destroy). 364 * 365 * A very restricted set of operations are always permitted, regardless of 366 * the amount of free space. These are the operations that call 367 * dsl_sync_task(ZFS_SPACE_CHECK_NONE). If these operations result in a net 368 * increase in the amount of space used, it is possible to run the pool 369 * completely out of space, causing it to be permanently read-only. 370 * 371 * Note that on very small pools, the slop space will be larger than 372 * 3.2%, in an effort to have it be at least spa_min_slop (128MB), 373 * but we never allow it to be more than half the pool size. 374 * 375 * Further, on very large pools, the slop space will be smaller than 376 * 3.2%, to avoid reserving much more space than we actually need; bounded 377 * by spa_max_slop (128GB). 378 * 379 * See also the comments in zfs_space_check_t. 380 */ 381 uint_t spa_slop_shift = 5; 382 static const uint64_t spa_min_slop = 128ULL * 1024 * 1024; 383 static const uint64_t spa_max_slop = 128ULL * 1024 * 1024 * 1024; 384 385 /* 386 * Number of allocators to use, per spa instance 387 */ 388 static int spa_num_allocators = 4; 389 static int spa_cpus_per_allocator = 4; 390 391 /* 392 * Spa active allocator. 393 * Valid values are zfs_active_allocator=<dynamic|cursor|new-dynamic>. 394 */ 395 const char *zfs_active_allocator = "dynamic"; 396 397 void 398 spa_load_failed(spa_t *spa, const char *fmt, ...) 399 { 400 va_list adx; 401 char buf[256]; 402 403 va_start(adx, fmt); 404 (void) vsnprintf(buf, sizeof (buf), fmt, adx); 405 va_end(adx); 406 407 zfs_dbgmsg("spa_load(%s, config %s): FAILED: %s", spa_load_name(spa), 408 spa->spa_trust_config ? "trusted" : "untrusted", buf); 409 } 410 411 void 412 spa_load_note(spa_t *spa, const char *fmt, ...) 413 { 414 va_list adx; 415 char buf[256]; 416 417 va_start(adx, fmt); 418 (void) vsnprintf(buf, sizeof (buf), fmt, adx); 419 va_end(adx); 420 421 zfs_dbgmsg("spa_load(%s, config %s): %s", spa_load_name(spa), 422 spa->spa_trust_config ? "trusted" : "untrusted", buf); 423 424 spa_import_progress_set_notes_nolog(spa, "%s", buf); 425 } 426 427 /* 428 * By default dedup and user data indirects land in the special class 429 */ 430 static int zfs_ddt_data_is_special = B_TRUE; 431 static int zfs_user_indirect_is_special = B_TRUE; 432 433 /* 434 * The percentage of special class final space reserved for metadata only. 435 * Once we allocate 100 - zfs_special_class_metadata_reserve_pct we only 436 * let metadata into the class. 437 */ 438 static uint_t zfs_special_class_metadata_reserve_pct = 25; 439 440 /* 441 * ========================================================================== 442 * SPA config locking 443 * ========================================================================== 444 */ 445 static void 446 spa_config_lock_init(spa_t *spa) 447 { 448 for (int i = 0; i < SCL_LOCKS; i++) { 449 spa_config_lock_t *scl = &spa->spa_config_lock[i]; 450 mutex_init(&scl->scl_lock, NULL, MUTEX_DEFAULT, NULL); 451 cv_init(&scl->scl_cv, NULL, CV_DEFAULT, NULL); 452 scl->scl_writer = NULL; 453 scl->scl_write_wanted = 0; 454 scl->scl_count = 0; 455 } 456 } 457 458 static void 459 spa_config_lock_destroy(spa_t *spa) 460 { 461 for (int i = 0; i < SCL_LOCKS; i++) { 462 spa_config_lock_t *scl = &spa->spa_config_lock[i]; 463 mutex_destroy(&scl->scl_lock); 464 cv_destroy(&scl->scl_cv); 465 ASSERT0P(scl->scl_writer); 466 ASSERT0(scl->scl_write_wanted); 467 ASSERT0(scl->scl_count); 468 } 469 } 470 471 /* A writer wants or holds the lock. Mutated only under scl_lock. */ 472 #define SCL_COUNT_WRITER 0x80000000U 473 #define SCL_COUNT_REFS(c) ((c) & ~SCL_COUNT_WRITER) 474 475 /* 476 * Drop one reader reference. The result tells us both that it was the last 477 * one and that a writer is waiting for it. 478 */ 479 static void 480 spa_config_exit_read(spa_config_lock_t *scl) 481 { 482 uint32_t count; 483 484 ASSERT3U(SCL_COUNT_REFS(scl->scl_count), >, 0); 485 486 count = atomic_dec_32_nv(&scl->scl_count); 487 if (count != SCL_COUNT_WRITER) 488 return; 489 mutex_enter(&scl->scl_lock); 490 cv_broadcast(&scl->scl_cv); 491 mutex_exit(&scl->scl_lock); 492 } 493 494 int 495 spa_config_tryenter(spa_t *spa, int locks, const void *tag, krw_t rw) 496 { 497 for (int i = 0; i < SCL_LOCKS; i++) { 498 spa_config_lock_t *scl = &spa->spa_config_lock[i]; 499 if (!(locks & (1 << i))) 500 continue; 501 if (rw == RW_READER) { 502 if (atomic_inc_32_nv(&scl->scl_count) & 503 SCL_COUNT_WRITER) { 504 spa_config_exit_read(scl); 505 spa_config_exit(spa, locks & ((1 << i) - 1), 506 tag); 507 return (0); 508 } 509 } else { 510 mutex_enter(&scl->scl_lock); 511 ASSERT(scl->scl_writer != curthread); 512 /* Take the reference together with the bit or fail. */ 513 if (atomic_cas_32(&scl->scl_count, 0, 514 SCL_COUNT_WRITER | 1) != 0) { 515 mutex_exit(&scl->scl_lock); 516 spa_config_exit(spa, locks & ((1 << i) - 1), 517 tag); 518 return (0); 519 } 520 scl->scl_write_wanted++; 521 scl->scl_writer = curthread; 522 mutex_exit(&scl->scl_lock); 523 } 524 } 525 membar_consumer(); 526 return (1); 527 } 528 529 static void 530 spa_config_enter_impl(spa_t *spa, int locks, const void *tag, krw_t rw, 531 int priority_flag) 532 { 533 (void) tag; 534 int wlocks_held = 0; 535 536 ASSERT3U(SCL_LOCKS, <, sizeof (wlocks_held) * NBBY); 537 538 for (int i = 0; i < SCL_LOCKS; i++) { 539 spa_config_lock_t *scl = &spa->spa_config_lock[i]; 540 if (scl->scl_writer == curthread) 541 wlocks_held |= (1 << i); 542 if (!(locks & (1 << i))) 543 continue; 544 545 /* 546 * Priority readers have to tell a waiting writer from a 547 * holding one, which one word can not express, so they take 548 * the lock. They are rare. 549 */ 550 if (rw == RW_READER && !priority_flag) { 551 if ((atomic_inc_32_nv(&scl->scl_count) & 552 SCL_COUNT_WRITER) == 0) 553 continue; 554 spa_config_exit_read(scl); 555 } 556 557 mutex_enter(&scl->scl_lock); 558 if (rw == RW_READER) { 559 while (scl->scl_writer || 560 (!priority_flag && scl->scl_write_wanted)) { 561 cv_wait(&scl->scl_cv, &scl->scl_lock); 562 } 563 } else { 564 ASSERT(scl->scl_writer != curthread); 565 scl->scl_write_wanted++; 566 atomic_or_32(&scl->scl_count, SCL_COUNT_WRITER); 567 while (SCL_COUNT_REFS(scl->scl_count) != 0) 568 cv_wait(&scl->scl_cv, &scl->scl_lock); 569 scl->scl_writer = curthread; 570 } 571 atomic_inc_32(&scl->scl_count); 572 mutex_exit(&scl->scl_lock); 573 } 574 575 /* Pair with the membar_producer() in spa_config_exit(). */ 576 membar_consumer(); 577 ASSERT3U(wlocks_held, <=, locks); 578 } 579 580 void 581 spa_config_enter(spa_t *spa, int locks, const void *tag, krw_t rw) 582 { 583 spa_config_enter_impl(spa, locks, tag, rw, 0); 584 } 585 586 /* 587 * The spa_config_enter_priority() allows the mmp thread to cut in front of 588 * outstanding write lock requests. This is needed since the mmp updates are 589 * time sensitive and failure to service them promptly will result in a 590 * suspended pool. This pool suspension has been seen in practice when there is 591 * a single disk in a pool that is responding slowly and presumably about to 592 * fail. 593 */ 594 595 void 596 spa_config_enter_priority(spa_t *spa, int locks, const void *tag, krw_t rw) 597 { 598 spa_config_enter_impl(spa, locks, tag, rw, 1); 599 } 600 601 void 602 spa_config_exit(spa_t *spa, int locks, const void *tag) 603 { 604 (void) tag; 605 for (int i = SCL_LOCKS - 1; i >= 0; i--) { 606 spa_config_lock_t *scl = &spa->spa_config_lock[i]; 607 if (!(locks & (1 << i))) 608 continue; 609 if (scl->scl_writer != curthread) { 610 spa_config_exit_read(scl); 611 continue; 612 } 613 614 /* 615 * A reader mid-backoff may hold a reference, so the count is 616 * not ours alone. Clearing SCL_COUNT_WRITER opens the reader 617 * fast path, so it goes last and after a barrier. 618 */ 619 mutex_enter(&scl->scl_lock); 620 ASSERT3U(SCL_COUNT_REFS(scl->scl_count), >, 0); 621 scl->scl_writer = NULL; 622 atomic_dec_32(&scl->scl_count); 623 if (--scl->scl_write_wanted == 0) { 624 membar_producer(); 625 atomic_and_32(&scl->scl_count, ~SCL_COUNT_WRITER); 626 } 627 cv_broadcast(&scl->scl_cv); 628 mutex_exit(&scl->scl_lock); 629 } 630 } 631 632 int 633 spa_config_held(spa_t *spa, int locks, krw_t rw) 634 { 635 int locks_held = 0; 636 637 for (int i = 0; i < SCL_LOCKS; i++) { 638 spa_config_lock_t *scl = &spa->spa_config_lock[i]; 639 if (!(locks & (1 << i))) 640 continue; 641 if ((rw == RW_READER && SCL_COUNT_REFS(scl->scl_count) != 0) || 642 (rw == RW_WRITER && scl->scl_writer == curthread)) 643 locks_held |= 1 << i; 644 } 645 646 return (locks_held); 647 } 648 649 /* 650 * ========================================================================== 651 * SPA namespace functions 652 * ========================================================================== 653 */ 654 655 void 656 spa_namespace_enter(const void *tag) 657 { 658 (void) tag; 659 ASSERT(!MUTEX_HELD(&spa_namespace_lock)); 660 mutex_enter(&spa_namespace_lock); 661 } 662 663 boolean_t 664 spa_namespace_tryenter(const void *tag) 665 { 666 (void) tag; 667 ASSERT(!MUTEX_HELD(&spa_namespace_lock)); 668 return (mutex_tryenter(&spa_namespace_lock)); 669 } 670 671 int 672 spa_namespace_enter_interruptible(const void *tag) 673 { 674 (void) tag; 675 ASSERT(!MUTEX_HELD(&spa_namespace_lock)); 676 return (mutex_enter_interruptible(&spa_namespace_lock)); 677 } 678 679 void 680 spa_namespace_exit(const void *tag) 681 { 682 (void) tag; 683 ASSERT(MUTEX_HELD(&spa_namespace_lock)); 684 mutex_exit(&spa_namespace_lock); 685 } 686 687 boolean_t 688 spa_namespace_held(void) 689 { 690 return (MUTEX_HELD(&spa_namespace_lock)); 691 } 692 693 void 694 spa_namespace_wait(void) 695 { 696 ASSERT(MUTEX_HELD(&spa_namespace_lock)); 697 cv_wait(&spa_namespace_cv, &spa_namespace_lock); 698 } 699 700 void 701 spa_namespace_broadcast(void) 702 { 703 ASSERT(MUTEX_HELD(&spa_namespace_lock)); 704 cv_broadcast(&spa_namespace_cv); 705 } 706 707 /* 708 * Lookup the named spa_t in the AVL tree. The spa_namespace_lock must be held. 709 * Returns NULL if no matching spa_t is found. 710 */ 711 spa_t * 712 spa_lookup(const char *name) 713 { 714 static spa_t search; /* spa_t is large; don't allocate on stack */ 715 spa_t *spa; 716 avl_index_t where; 717 char *cp; 718 719 ASSERT(spa_namespace_held()); 720 721 retry: 722 (void) strlcpy(search.spa_name, name, sizeof (search.spa_name)); 723 724 /* 725 * If it's a full dataset name, figure out the pool name and 726 * just use that. 727 */ 728 cp = strpbrk(search.spa_name, "/@#"); 729 if (cp != NULL) 730 *cp = '\0'; 731 732 spa = avl_find(&spa_namespace_avl, &search, &where); 733 if (spa == NULL) 734 return (NULL); 735 736 /* 737 * Avoid racing with import/export, which don't hold the namespace 738 * lock for their entire duration. 739 */ 740 if ((spa->spa_load_thread != NULL && 741 spa->spa_load_thread != curthread) || 742 (spa->spa_export_thread != NULL && 743 spa->spa_export_thread != curthread)) { 744 spa_namespace_wait(); 745 goto retry; 746 } 747 748 return (spa); 749 } 750 751 /* 752 * Fires when spa_sync has not completed within zfs_deadman_synctime_ms. 753 * If the zfs_deadman_enabled flag is set then it inspects all vdev queues 754 * looking for potentially hung I/Os. 755 */ 756 void 757 spa_deadman(void *arg) 758 { 759 spa_t *spa = arg; 760 761 /* Disable the deadman if the pool is suspended. */ 762 if (spa_suspended(spa)) 763 return; 764 765 zfs_dbgmsg("slow spa_sync: started %llu seconds ago, calls %llu", 766 (getlrtime() - spa->spa_sync_starttime) / NANOSEC, 767 (u_longlong_t)++spa->spa_deadman_calls); 768 if (zfs_deadman_enabled) 769 vdev_deadman(spa->spa_root_vdev, FTAG); 770 771 spa->spa_deadman_tqid = taskq_dispatch_delay(system_delay_taskq, 772 spa_deadman, spa, TQ_SLEEP, ddi_get_lbolt() + 773 MSEC_TO_TICK(zfs_deadman_checktime_ms)); 774 } 775 776 static int 777 spa_log_sm_sort_by_txg(const void *va, const void *vb) 778 { 779 const spa_log_sm_t *a = va; 780 const spa_log_sm_t *b = vb; 781 782 return (TREE_CMP(a->sls_txg, b->sls_txg)); 783 } 784 785 /* 786 * Create an uninitialized spa_t with the given name. Requires 787 * spa_namespace_lock. The caller must ensure that the spa_t doesn't already 788 * exist by calling spa_lookup() first. 789 */ 790 spa_t * 791 spa_add(const char *name, nvlist_t *config, const char *altroot) 792 { 793 spa_t *spa; 794 spa_config_dirent_t *dp; 795 796 ASSERT(spa_namespace_held()); 797 798 spa = kmem_zalloc(sizeof (spa_t), KM_SLEEP); 799 800 mutex_init(&spa->spa_async_lock, NULL, MUTEX_DEFAULT, NULL); 801 mutex_init(&spa->spa_errlist_lock, NULL, MUTEX_DEFAULT, NULL); 802 mutex_init(&spa->spa_errlog_lock, NULL, MUTEX_DEFAULT, NULL); 803 mutex_init(&spa->spa_evicting_os_lock, NULL, MUTEX_DEFAULT, NULL); 804 mutex_init(&spa->spa_history_lock, NULL, MUTEX_DEFAULT, NULL); 805 mutex_init(&spa->spa_proc_lock, NULL, MUTEX_DEFAULT, NULL); 806 mutex_init(&spa->spa_props_lock, NULL, MUTEX_DEFAULT, NULL); 807 mutex_init(&spa->spa_cksum_tmpls_lock, NULL, MUTEX_DEFAULT, NULL); 808 mutex_init(&spa->spa_scrub_lock, NULL, MUTEX_DEFAULT, NULL); 809 mutex_init(&spa->spa_suspend_lock, NULL, MUTEX_DEFAULT, NULL); 810 mutex_init(&spa->spa_vdev_top_lock, NULL, MUTEX_DEFAULT, NULL); 811 mutex_init(&spa->spa_feat_stats_lock, NULL, MUTEX_DEFAULT, NULL); 812 mutex_init(&spa->spa_flushed_ms_lock, NULL, MUTEX_DEFAULT, NULL); 813 mutex_init(&spa->spa_activities_lock, NULL, MUTEX_DEFAULT, NULL); 814 mutex_init(&spa->spa_txg_log_time_lock, NULL, MUTEX_DEFAULT, NULL); 815 mutex_init(&spa->spa_condense_stats_lock, NULL, MUTEX_DEFAULT, NULL); 816 817 cv_init(&spa->spa_async_cv, NULL, CV_DEFAULT, NULL); 818 cv_init(&spa->spa_evicting_os_cv, NULL, CV_DEFAULT, NULL); 819 cv_init(&spa->spa_proc_cv, NULL, CV_DEFAULT, NULL); 820 cv_init(&spa->spa_scrub_io_cv, NULL, CV_DEFAULT, NULL); 821 cv_init(&spa->spa_suspend_cv, NULL, CV_DEFAULT, NULL); 822 cv_init(&spa->spa_activities_cv, NULL, CV_DEFAULT, NULL); 823 cv_init(&spa->spa_waiters_cv, NULL, CV_DEFAULT, NULL); 824 825 for (int t = 0; t < TXG_SIZE; t++) 826 bplist_create(&spa->spa_free_bplist[t]); 827 828 (void) strlcpy(spa->spa_name, name, sizeof (spa->spa_name)); 829 spa->spa_state = POOL_STATE_UNINITIALIZED; 830 spa->spa_freeze_txg = UINT64_MAX; 831 spa->spa_final_txg = UINT64_MAX; 832 spa->spa_load_max_txg = UINT64_MAX; 833 spa->spa_proc = &p0; 834 spa->spa_proc_state = SPA_PROC_NONE; 835 spa->spa_trust_config = B_TRUE; 836 spa->spa_hostid = zone_get_hostid(NULL); 837 838 spa->spa_deadman_synctime = MSEC2NSEC(zfs_deadman_synctime_ms); 839 spa->spa_deadman_ziotime = MSEC2NSEC(zfs_deadman_ziotime_ms); 840 spa_set_deadman_failmode(spa, zfs_deadman_failmode); 841 spa_set_allocator(spa, zfs_active_allocator); 842 843 zfs_refcount_create(&spa->spa_refcount); 844 spa_config_lock_init(spa); 845 spa_stats_init(spa); 846 847 ASSERT(spa_namespace_held()); 848 avl_add(&spa_namespace_avl, spa); 849 850 /* 851 * Set the alternate root, if there is one. 852 */ 853 if (altroot) 854 spa->spa_root = spa_strdup(altroot); 855 856 /* Do not allow more allocators than fraction of CPUs. */ 857 spa->spa_alloc_count = MAX(MIN(spa_num_allocators, 858 boot_ncpus / MAX(spa_cpus_per_allocator, 1)), 1); 859 860 if (spa->spa_alloc_count > 1) { 861 spa->spa_allocs_use = kmem_zalloc(offsetof(spa_allocs_use_t, 862 sau_inuse[spa->spa_alloc_count]), KM_SLEEP); 863 mutex_init(&spa->spa_allocs_use->sau_lock, NULL, MUTEX_DEFAULT, 864 NULL); 865 } 866 867 avl_create(&spa->spa_metaslabs_by_flushed, metaslab_sort_by_flushed, 868 sizeof (metaslab_t), offsetof(metaslab_t, ms_spa_txg_node)); 869 avl_create(&spa->spa_sm_logs_by_txg, spa_log_sm_sort_by_txg, 870 sizeof (spa_log_sm_t), offsetof(spa_log_sm_t, sls_node)); 871 list_create(&spa->spa_log_summary, sizeof (log_summary_entry_t), 872 offsetof(log_summary_entry_t, lse_node)); 873 874 /* 875 * Every pool starts with the default cachefile 876 */ 877 list_create(&spa->spa_config_list, sizeof (spa_config_dirent_t), 878 offsetof(spa_config_dirent_t, scd_link)); 879 880 dp = kmem_zalloc(sizeof (spa_config_dirent_t), KM_SLEEP); 881 dp->scd_path = altroot ? NULL : spa_strdup(spa_config_path); 882 list_insert_head(&spa->spa_config_list, dp); 883 884 VERIFY0(nvlist_alloc(&spa->spa_load_info, NV_UNIQUE_NAME, KM_SLEEP)); 885 886 if (config != NULL) { 887 nvlist_t *features; 888 889 if (nvlist_lookup_nvlist(config, ZPOOL_CONFIG_FEATURES_FOR_READ, 890 &features) == 0) { 891 VERIFY0(nvlist_dup(features, 892 &spa->spa_label_features, 0)); 893 } 894 895 VERIFY0(nvlist_dup(config, &spa->spa_config, 0)); 896 } 897 898 if (spa->spa_label_features == NULL) { 899 VERIFY0(nvlist_alloc(&spa->spa_label_features, NV_UNIQUE_NAME, 900 KM_SLEEP)); 901 } 902 903 spa->spa_min_ashift = INT_MAX; 904 spa->spa_max_ashift = 0; 905 spa->spa_min_alloc = INT_MAX; 906 spa->spa_max_alloc = 0; 907 spa->spa_gcd_alloc = INT_MAX; 908 909 /* Reset cached value */ 910 spa->spa_dedup_dspace = ~0ULL; 911 912 /* 913 * As a pool is being created, treat all features as disabled by 914 * setting SPA_FEATURE_DISABLED for all entries in the feature 915 * refcount cache. 916 */ 917 for (int i = 0; i < SPA_FEATURES; i++) { 918 spa->spa_feat_refcount_cache[i] = SPA_FEATURE_DISABLED; 919 } 920 921 list_create(&spa->spa_leaf_list, sizeof (vdev_t), 922 offsetof(vdev_t, vdev_leaf_node)); 923 924 return (spa); 925 } 926 927 /* 928 * Removes a spa_t from the namespace, freeing up any memory used. Requires 929 * spa_namespace_lock. This is called only after the spa_t has been closed and 930 * deactivated. 931 */ 932 void 933 spa_remove(spa_t *spa) 934 { 935 spa_config_dirent_t *dp; 936 937 ASSERT(spa_namespace_held()); 938 ASSERT(spa_state(spa) == POOL_STATE_UNINITIALIZED); 939 ASSERT3U(zfs_refcount_count(&spa->spa_refcount), ==, 0); 940 ASSERT0(spa->spa_waiters); 941 942 nvlist_free(spa->spa_config_splitting); 943 944 avl_remove(&spa_namespace_avl, spa); 945 946 if (spa->spa_root) 947 spa_strfree(spa->spa_root); 948 949 if (spa->spa_load_name) 950 spa_strfree(spa->spa_load_name); 951 952 while ((dp = list_remove_head(&spa->spa_config_list)) != NULL) { 953 if (dp->scd_path != NULL) 954 spa_strfree(dp->scd_path); 955 kmem_free(dp, sizeof (spa_config_dirent_t)); 956 } 957 958 if (spa->spa_alloc_count > 1) { 959 mutex_destroy(&spa->spa_allocs_use->sau_lock); 960 kmem_free(spa->spa_allocs_use, offsetof(spa_allocs_use_t, 961 sau_inuse[spa->spa_alloc_count])); 962 } 963 964 avl_destroy(&spa->spa_metaslabs_by_flushed); 965 avl_destroy(&spa->spa_sm_logs_by_txg); 966 list_destroy(&spa->spa_log_summary); 967 list_destroy(&spa->spa_config_list); 968 list_destroy(&spa->spa_leaf_list); 969 970 nvlist_free(spa->spa_label_features); 971 nvlist_free(spa->spa_load_info); 972 nvlist_free(spa->spa_feat_stats); 973 spa_config_set(spa, NULL); 974 975 zfs_refcount_destroy(&spa->spa_refcount); 976 977 spa_stats_destroy(spa); 978 spa_config_lock_destroy(spa); 979 980 for (int t = 0; t < TXG_SIZE; t++) 981 bplist_destroy(&spa->spa_free_bplist[t]); 982 983 zio_checksum_templates_free(spa); 984 985 cv_destroy(&spa->spa_async_cv); 986 cv_destroy(&spa->spa_evicting_os_cv); 987 cv_destroy(&spa->spa_proc_cv); 988 cv_destroy(&spa->spa_scrub_io_cv); 989 cv_destroy(&spa->spa_suspend_cv); 990 cv_destroy(&spa->spa_activities_cv); 991 cv_destroy(&spa->spa_waiters_cv); 992 993 mutex_destroy(&spa->spa_flushed_ms_lock); 994 mutex_destroy(&spa->spa_async_lock); 995 mutex_destroy(&spa->spa_errlist_lock); 996 mutex_destroy(&spa->spa_errlog_lock); 997 mutex_destroy(&spa->spa_evicting_os_lock); 998 mutex_destroy(&spa->spa_history_lock); 999 mutex_destroy(&spa->spa_proc_lock); 1000 mutex_destroy(&spa->spa_props_lock); 1001 mutex_destroy(&spa->spa_cksum_tmpls_lock); 1002 mutex_destroy(&spa->spa_scrub_lock); 1003 mutex_destroy(&spa->spa_suspend_lock); 1004 mutex_destroy(&spa->spa_vdev_top_lock); 1005 mutex_destroy(&spa->spa_feat_stats_lock); 1006 mutex_destroy(&spa->spa_activities_lock); 1007 mutex_destroy(&spa->spa_txg_log_time_lock); 1008 mutex_destroy(&spa->spa_condense_stats_lock); 1009 1010 kmem_free(spa, sizeof (spa_t)); 1011 } 1012 1013 /* 1014 * Given a pool, return the next pool in the namespace, or NULL if there is 1015 * none. If 'prev' is NULL, return the first pool. 1016 */ 1017 spa_t * 1018 spa_next(spa_t *prev) 1019 { 1020 ASSERT(spa_namespace_held()); 1021 1022 if (prev) 1023 return (AVL_NEXT(&spa_namespace_avl, prev)); 1024 else 1025 return (avl_first(&spa_namespace_avl)); 1026 } 1027 1028 /* 1029 * ========================================================================== 1030 * SPA refcount functions 1031 * ========================================================================== 1032 */ 1033 1034 /* 1035 * Add a reference to the given spa_t. Must have at least one reference, or 1036 * have the namespace lock held. 1037 */ 1038 void 1039 spa_open_ref(spa_t *spa, const void *tag) 1040 { 1041 ASSERT(zfs_refcount_count(&spa->spa_refcount) >= spa->spa_minref || 1042 spa_namespace_held() || 1043 spa->spa_load_thread == curthread); 1044 (void) zfs_refcount_add(&spa->spa_refcount, tag); 1045 } 1046 1047 /* 1048 * Remove a reference to the given spa_t. Must have at least one reference, or 1049 * have the namespace lock held or be part of a pool import/export. 1050 */ 1051 void 1052 spa_close(spa_t *spa, const void *tag) 1053 { 1054 ASSERT(zfs_refcount_count(&spa->spa_refcount) > spa->spa_minref || 1055 spa_namespace_held() || 1056 spa->spa_load_thread == curthread || 1057 spa->spa_export_thread == curthread); 1058 (void) zfs_refcount_remove(&spa->spa_refcount, tag); 1059 } 1060 1061 /* 1062 * Remove a reference to the given spa_t held by a dsl dir that is 1063 * being asynchronously released. Async releases occur from a taskq 1064 * performing eviction of dsl datasets and dirs. The namespace lock 1065 * isn't held and the hold by the object being evicted may contribute to 1066 * spa_minref (e.g. dataset or directory released during pool export), 1067 * so the asserts in spa_close() do not apply. 1068 */ 1069 void 1070 spa_async_close(spa_t *spa, const void *tag) 1071 { 1072 (void) zfs_refcount_remove(&spa->spa_refcount, tag); 1073 } 1074 1075 /* 1076 * Check to see if the spa refcount is zero. Must be called with 1077 * spa_namespace_lock held or be the spa export thread. We really 1078 * compare against spa_minref, which is the number of references 1079 * acquired when opening a pool 1080 */ 1081 boolean_t 1082 spa_refcount_zero(spa_t *spa) 1083 { 1084 ASSERT(spa_namespace_held() || 1085 spa->spa_export_thread == curthread); 1086 1087 return (zfs_refcount_count(&spa->spa_refcount) == spa->spa_minref); 1088 } 1089 1090 /* 1091 * ========================================================================== 1092 * SPA spare and l2cache tracking 1093 * ========================================================================== 1094 */ 1095 1096 /* 1097 * Hot spares and cache devices are tracked using the same code below, 1098 * for 'auxiliary' devices. 1099 */ 1100 1101 typedef struct spa_aux { 1102 uint64_t aux_guid; 1103 uint64_t aux_pool; 1104 avl_node_t aux_avl; 1105 int aux_count; 1106 } spa_aux_t; 1107 1108 static inline int 1109 spa_aux_compare(const void *a, const void *b) 1110 { 1111 const spa_aux_t *sa = (const spa_aux_t *)a; 1112 const spa_aux_t *sb = (const spa_aux_t *)b; 1113 1114 return (TREE_CMP(sa->aux_guid, sb->aux_guid)); 1115 } 1116 1117 static void 1118 spa_aux_add(vdev_t *vd, avl_tree_t *avl) 1119 { 1120 avl_index_t where; 1121 spa_aux_t search; 1122 spa_aux_t *aux; 1123 1124 search.aux_guid = vd->vdev_guid; 1125 if ((aux = avl_find(avl, &search, &where)) != NULL) { 1126 aux->aux_count++; 1127 } else { 1128 aux = kmem_zalloc(sizeof (spa_aux_t), KM_SLEEP); 1129 aux->aux_guid = vd->vdev_guid; 1130 aux->aux_count = 1; 1131 avl_insert(avl, aux, where); 1132 } 1133 } 1134 1135 static void 1136 spa_aux_remove(vdev_t *vd, avl_tree_t *avl) 1137 { 1138 spa_aux_t search; 1139 spa_aux_t *aux; 1140 avl_index_t where; 1141 1142 search.aux_guid = vd->vdev_guid; 1143 aux = avl_find(avl, &search, &where); 1144 1145 ASSERT(aux != NULL); 1146 1147 if (--aux->aux_count == 0) { 1148 avl_remove(avl, aux); 1149 kmem_free(aux, sizeof (spa_aux_t)); 1150 } else if (aux->aux_pool == spa_guid(vd->vdev_spa)) { 1151 aux->aux_pool = 0ULL; 1152 } 1153 } 1154 1155 static boolean_t 1156 spa_aux_exists(uint64_t guid, uint64_t *pool, int *refcnt, avl_tree_t *avl) 1157 { 1158 spa_aux_t search, *found; 1159 1160 search.aux_guid = guid; 1161 found = avl_find(avl, &search, NULL); 1162 1163 if (pool) { 1164 if (found) 1165 *pool = found->aux_pool; 1166 else 1167 *pool = 0ULL; 1168 } 1169 1170 if (refcnt) { 1171 if (found) 1172 *refcnt = found->aux_count; 1173 else 1174 *refcnt = 0; 1175 } 1176 1177 return (found != NULL); 1178 } 1179 1180 static void 1181 spa_aux_activate(vdev_t *vd, avl_tree_t *avl) 1182 { 1183 spa_aux_t search, *found; 1184 avl_index_t where; 1185 1186 search.aux_guid = vd->vdev_guid; 1187 found = avl_find(avl, &search, &where); 1188 ASSERT(found != NULL); 1189 ASSERT(found->aux_pool == 0ULL); 1190 1191 found->aux_pool = spa_guid(vd->vdev_spa); 1192 } 1193 1194 /* 1195 * Spares are tracked globally due to the following constraints: 1196 * 1197 * - A spare may be part of multiple pools. 1198 * - A spare may be added to a pool even if it's actively in use within 1199 * another pool. 1200 * - A spare in use in any pool can only be the source of a replacement if 1201 * the target is a spare in the same pool. 1202 * 1203 * We keep track of all spares on the system through the use of a reference 1204 * counted AVL tree. When a vdev is added as a spare, or used as a replacement 1205 * spare, then we bump the reference count in the AVL tree. In addition, we set 1206 * the 'vdev_isspare' member to indicate that the device is a spare (active or 1207 * inactive). When a spare is made active (used to replace a device in the 1208 * pool), we also keep track of which pool its been made a part of. 1209 * 1210 * The 'spa_spare_lock' protects the AVL tree. These functions are normally 1211 * called under the spa_namespace lock as part of vdev reconfiguration. The 1212 * separate spare lock exists for the status query path, which does not need to 1213 * be completely consistent with respect to other vdev configuration changes. 1214 */ 1215 1216 static int 1217 spa_spare_compare(const void *a, const void *b) 1218 { 1219 return (spa_aux_compare(a, b)); 1220 } 1221 1222 void 1223 spa_spare_add(vdev_t *vd) 1224 { 1225 mutex_enter(&spa_spare_lock); 1226 ASSERT(!vd->vdev_isspare); 1227 spa_aux_add(vd, &spa_spare_avl); 1228 vd->vdev_isspare = B_TRUE; 1229 mutex_exit(&spa_spare_lock); 1230 } 1231 1232 void 1233 spa_spare_remove(vdev_t *vd) 1234 { 1235 mutex_enter(&spa_spare_lock); 1236 ASSERT(vd->vdev_isspare); 1237 spa_aux_remove(vd, &spa_spare_avl); 1238 vd->vdev_isspare = B_FALSE; 1239 mutex_exit(&spa_spare_lock); 1240 } 1241 1242 boolean_t 1243 spa_spare_exists(uint64_t guid, uint64_t *pool, int *refcnt) 1244 { 1245 boolean_t found; 1246 1247 mutex_enter(&spa_spare_lock); 1248 found = spa_aux_exists(guid, pool, refcnt, &spa_spare_avl); 1249 mutex_exit(&spa_spare_lock); 1250 1251 return (found); 1252 } 1253 1254 void 1255 spa_spare_activate(vdev_t *vd) 1256 { 1257 mutex_enter(&spa_spare_lock); 1258 ASSERT(vd->vdev_isspare); 1259 spa_aux_activate(vd, &spa_spare_avl); 1260 mutex_exit(&spa_spare_lock); 1261 } 1262 1263 /* 1264 * Level 2 ARC devices are tracked globally for the same reasons as spares. 1265 * Cache devices currently only support one pool per cache device, and so 1266 * for these devices the aux reference count is currently unused beyond 1. 1267 */ 1268 1269 static int 1270 spa_l2cache_compare(const void *a, const void *b) 1271 { 1272 return (spa_aux_compare(a, b)); 1273 } 1274 1275 void 1276 spa_l2cache_add(vdev_t *vd) 1277 { 1278 mutex_enter(&spa_l2cache_lock); 1279 ASSERT(!vd->vdev_isl2cache); 1280 spa_aux_add(vd, &spa_l2cache_avl); 1281 vd->vdev_isl2cache = B_TRUE; 1282 mutex_exit(&spa_l2cache_lock); 1283 } 1284 1285 void 1286 spa_l2cache_remove(vdev_t *vd) 1287 { 1288 mutex_enter(&spa_l2cache_lock); 1289 ASSERT(vd->vdev_isl2cache); 1290 spa_aux_remove(vd, &spa_l2cache_avl); 1291 vd->vdev_isl2cache = B_FALSE; 1292 mutex_exit(&spa_l2cache_lock); 1293 } 1294 1295 boolean_t 1296 spa_l2cache_exists(uint64_t guid, uint64_t *pool) 1297 { 1298 boolean_t found; 1299 1300 mutex_enter(&spa_l2cache_lock); 1301 found = spa_aux_exists(guid, pool, NULL, &spa_l2cache_avl); 1302 mutex_exit(&spa_l2cache_lock); 1303 1304 return (found); 1305 } 1306 1307 void 1308 spa_l2cache_activate(vdev_t *vd) 1309 { 1310 mutex_enter(&spa_l2cache_lock); 1311 ASSERT(vd->vdev_isl2cache); 1312 spa_aux_activate(vd, &spa_l2cache_avl); 1313 mutex_exit(&spa_l2cache_lock); 1314 } 1315 1316 /* 1317 * ========================================================================== 1318 * SPA vdev locking 1319 * ========================================================================== 1320 */ 1321 1322 /* 1323 * Lock the given spa_t for the purpose of adding or removing a vdev. 1324 * Grabs the global spa_namespace_lock plus the spa config lock for writing. 1325 * It returns the next transaction group for the spa_t. 1326 */ 1327 uint64_t 1328 spa_vdev_enter(spa_t *spa) 1329 { 1330 mutex_enter(&spa->spa_vdev_top_lock); 1331 spa_namespace_enter(FTAG); 1332 1333 ASSERT0P(spa->spa_export_thread); 1334 1335 vdev_autotrim_stop_all(spa); 1336 1337 return (spa_vdev_config_enter(spa)); 1338 } 1339 1340 /* 1341 * The same as spa_vdev_enter() above but additionally takes the guid of 1342 * the vdev being detached. When there is a rebuild in process it will be 1343 * suspended while the vdev tree is modified then resumed by spa_vdev_exit(). 1344 * The rebuild is canceled if only a single child remains after the detach. 1345 */ 1346 uint64_t 1347 spa_vdev_detach_enter(spa_t *spa, uint64_t guid) 1348 { 1349 mutex_enter(&spa->spa_vdev_top_lock); 1350 spa_namespace_enter(FTAG); 1351 1352 ASSERT0P(spa->spa_export_thread); 1353 1354 vdev_autotrim_stop_all(spa); 1355 1356 if (guid != 0) { 1357 vdev_t *vd = spa_lookup_by_guid(spa, guid, B_FALSE); 1358 if (vd) { 1359 vdev_rebuild_stop_wait(vd->vdev_top); 1360 } 1361 } 1362 1363 return (spa_vdev_config_enter(spa)); 1364 } 1365 1366 /* 1367 * Internal implementation for spa_vdev_enter(). Used when a vdev 1368 * operation requires multiple syncs (i.e. removing a device) while 1369 * keeping the spa_namespace_lock held. 1370 */ 1371 uint64_t 1372 spa_vdev_config_enter(spa_t *spa) 1373 { 1374 ASSERT(spa_namespace_held()); 1375 1376 spa_config_enter(spa, SCL_ALL, spa, RW_WRITER); 1377 1378 return (spa_last_synced_txg(spa) + 1); 1379 } 1380 1381 /* 1382 * Used in combination with spa_vdev_config_enter() to allow the syncing 1383 * of multiple transactions without releasing the spa_namespace_lock. 1384 */ 1385 void 1386 spa_vdev_config_exit(spa_t *spa, vdev_t *vd, uint64_t txg, int error, 1387 const char *tag) 1388 { 1389 ASSERT(spa_namespace_held()); 1390 1391 int config_changed = B_FALSE; 1392 1393 ASSERT(txg > spa_last_synced_txg(spa)); 1394 1395 spa->spa_pending_vdev = NULL; 1396 1397 /* 1398 * Reassess the DTLs. 1399 */ 1400 vdev_dtl_reassess(spa->spa_root_vdev, 0, 0, B_FALSE, B_FALSE); 1401 1402 if (error == 0 && !list_is_empty(&spa->spa_config_dirty_list)) { 1403 config_changed = B_TRUE; 1404 spa->spa_config_generation++; 1405 } 1406 1407 /* 1408 * Verify the metaslab classes. 1409 */ 1410 metaslab_class_validate(spa_normal_class(spa)); 1411 metaslab_class_validate(spa_log_class(spa)); 1412 metaslab_class_validate(spa_embedded_log_class(spa)); 1413 metaslab_class_validate(spa_special_class(spa)); 1414 metaslab_class_validate(spa_special_embedded_log_class(spa)); 1415 metaslab_class_validate(spa_dedup_class(spa)); 1416 1417 spa_config_exit(spa, SCL_ALL, spa); 1418 1419 /* 1420 * Panic the system if the specified tag requires it. This 1421 * is useful for ensuring that configurations are updated 1422 * transactionally. 1423 */ 1424 if (zio_injection_enabled) 1425 zio_handle_panic_injection(spa, tag, 0); 1426 1427 /* 1428 * Note: this txg_wait_synced() is important because it ensures 1429 * that there won't be more than one config change per txg. 1430 * This allows us to use the txg as the generation number. 1431 */ 1432 if (error == 0) 1433 txg_wait_synced(spa->spa_dsl_pool, txg); 1434 1435 if (vd != NULL) { 1436 ASSERT(!vd->vdev_detached || vd->vdev_dtl_sm == NULL); 1437 if (vd->vdev_ops->vdev_op_leaf) { 1438 mutex_enter(&vd->vdev_initialize_lock); 1439 vdev_initialize_stop(vd, VDEV_INITIALIZE_CANCELED, 1440 NULL); 1441 mutex_exit(&vd->vdev_initialize_lock); 1442 1443 mutex_enter(&vd->vdev_trim_lock); 1444 vdev_trim_stop(vd, VDEV_TRIM_CANCELED, NULL); 1445 mutex_exit(&vd->vdev_trim_lock); 1446 } 1447 1448 /* 1449 * The vdev may be both a leaf and top-level device. 1450 */ 1451 vdev_autotrim_stop_wait(vd); 1452 1453 spa_config_enter(spa, SCL_STATE_ALL, spa, RW_WRITER); 1454 vdev_free(vd); 1455 spa_config_exit(spa, SCL_STATE_ALL, spa); 1456 } 1457 1458 /* 1459 * If the config changed, update the config cache. 1460 */ 1461 if (config_changed) 1462 spa_write_cachefile(spa, B_FALSE, B_TRUE, B_TRUE); 1463 } 1464 1465 /* 1466 * Unlock the spa_t after adding or removing a vdev. Besides undoing the 1467 * locking of spa_vdev_enter(), we also want make sure the transactions have 1468 * synced to disk, and then update the global configuration cache with the new 1469 * information. 1470 */ 1471 int 1472 spa_vdev_exit(spa_t *spa, vdev_t *vd, uint64_t txg, int error) 1473 { 1474 vdev_autotrim_restart(spa); 1475 vdev_rebuild_restart(spa); 1476 1477 spa_vdev_config_exit(spa, vd, txg, error, FTAG); 1478 spa_namespace_exit(FTAG); 1479 mutex_exit(&spa->spa_vdev_top_lock); 1480 1481 return (error); 1482 } 1483 1484 /* 1485 * Lock the given spa_t for the purpose of changing vdev state. 1486 */ 1487 void 1488 spa_vdev_state_enter(spa_t *spa, int oplocks) 1489 { 1490 int locks = SCL_STATE_ALL | oplocks; 1491 1492 /* 1493 * Root pools may need to read of the underlying devfs filesystem 1494 * when opening up a vdev. Unfortunately if we're holding the 1495 * SCL_ZIO lock it will result in a deadlock when we try to issue 1496 * the read from the root filesystem. Instead we "prefetch" 1497 * the associated vnodes that we need prior to opening the 1498 * underlying devices and cache them so that we can prevent 1499 * any I/O when we are doing the actual open. 1500 */ 1501 if (spa_is_root(spa)) { 1502 int low = locks & ~(SCL_ZIO - 1); 1503 int high = locks & ~low; 1504 1505 spa_config_enter(spa, high, spa, RW_WRITER); 1506 vdev_hold(spa->spa_root_vdev); 1507 spa_config_enter(spa, low, spa, RW_WRITER); 1508 } else { 1509 spa_config_enter(spa, locks, spa, RW_WRITER); 1510 } 1511 spa->spa_vdev_locks = locks; 1512 } 1513 1514 int 1515 spa_vdev_state_exit(spa_t *spa, vdev_t *vd, int error) 1516 { 1517 boolean_t config_changed = B_FALSE; 1518 vdev_t *vdev_top; 1519 1520 if (vd == NULL || vd == spa->spa_root_vdev) { 1521 vdev_top = spa->spa_root_vdev; 1522 } else { 1523 vdev_top = vd->vdev_top; 1524 } 1525 1526 if (vd != NULL || error == 0) 1527 vdev_dtl_reassess(vdev_top, 0, 0, B_FALSE, B_FALSE); 1528 1529 if (vd != NULL) { 1530 if (vd != spa->spa_root_vdev) 1531 vdev_state_dirty(vdev_top); 1532 1533 config_changed = B_TRUE; 1534 spa->spa_config_generation++; 1535 } 1536 1537 if (spa_is_root(spa)) 1538 vdev_rele(spa->spa_root_vdev); 1539 1540 ASSERT3U(spa->spa_vdev_locks, >=, SCL_STATE_ALL); 1541 spa_config_exit(spa, spa->spa_vdev_locks, spa); 1542 1543 /* 1544 * If anything changed, wait for it to sync. This ensures that, 1545 * from the system administrator's perspective, zpool(8) commands 1546 * are synchronous. This is important for things like zpool offline: 1547 * when the command completes, you expect no further I/O from ZFS. 1548 */ 1549 if (vd != NULL) 1550 txg_wait_synced(spa->spa_dsl_pool, 0); 1551 1552 /* 1553 * If the config changed, update the config cache. 1554 */ 1555 if (config_changed) { 1556 spa_namespace_enter(FTAG); 1557 spa_write_cachefile(spa, B_FALSE, B_TRUE, B_FALSE); 1558 spa_namespace_exit(FTAG); 1559 } 1560 1561 return (error); 1562 } 1563 1564 /* 1565 * ========================================================================== 1566 * Miscellaneous functions 1567 * ========================================================================== 1568 */ 1569 1570 void 1571 spa_activate_mos_feature(spa_t *spa, const char *feature, dmu_tx_t *tx) 1572 { 1573 if (!nvlist_exists(spa->spa_label_features, feature)) { 1574 fnvlist_add_boolean(spa->spa_label_features, feature); 1575 /* 1576 * When we are creating the pool (tx_txg==TXG_INITIAL), we can't 1577 * dirty the vdev config because lock SCL_CONFIG is not held. 1578 * Thankfully, in this case we don't need to dirty the config 1579 * because it will be written out anyway when we finish 1580 * creating the pool. 1581 */ 1582 if (tx->tx_txg != TXG_INITIAL) 1583 vdev_config_dirty(spa->spa_root_vdev); 1584 } 1585 } 1586 1587 void 1588 spa_deactivate_mos_feature(spa_t *spa, const char *feature) 1589 { 1590 if (nvlist_remove_all(spa->spa_label_features, feature) == 0) 1591 vdev_config_dirty(spa->spa_root_vdev); 1592 } 1593 1594 /* 1595 * Return the spa_t associated with given pool_guid, if it exists. If 1596 * device_guid is non-zero, determine whether the pool exists *and* contains 1597 * a device with the specified device_guid. 1598 */ 1599 spa_t * 1600 spa_by_guid(uint64_t pool_guid, uint64_t device_guid) 1601 { 1602 spa_t *spa; 1603 avl_tree_t *t = &spa_namespace_avl; 1604 1605 ASSERT(spa_namespace_held()); 1606 1607 for (spa = avl_first(t); spa != NULL; spa = AVL_NEXT(t, spa)) { 1608 if (spa->spa_state == POOL_STATE_UNINITIALIZED) 1609 continue; 1610 if (spa->spa_root_vdev == NULL) 1611 continue; 1612 if (spa_guid(spa) == pool_guid) { 1613 if (device_guid == 0) 1614 break; 1615 1616 if (vdev_lookup_by_guid(spa->spa_root_vdev, 1617 device_guid) != NULL) 1618 break; 1619 1620 /* 1621 * Check any devices we may be in the process of adding. 1622 */ 1623 if (spa->spa_pending_vdev) { 1624 if (vdev_lookup_by_guid(spa->spa_pending_vdev, 1625 device_guid) != NULL) 1626 break; 1627 } 1628 } 1629 } 1630 1631 return (spa); 1632 } 1633 1634 /* 1635 * Determine whether a pool with the given pool_guid exists. 1636 */ 1637 boolean_t 1638 spa_guid_exists(uint64_t pool_guid, uint64_t device_guid) 1639 { 1640 return (spa_by_guid(pool_guid, device_guid) != NULL); 1641 } 1642 1643 char * 1644 spa_strdup(const char *s) 1645 { 1646 size_t len; 1647 char *new; 1648 1649 len = strlen(s); 1650 new = kmem_alloc(len + 1, KM_SLEEP); 1651 memcpy(new, s, len + 1); 1652 1653 return (new); 1654 } 1655 1656 void 1657 spa_strfree(char *s) 1658 { 1659 kmem_free(s, strlen(s) + 1); 1660 } 1661 1662 uint64_t 1663 spa_generate_guid(spa_t *spa) 1664 { 1665 uint64_t guid; 1666 1667 if (spa != NULL) { 1668 do { 1669 (void) random_get_pseudo_bytes((void *)&guid, 1670 sizeof (guid)); 1671 } while (guid == 0 || spa_guid_exists(spa_guid(spa), guid)); 1672 } else { 1673 do { 1674 (void) random_get_pseudo_bytes((void *)&guid, 1675 sizeof (guid)); 1676 } while (guid == 0 || spa_guid_exists(guid, 0)); 1677 } 1678 1679 return (guid); 1680 } 1681 1682 static boolean_t 1683 spa_load_guid_exists(uint64_t guid) 1684 { 1685 avl_tree_t *t = &spa_namespace_avl; 1686 1687 ASSERT(spa_namespace_held()); 1688 1689 for (spa_t *spa = avl_first(t); spa != NULL; spa = AVL_NEXT(t, spa)) { 1690 if (spa_load_guid(spa) == guid) 1691 return (B_TRUE); 1692 } 1693 1694 return (arc_async_flush_guid_inuse(guid)); 1695 } 1696 1697 uint64_t 1698 spa_generate_load_guid(void) 1699 { 1700 uint64_t guid; 1701 1702 do { 1703 (void) random_get_pseudo_bytes((void *)&guid, 1704 sizeof (guid)); 1705 } while (guid == 0 || spa_load_guid_exists(guid)); 1706 1707 return (guid); 1708 } 1709 1710 void 1711 snprintf_blkptr(char *buf, size_t buflen, const blkptr_t *bp) 1712 { 1713 char type[256]; 1714 const char *checksum = NULL; 1715 const char *compress = NULL; 1716 1717 if (bp != NULL) { 1718 if (BP_GET_TYPE(bp) & DMU_OT_NEWTYPE) { 1719 dmu_object_byteswap_t bswap = 1720 DMU_OT_BYTESWAP(BP_GET_TYPE(bp)); 1721 (void) snprintf(type, sizeof (type), "bswap %s %s", 1722 DMU_OT_IS_METADATA(BP_GET_TYPE(bp)) ? 1723 "metadata" : "data", 1724 dmu_ot_byteswap[bswap].ob_name); 1725 } else { 1726 (void) strlcpy(type, dmu_ot[BP_GET_TYPE(bp)].ot_name, 1727 sizeof (type)); 1728 } 1729 if (!BP_IS_EMBEDDED(bp)) { 1730 checksum = 1731 zio_checksum_table[BP_GET_CHECKSUM(bp)].ci_name; 1732 } 1733 compress = zio_compress_table[BP_GET_COMPRESS(bp)].ci_name; 1734 } 1735 1736 SNPRINTF_BLKPTR(kmem_scnprintf, ' ', buf, buflen, bp, type, checksum, 1737 compress); 1738 } 1739 1740 void 1741 spa_freeze(spa_t *spa) 1742 { 1743 uint64_t freeze_txg = 0; 1744 1745 spa_config_enter(spa, SCL_ALL, FTAG, RW_WRITER); 1746 if (spa->spa_freeze_txg == UINT64_MAX) { 1747 freeze_txg = spa_last_synced_txg(spa) + TXG_SIZE; 1748 spa->spa_freeze_txg = freeze_txg; 1749 } 1750 spa_config_exit(spa, SCL_ALL, FTAG); 1751 if (freeze_txg != 0) 1752 txg_wait_synced(spa_get_dsl(spa), freeze_txg); 1753 } 1754 1755 void 1756 zfs_panic_recover(const char *fmt, ...) 1757 { 1758 va_list adx; 1759 1760 va_start(adx, fmt); 1761 vcmn_err(zfs_recover ? CE_WARN : CE_PANIC, fmt, adx); 1762 va_end(adx); 1763 } 1764 1765 /* 1766 * This is a stripped-down version of strtoull, suitable only for converting 1767 * lowercase hexadecimal numbers that don't overflow. 1768 */ 1769 uint64_t 1770 zfs_strtonum(const char *str, char **nptr) 1771 { 1772 uint64_t val = 0; 1773 char c; 1774 int digit; 1775 1776 while ((c = *str) != '\0') { 1777 if (c >= '0' && c <= '9') 1778 digit = c - '0'; 1779 else if (c >= 'a' && c <= 'f') 1780 digit = 10 + c - 'a'; 1781 else 1782 break; 1783 1784 val *= 16; 1785 val += digit; 1786 1787 str++; 1788 } 1789 1790 if (nptr) 1791 *nptr = (char *)str; 1792 1793 return (val); 1794 } 1795 1796 void 1797 spa_activate_allocation_classes(spa_t *spa, dmu_tx_t *tx) 1798 { 1799 /* 1800 * We bump the feature refcount for each special vdev added to the pool 1801 */ 1802 ASSERT(spa_feature_is_enabled(spa, SPA_FEATURE_ALLOCATION_CLASSES)); 1803 spa_feature_incr(spa, SPA_FEATURE_ALLOCATION_CLASSES, tx); 1804 } 1805 1806 /* 1807 * ========================================================================== 1808 * Accessor functions 1809 * ========================================================================== 1810 */ 1811 1812 boolean_t 1813 spa_shutting_down(spa_t *spa) 1814 { 1815 return (spa->spa_async_suspended); 1816 } 1817 1818 dsl_pool_t * 1819 spa_get_dsl(spa_t *spa) 1820 { 1821 return (spa->spa_dsl_pool); 1822 } 1823 1824 boolean_t 1825 spa_is_initializing(spa_t *spa) 1826 { 1827 return (spa->spa_is_initializing); 1828 } 1829 1830 boolean_t 1831 spa_indirect_vdevs_loaded(spa_t *spa) 1832 { 1833 return (spa->spa_indirect_vdevs_loaded); 1834 } 1835 1836 blkptr_t * 1837 spa_get_rootblkptr(spa_t *spa) 1838 { 1839 return (&spa->spa_ubsync.ub_rootbp); 1840 } 1841 1842 void 1843 spa_set_rootblkptr(spa_t *spa, const blkptr_t *bp) 1844 { 1845 spa->spa_uberblock.ub_rootbp = *bp; 1846 } 1847 1848 void 1849 spa_altroot(spa_t *spa, char *buf, size_t buflen) 1850 { 1851 if (spa->spa_root == NULL) 1852 buf[0] = '\0'; 1853 else 1854 (void) strlcpy(buf, spa->spa_root, buflen); 1855 } 1856 1857 uint32_t 1858 spa_sync_pass(spa_t *spa) 1859 { 1860 return (spa->spa_sync_pass); 1861 } 1862 1863 char * 1864 spa_name(spa_t *spa) 1865 { 1866 return (spa->spa_name); 1867 } 1868 1869 char * 1870 spa_load_name(spa_t *spa) 1871 { 1872 /* 1873 * During spa_tryimport() the pool name includes a unique prefix. 1874 * Returns the original name which can be used for log messages. 1875 */ 1876 if (spa->spa_load_name) 1877 return (spa->spa_load_name); 1878 1879 return (spa->spa_name); 1880 } 1881 1882 uint64_t 1883 spa_guid(spa_t *spa) 1884 { 1885 dsl_pool_t *dp = spa_get_dsl(spa); 1886 uint64_t guid; 1887 1888 /* 1889 * If we fail to parse the config during spa_load(), we can go through 1890 * the error path (which posts an ereport) and end up here with no root 1891 * vdev. We stash the original pool guid in 'spa_config_guid' to handle 1892 * this case. 1893 */ 1894 if (spa->spa_root_vdev == NULL) 1895 return (spa->spa_config_guid); 1896 1897 guid = spa->spa_last_synced_guid != 0 ? 1898 spa->spa_last_synced_guid : spa->spa_root_vdev->vdev_guid; 1899 1900 /* 1901 * Return the most recently synced out guid unless we're 1902 * in syncing context. 1903 */ 1904 if (dp && dsl_pool_sync_context(dp)) 1905 return (spa->spa_root_vdev->vdev_guid); 1906 else 1907 return (guid); 1908 } 1909 1910 uint64_t 1911 spa_load_guid(spa_t *spa) 1912 { 1913 /* 1914 * This is a GUID that exists solely as a reference for the 1915 * purposes of the arc. It is generated at load time, and 1916 * is never written to persistent storage. 1917 */ 1918 return (spa->spa_load_guid); 1919 } 1920 1921 uint64_t 1922 spa_last_synced_txg(spa_t *spa) 1923 { 1924 return (spa->spa_ubsync.ub_txg); 1925 } 1926 1927 uint64_t 1928 spa_first_txg(spa_t *spa) 1929 { 1930 return (spa->spa_first_txg); 1931 } 1932 1933 uint64_t 1934 spa_syncing_txg(spa_t *spa) 1935 { 1936 return (spa->spa_syncing_txg); 1937 } 1938 1939 uint64_t 1940 spa_open_txg(spa_t *spa) 1941 { 1942 return (spa->spa_dsl_pool->dp_tx.tx_open_txg); 1943 } 1944 1945 /* 1946 * Return the last txg where data can be dirtied. The final txgs 1947 * will be used to just clear out any deferred frees that remain. 1948 */ 1949 uint64_t 1950 spa_final_dirty_txg(spa_t *spa) 1951 { 1952 return (spa->spa_final_txg - TXG_DEFER_SIZE); 1953 } 1954 1955 pool_state_t 1956 spa_state(spa_t *spa) 1957 { 1958 return (spa->spa_state); 1959 } 1960 1961 spa_load_state_t 1962 spa_load_state(spa_t *spa) 1963 { 1964 return (spa->spa_load_state); 1965 } 1966 1967 uint64_t 1968 spa_freeze_txg(spa_t *spa) 1969 { 1970 return (spa->spa_freeze_txg); 1971 } 1972 1973 /* 1974 * Return the inflated asize for a logical write in bytes. This is used by the 1975 * DMU to calculate the space a logical write will require on disk. 1976 * If lsize is smaller than the largest physical block size allocatable on this 1977 * pool we use its value instead, since the write will end up using the whole 1978 * block anyway. 1979 */ 1980 uint64_t 1981 spa_get_worst_case_asize(spa_t *spa, uint64_t lsize) 1982 { 1983 if (lsize == 0) 1984 return (0); /* No inflation needed */ 1985 return (MAX(lsize, 1 << spa->spa_max_ashift) * spa_asize_inflation); 1986 } 1987 1988 /* 1989 * Return the range of minimum allocation sizes for the normal allocation 1990 * class. This can be used by external consumers of the DMU to estimate 1991 * potential wasted capacity when setting the recordsize for an object. 1992 * This is mainly for dRAID pools which always pad to a full stripe width. 1993 */ 1994 void 1995 spa_get_min_alloc_range(spa_t *spa, uint64_t *min_alloc, uint64_t *max_alloc) 1996 { 1997 *min_alloc = spa->spa_min_alloc; 1998 *max_alloc = spa->spa_max_alloc; 1999 } 2000 2001 /* 2002 * Return the amount of slop space in bytes. It is typically 1/32 of the pool 2003 * (3.2%), minus the embedded log space. On very small pools, it may be 2004 * slightly larger than this. On very large pools, it will be capped to 2005 * the value of spa_max_slop. The embedded log space is not included in 2006 * spa_dspace. By subtracting it, the usable space (per "zfs list") is a 2007 * constant 97% of the total space, regardless of metaslab size (assuming the 2008 * default spa_slop_shift=5 and a non-tiny pool). 2009 * 2010 * See the comment above spa_slop_shift for more details. 2011 */ 2012 uint64_t 2013 spa_get_slop_space(spa_t *spa) 2014 { 2015 uint64_t space = 0; 2016 uint64_t slop = 0; 2017 2018 /* 2019 * Make sure spa_dedup_dspace has been set. 2020 */ 2021 if (spa->spa_dedup_dspace == ~0ULL) 2022 spa_update_dspace(spa); 2023 2024 space = spa->spa_rdspace; 2025 slop = MIN(space >> spa_slop_shift, spa_max_slop); 2026 2027 /* 2028 * Subtract the embedded log space, but no more than half the (3.2%) 2029 * unusable space. Note, the "no more than half" is only relevant if 2030 * zfs_embedded_slog_min_ms >> spa_slop_shift < 2, which is not true by 2031 * default. 2032 */ 2033 uint64_t embedded_log = 2034 metaslab_class_get_dspace(spa_embedded_log_class(spa)); 2035 embedded_log += metaslab_class_get_dspace( 2036 spa_special_embedded_log_class(spa)); 2037 slop -= MIN(embedded_log, slop >> 1); 2038 2039 /* 2040 * Slop space should be at least spa_min_slop, but no more than half 2041 * the entire pool. 2042 */ 2043 slop = MAX(slop, MIN(space >> 1, spa_min_slop)); 2044 return (slop); 2045 } 2046 2047 uint64_t 2048 spa_get_dspace(spa_t *spa) 2049 { 2050 return (spa->spa_dspace); 2051 } 2052 2053 uint64_t 2054 spa_get_checkpoint_space(spa_t *spa) 2055 { 2056 return (spa->spa_checkpoint_info.sci_dspace); 2057 } 2058 2059 void 2060 spa_update_dspace(spa_t *spa) 2061 { 2062 spa->spa_rdspace = metaslab_class_get_dspace(spa_normal_class(spa)); 2063 if (spa->spa_nonallocating_dspace > 0) { 2064 /* 2065 * Subtract the space provided by all non-allocating vdevs that 2066 * contribute to dspace. If a file is overwritten, its old 2067 * blocks are freed and new blocks are allocated. If there are 2068 * no snapshots of the file, the available space should remain 2069 * the same. The old blocks could be freed from the 2070 * non-allocating vdev, but the new blocks must be allocated on 2071 * other (allocating) vdevs. By reserving the entire size of 2072 * the non-allocating vdevs (including allocated space), we 2073 * ensure that there will be enough space on the allocating 2074 * vdevs for this file overwrite to succeed. 2075 * 2076 * Note that the DMU/DSL doesn't actually know or care 2077 * how much space is allocated (it does its own tracking 2078 * of how much space has been logically used). So it 2079 * doesn't matter that the data we are moving may be 2080 * allocated twice (on the old device and the new device). 2081 */ 2082 ASSERT3U(spa->spa_rdspace, >=, spa->spa_nonallocating_dspace); 2083 spa->spa_rdspace -= spa->spa_nonallocating_dspace; 2084 } 2085 spa->spa_dspace = spa->spa_rdspace + 2086 metaslab_class_get_dalloc(spa_special_class(spa)) + 2087 metaslab_class_get_dalloc(spa_dedup_class(spa)) + 2088 ddt_get_dedup_dspace(spa) + 2089 brt_get_dspace(spa); 2090 } 2091 2092 /* 2093 * Return the failure mode that has been set to this pool. The default 2094 * behavior will be to block all I/Os when a complete failure occurs. 2095 */ 2096 uint64_t 2097 spa_get_failmode(spa_t *spa) 2098 { 2099 return (spa->spa_failmode); 2100 } 2101 2102 boolean_t 2103 spa_suspended(spa_t *spa) 2104 { 2105 return (spa->spa_suspended != ZIO_SUSPEND_NONE); 2106 } 2107 2108 uint64_t 2109 spa_version(spa_t *spa) 2110 { 2111 return (spa->spa_ubsync.ub_version); 2112 } 2113 2114 boolean_t 2115 spa_deflate(spa_t *spa) 2116 { 2117 return (spa->spa_deflate); 2118 } 2119 2120 metaslab_class_t * 2121 spa_normal_class(spa_t *spa) 2122 { 2123 return (spa->spa_normal_class); 2124 } 2125 2126 metaslab_class_t * 2127 spa_log_class(spa_t *spa) 2128 { 2129 return (spa->spa_log_class); 2130 } 2131 2132 metaslab_class_t * 2133 spa_embedded_log_class(spa_t *spa) 2134 { 2135 return (spa->spa_embedded_log_class); 2136 } 2137 2138 metaslab_class_t * 2139 spa_special_class(spa_t *spa) 2140 { 2141 return (spa->spa_special_class); 2142 } 2143 2144 metaslab_class_t * 2145 spa_special_embedded_log_class(spa_t *spa) 2146 { 2147 return (spa->spa_special_embedded_log_class); 2148 } 2149 2150 metaslab_class_t * 2151 spa_dedup_class(spa_t *spa) 2152 { 2153 return (spa->spa_dedup_class); 2154 } 2155 2156 boolean_t 2157 spa_special_has_ddt(spa_t *spa) 2158 { 2159 return (zfs_ddt_data_is_special && spa_has_special(spa)); 2160 } 2161 2162 /* 2163 * Locate an appropriate allocation class 2164 */ 2165 metaslab_class_t * 2166 spa_preferred_class(spa_t *spa, const zio_t *zio) 2167 { 2168 metaslab_class_t *mc = zio->io_metaslab_class; 2169 boolean_t tried_dedup = (mc == spa_dedup_class(spa)); 2170 boolean_t tried_special = (mc == spa_special_class(spa)); 2171 const zio_prop_t *zp = &zio->io_prop; 2172 2173 /* Gang children should always use the class of their parents. */ 2174 if (zio->io_flags & ZIO_FLAG_GANG_CHILD) { 2175 ASSERT(mc != NULL); 2176 return (mc); 2177 } 2178 2179 /* 2180 * Override object type for the purposes of selecting a storage class. 2181 * Primarily for DMU_OTN_ types where we can't explicitly control their 2182 * storage class; instead, choose a static type most closely matches 2183 * what we want. 2184 */ 2185 dmu_object_type_t objtype = 2186 zp->zp_storage_type == DMU_OT_NONE ? 2187 zp->zp_type : zp->zp_storage_type; 2188 2189 /* 2190 * ZIL allocations determine their class in zio_alloc_zil(). 2191 */ 2192 ASSERT(objtype != DMU_OT_INTENT_LOG); 2193 2194 if (DMU_OT_IS_DDT(objtype)) { 2195 if (spa_has_dedup(spa) && !tried_dedup && !tried_special) 2196 return (spa_dedup_class(spa)); 2197 else if (spa_special_has_ddt(spa) && !tried_special) 2198 return (spa_special_class(spa)); 2199 else 2200 return (spa_normal_class(spa)); 2201 } 2202 2203 if (!spa_has_special(spa) || tried_special) 2204 return (spa_normal_class(spa)); 2205 2206 if (DMU_OT_IS_METADATA(objtype) || 2207 (zfs_user_indirect_is_special && zp->zp_level > 0)) 2208 return (spa_special_class(spa)); 2209 2210 /* 2211 * Allow small blocks in special class. However, leave a reserve of 2212 * zfs_special_class_metadata_reserve_pct exclusively for metadata. 2213 */ 2214 if (zio->io_size <= zp->zp_zpl_smallblk) { 2215 metaslab_class_t *special = spa_special_class(spa); 2216 uint64_t limit = metaslab_class_get_space(special) * 2217 (100 - zfs_special_class_metadata_reserve_pct) / 100; 2218 2219 if (metaslab_class_get_alloc(special) < limit) 2220 return (special); 2221 } 2222 2223 return (spa_normal_class(spa)); 2224 } 2225 2226 void 2227 spa_evicting_os_register(spa_t *spa, objset_t *os) 2228 { 2229 mutex_enter(&spa->spa_evicting_os_lock); 2230 list_insert_head(&spa->spa_evicting_os_list, os); 2231 mutex_exit(&spa->spa_evicting_os_lock); 2232 } 2233 2234 void 2235 spa_evicting_os_deregister(spa_t *spa, objset_t *os) 2236 { 2237 mutex_enter(&spa->spa_evicting_os_lock); 2238 list_remove(&spa->spa_evicting_os_list, os); 2239 cv_broadcast(&spa->spa_evicting_os_cv); 2240 mutex_exit(&spa->spa_evicting_os_lock); 2241 } 2242 2243 void 2244 spa_evicting_os_wait(spa_t *spa) 2245 { 2246 mutex_enter(&spa->spa_evicting_os_lock); 2247 while (!list_is_empty(&spa->spa_evicting_os_list)) 2248 cv_wait(&spa->spa_evicting_os_cv, &spa->spa_evicting_os_lock); 2249 mutex_exit(&spa->spa_evicting_os_lock); 2250 2251 dmu_buf_user_evict_wait(); 2252 } 2253 2254 int 2255 spa_max_replication(spa_t *spa) 2256 { 2257 /* 2258 * As of SPA_VERSION == SPA_VERSION_DITTO_BLOCKS, we are able to 2259 * handle BPs with more than one DVA allocated. Set our max 2260 * replication level accordingly. 2261 */ 2262 if (spa_version(spa) < SPA_VERSION_DITTO_BLOCKS) 2263 return (1); 2264 return (MIN(SPA_DVAS_PER_BP, spa_max_replication_override)); 2265 } 2266 2267 int 2268 spa_prev_software_version(spa_t *spa) 2269 { 2270 return (spa->spa_prev_software_version); 2271 } 2272 2273 uint64_t 2274 spa_deadman_synctime(spa_t *spa) 2275 { 2276 return (spa->spa_deadman_synctime); 2277 } 2278 2279 spa_autotrim_t 2280 spa_get_autotrim(spa_t *spa) 2281 { 2282 return (spa->spa_autotrim); 2283 } 2284 2285 uint64_t 2286 spa_deadman_ziotime(spa_t *spa) 2287 { 2288 return (spa->spa_deadman_ziotime); 2289 } 2290 2291 uint64_t 2292 spa_get_deadman_failmode(spa_t *spa) 2293 { 2294 return (spa->spa_deadman_failmode); 2295 } 2296 2297 void 2298 spa_set_deadman_failmode(spa_t *spa, const char *failmode) 2299 { 2300 if (strcmp(failmode, "wait") == 0) 2301 spa->spa_deadman_failmode = ZIO_FAILURE_MODE_WAIT; 2302 else if (strcmp(failmode, "continue") == 0) 2303 spa->spa_deadman_failmode = ZIO_FAILURE_MODE_CONTINUE; 2304 else if (strcmp(failmode, "panic") == 0) 2305 spa->spa_deadman_failmode = ZIO_FAILURE_MODE_PANIC; 2306 else 2307 spa->spa_deadman_failmode = ZIO_FAILURE_MODE_WAIT; 2308 } 2309 2310 void 2311 spa_set_deadman_ziotime(hrtime_t ns) 2312 { 2313 spa_t *spa = NULL; 2314 2315 if (spa_mode_global != SPA_MODE_UNINIT) { 2316 spa_namespace_enter(FTAG); 2317 while ((spa = spa_next(spa)) != NULL) 2318 spa->spa_deadman_ziotime = ns; 2319 spa_namespace_exit(FTAG); 2320 } 2321 } 2322 2323 void 2324 spa_set_deadman_synctime(hrtime_t ns) 2325 { 2326 spa_t *spa = NULL; 2327 2328 if (spa_mode_global != SPA_MODE_UNINIT) { 2329 spa_namespace_enter(FTAG); 2330 while ((spa = spa_next(spa)) != NULL) 2331 spa->spa_deadman_synctime = ns; 2332 spa_namespace_exit(FTAG); 2333 } 2334 } 2335 2336 uint64_t 2337 dva_get_dsize_sync(spa_t *spa, const dva_t *dva) 2338 { 2339 uint64_t asize = DVA_GET_ASIZE(dva); 2340 uint64_t dsize = asize; 2341 2342 ASSERT(spa_config_held(spa, SCL_ALL, RW_READER) != 0); 2343 2344 if (asize != 0 && spa->spa_deflate) { 2345 vdev_t *vd = vdev_lookup_top(spa, DVA_GET_VDEV(dva)); 2346 if (vd != NULL) 2347 dsize = (asize >> SPA_MINBLOCKSHIFT) * 2348 vd->vdev_deflate_ratio; 2349 } 2350 2351 return (dsize); 2352 } 2353 2354 uint64_t 2355 bp_get_dsize_sync(spa_t *spa, const blkptr_t *bp) 2356 { 2357 uint64_t dsize = 0; 2358 2359 for (int d = 0; d < BP_GET_NDVAS(bp); d++) 2360 dsize += dva_get_dsize_sync(spa, &bp->blk_dva[d]); 2361 2362 return (dsize); 2363 } 2364 2365 uint64_t 2366 bp_get_dsize(spa_t *spa, const blkptr_t *bp) 2367 { 2368 uint64_t dsize = 0; 2369 2370 spa_config_enter(spa, SCL_VDEV, FTAG, RW_READER); 2371 2372 for (int d = 0; d < BP_GET_NDVAS(bp); d++) 2373 dsize += dva_get_dsize_sync(spa, &bp->blk_dva[d]); 2374 2375 spa_config_exit(spa, SCL_VDEV, FTAG); 2376 2377 return (dsize); 2378 } 2379 2380 uint64_t 2381 spa_dirty_data(spa_t *spa) 2382 { 2383 return (spa->spa_dsl_pool->dp_dirty_total); 2384 } 2385 2386 /* 2387 * ========================================================================== 2388 * SPA Import Progress Routines 2389 * ========================================================================== 2390 */ 2391 2392 typedef struct spa_import_progress { 2393 uint64_t pool_guid; /* unique id for updates */ 2394 char *pool_name; 2395 spa_load_state_t spa_load_state; 2396 char *spa_load_notes; 2397 uint64_t mmp_sec_remaining; /* MMP activity check */ 2398 uint64_t spa_load_max_txg; /* rewind txg */ 2399 procfs_list_node_t smh_node; 2400 } spa_import_progress_t; 2401 2402 spa_history_list_t *spa_import_progress_list = NULL; 2403 2404 static int 2405 spa_import_progress_show_header(struct seq_file *f) 2406 { 2407 seq_printf(f, "%-20s %-14s %-14s %-12s %-16s %s\n", "pool_guid", 2408 "load_state", "multihost_secs", "max_txg", 2409 "pool_name", "notes"); 2410 return (0); 2411 } 2412 2413 static int 2414 spa_import_progress_show(struct seq_file *f, void *data) 2415 { 2416 spa_import_progress_t *sip = (spa_import_progress_t *)data; 2417 2418 seq_printf(f, "%-20llu %-14llu %-14llu %-12llu %-16s %s\n", 2419 (u_longlong_t)sip->pool_guid, (u_longlong_t)sip->spa_load_state, 2420 (u_longlong_t)sip->mmp_sec_remaining, 2421 (u_longlong_t)sip->spa_load_max_txg, 2422 (sip->pool_name ? sip->pool_name : "-"), 2423 (sip->spa_load_notes ? sip->spa_load_notes : "-")); 2424 2425 return (0); 2426 } 2427 2428 /* Remove oldest elements from list until there are no more than 'size' left */ 2429 static void 2430 spa_import_progress_truncate(spa_history_list_t *shl, unsigned int size) 2431 { 2432 spa_import_progress_t *sip; 2433 while (shl->size > size) { 2434 sip = list_remove_head(&shl->procfs_list.pl_list); 2435 if (sip->pool_name) 2436 spa_strfree(sip->pool_name); 2437 if (sip->spa_load_notes) 2438 kmem_strfree(sip->spa_load_notes); 2439 kmem_free(sip, sizeof (spa_import_progress_t)); 2440 shl->size--; 2441 } 2442 2443 IMPLY(size == 0, list_is_empty(&shl->procfs_list.pl_list)); 2444 } 2445 2446 static void 2447 spa_import_progress_init(void) 2448 { 2449 spa_import_progress_list = kmem_zalloc(sizeof (spa_history_list_t), 2450 KM_SLEEP); 2451 2452 spa_import_progress_list->size = 0; 2453 2454 spa_import_progress_list->procfs_list.pl_private = 2455 spa_import_progress_list; 2456 2457 procfs_list_install("zfs", 2458 NULL, 2459 "import_progress", 2460 0644, 2461 &spa_import_progress_list->procfs_list, 2462 spa_import_progress_show, 2463 spa_import_progress_show_header, 2464 NULL, 2465 offsetof(spa_import_progress_t, smh_node)); 2466 } 2467 2468 static void 2469 spa_import_progress_destroy(void) 2470 { 2471 spa_history_list_t *shl = spa_import_progress_list; 2472 procfs_list_uninstall(&shl->procfs_list); 2473 spa_import_progress_truncate(shl, 0); 2474 procfs_list_destroy(&shl->procfs_list); 2475 kmem_free(shl, sizeof (spa_history_list_t)); 2476 } 2477 2478 int 2479 spa_import_progress_set_state(uint64_t pool_guid, 2480 spa_load_state_t load_state) 2481 { 2482 spa_history_list_t *shl = spa_import_progress_list; 2483 spa_import_progress_t *sip; 2484 int error = ENOENT; 2485 2486 if (shl->size == 0) 2487 return (0); 2488 2489 mutex_enter(&shl->procfs_list.pl_lock); 2490 for (sip = list_tail(&shl->procfs_list.pl_list); sip != NULL; 2491 sip = list_prev(&shl->procfs_list.pl_list, sip)) { 2492 if (sip->pool_guid == pool_guid) { 2493 sip->spa_load_state = load_state; 2494 if (sip->spa_load_notes != NULL) { 2495 kmem_strfree(sip->spa_load_notes); 2496 sip->spa_load_notes = NULL; 2497 } 2498 error = 0; 2499 break; 2500 } 2501 } 2502 mutex_exit(&shl->procfs_list.pl_lock); 2503 2504 return (error); 2505 } 2506 2507 static void 2508 spa_import_progress_set_notes_impl(spa_t *spa, boolean_t log_dbgmsg, 2509 const char *fmt, va_list adx) 2510 { 2511 spa_history_list_t *shl = spa_import_progress_list; 2512 spa_import_progress_t *sip; 2513 uint64_t pool_guid = spa_guid(spa); 2514 2515 if (shl->size == 0) 2516 return; 2517 2518 char *notes = kmem_vasprintf(fmt, adx); 2519 2520 mutex_enter(&shl->procfs_list.pl_lock); 2521 for (sip = list_tail(&shl->procfs_list.pl_list); sip != NULL; 2522 sip = list_prev(&shl->procfs_list.pl_list, sip)) { 2523 if (sip->pool_guid == pool_guid) { 2524 if (sip->spa_load_notes != NULL) { 2525 kmem_strfree(sip->spa_load_notes); 2526 sip->spa_load_notes = NULL; 2527 } 2528 sip->spa_load_notes = notes; 2529 if (log_dbgmsg) 2530 zfs_dbgmsg("'%s' %s", sip->pool_name, notes); 2531 notes = NULL; 2532 break; 2533 } 2534 } 2535 mutex_exit(&shl->procfs_list.pl_lock); 2536 if (notes != NULL) 2537 kmem_strfree(notes); 2538 } 2539 2540 void 2541 spa_import_progress_set_notes(spa_t *spa, const char *fmt, ...) 2542 { 2543 va_list adx; 2544 2545 va_start(adx, fmt); 2546 spa_import_progress_set_notes_impl(spa, B_TRUE, fmt, adx); 2547 va_end(adx); 2548 } 2549 2550 void 2551 spa_import_progress_set_notes_nolog(spa_t *spa, const char *fmt, ...) 2552 { 2553 va_list adx; 2554 2555 va_start(adx, fmt); 2556 spa_import_progress_set_notes_impl(spa, B_FALSE, fmt, adx); 2557 va_end(adx); 2558 } 2559 2560 int 2561 spa_import_progress_set_max_txg(uint64_t pool_guid, uint64_t load_max_txg) 2562 { 2563 spa_history_list_t *shl = spa_import_progress_list; 2564 spa_import_progress_t *sip; 2565 int error = ENOENT; 2566 2567 if (shl->size == 0) 2568 return (0); 2569 2570 mutex_enter(&shl->procfs_list.pl_lock); 2571 for (sip = list_tail(&shl->procfs_list.pl_list); sip != NULL; 2572 sip = list_prev(&shl->procfs_list.pl_list, sip)) { 2573 if (sip->pool_guid == pool_guid) { 2574 sip->spa_load_max_txg = load_max_txg; 2575 error = 0; 2576 break; 2577 } 2578 } 2579 mutex_exit(&shl->procfs_list.pl_lock); 2580 2581 return (error); 2582 } 2583 2584 int 2585 spa_import_progress_set_mmp_check(uint64_t pool_guid, 2586 uint64_t mmp_sec_remaining) 2587 { 2588 spa_history_list_t *shl = spa_import_progress_list; 2589 spa_import_progress_t *sip; 2590 int error = ENOENT; 2591 2592 if (shl->size == 0) 2593 return (0); 2594 2595 mutex_enter(&shl->procfs_list.pl_lock); 2596 for (sip = list_tail(&shl->procfs_list.pl_list); sip != NULL; 2597 sip = list_prev(&shl->procfs_list.pl_list, sip)) { 2598 if (sip->pool_guid == pool_guid) { 2599 sip->mmp_sec_remaining = mmp_sec_remaining; 2600 error = 0; 2601 break; 2602 } 2603 } 2604 mutex_exit(&shl->procfs_list.pl_lock); 2605 2606 return (error); 2607 } 2608 2609 /* 2610 * A new import is in progress, add an entry. 2611 */ 2612 void 2613 spa_import_progress_add(spa_t *spa) 2614 { 2615 spa_history_list_t *shl = spa_import_progress_list; 2616 spa_import_progress_t *sip; 2617 const char *poolname = NULL; 2618 2619 sip = kmem_zalloc(sizeof (spa_import_progress_t), KM_SLEEP); 2620 sip->pool_guid = spa_guid(spa); 2621 2622 (void) nvlist_lookup_string(spa->spa_config, ZPOOL_CONFIG_POOL_NAME, 2623 &poolname); 2624 if (poolname == NULL) 2625 poolname = spa_name(spa); 2626 sip->pool_name = spa_strdup(poolname); 2627 sip->spa_load_state = spa_load_state(spa); 2628 sip->spa_load_notes = NULL; 2629 2630 mutex_enter(&shl->procfs_list.pl_lock); 2631 procfs_list_add(&shl->procfs_list, sip); 2632 shl->size++; 2633 mutex_exit(&shl->procfs_list.pl_lock); 2634 } 2635 2636 void 2637 spa_import_progress_remove(uint64_t pool_guid) 2638 { 2639 spa_history_list_t *shl = spa_import_progress_list; 2640 spa_import_progress_t *sip; 2641 2642 mutex_enter(&shl->procfs_list.pl_lock); 2643 for (sip = list_tail(&shl->procfs_list.pl_list); sip != NULL; 2644 sip = list_prev(&shl->procfs_list.pl_list, sip)) { 2645 if (sip->pool_guid == pool_guid) { 2646 if (sip->pool_name) 2647 spa_strfree(sip->pool_name); 2648 if (sip->spa_load_notes) 2649 spa_strfree(sip->spa_load_notes); 2650 list_remove(&shl->procfs_list.pl_list, sip); 2651 shl->size--; 2652 kmem_free(sip, sizeof (spa_import_progress_t)); 2653 break; 2654 } 2655 } 2656 mutex_exit(&shl->procfs_list.pl_lock); 2657 } 2658 2659 /* 2660 * ========================================================================== 2661 * Initialization and Termination 2662 * ========================================================================== 2663 */ 2664 2665 static int 2666 spa_name_compare(const void *a1, const void *a2) 2667 { 2668 const spa_t *s1 = a1; 2669 const spa_t *s2 = a2; 2670 2671 return (TREE_ISIGN(strcmp(s1->spa_name, s2->spa_name))); 2672 } 2673 2674 void 2675 spa_init(spa_mode_t mode) 2676 { 2677 mutex_init(&spa_namespace_lock, NULL, MUTEX_DEFAULT, NULL); 2678 mutex_init(&spa_spare_lock, NULL, MUTEX_DEFAULT, NULL); 2679 mutex_init(&spa_l2cache_lock, NULL, MUTEX_DEFAULT, NULL); 2680 cv_init(&spa_namespace_cv, NULL, CV_DEFAULT, NULL); 2681 2682 avl_create(&spa_namespace_avl, spa_name_compare, sizeof (spa_t), 2683 offsetof(spa_t, spa_avl)); 2684 2685 avl_create(&spa_spare_avl, spa_spare_compare, sizeof (spa_aux_t), 2686 offsetof(spa_aux_t, aux_avl)); 2687 2688 avl_create(&spa_l2cache_avl, spa_l2cache_compare, sizeof (spa_aux_t), 2689 offsetof(spa_aux_t, aux_avl)); 2690 2691 spa_mode_global = mode; 2692 2693 #ifndef _KERNEL 2694 if (spa_mode_global != SPA_MODE_READ && dprintf_find_string("watch")) { 2695 struct sigaction sa; 2696 2697 sa.sa_flags = SA_SIGINFO; 2698 sigemptyset(&sa.sa_mask); 2699 sa.sa_sigaction = arc_buf_sigsegv; 2700 2701 if (sigaction(SIGSEGV, &sa, NULL) == -1) { 2702 perror("could not enable watchpoints: " 2703 "sigaction(SIGSEGV, ...) = "); 2704 } else { 2705 arc_watch = B_TRUE; 2706 } 2707 } 2708 #endif 2709 2710 fm_init(); 2711 zfs_refcount_init(); 2712 unique_init(); 2713 zfs_btree_init(); 2714 metaslab_stat_init(); 2715 brt_init(); 2716 ddt_init(); 2717 zio_init(); 2718 dmu_init(); 2719 zil_init(); 2720 vdev_mirror_stat_init(); 2721 vdev_raidz_math_init(); 2722 vdev_file_init(); 2723 zfs_prop_init(); 2724 chksum_init(); 2725 zpool_prop_init(); 2726 zpool_feature_init(); 2727 vdev_prop_init(); 2728 scan_init(); 2729 qat_init(); 2730 spa_import_progress_init(); 2731 zap_init(); 2732 } 2733 2734 void 2735 spa_fini(void) 2736 { 2737 spa_evict_all(); 2738 2739 vdev_file_fini(); 2740 vdev_mirror_stat_fini(); 2741 vdev_raidz_math_fini(); 2742 chksum_fini(); 2743 zil_fini(); 2744 dmu_fini(); 2745 zio_fini(); 2746 ddt_fini(); 2747 brt_fini(); 2748 metaslab_stat_fini(); 2749 zfs_btree_fini(); 2750 unique_fini(); 2751 zfs_refcount_fini(); 2752 fm_fini(); 2753 scan_fini(); 2754 qat_fini(); 2755 spa_import_progress_destroy(); 2756 zap_fini(); 2757 2758 avl_destroy(&spa_namespace_avl); 2759 avl_destroy(&spa_spare_avl); 2760 avl_destroy(&spa_l2cache_avl); 2761 2762 cv_destroy(&spa_namespace_cv); 2763 mutex_destroy(&spa_namespace_lock); 2764 mutex_destroy(&spa_spare_lock); 2765 mutex_destroy(&spa_l2cache_lock); 2766 } 2767 2768 boolean_t 2769 spa_has_dedup(spa_t *spa) 2770 { 2771 return (spa->spa_dedup_class->mc_groups != 0); 2772 } 2773 2774 /* 2775 * Return whether this pool has a dedicated slog device. No locking needed. 2776 * It's not a problem if the wrong answer is returned as it's only for 2777 * performance and not correctness. 2778 */ 2779 boolean_t 2780 spa_has_slogs(spa_t *spa) 2781 { 2782 return (spa->spa_log_class->mc_groups != 0); 2783 } 2784 2785 boolean_t 2786 spa_has_special(spa_t *spa) 2787 { 2788 return (spa->spa_special_class->mc_groups != 0); 2789 } 2790 2791 spa_log_state_t 2792 spa_get_log_state(spa_t *spa) 2793 { 2794 return (spa->spa_log_state); 2795 } 2796 2797 void 2798 spa_set_log_state(spa_t *spa, spa_log_state_t state) 2799 { 2800 spa->spa_log_state = state; 2801 } 2802 2803 boolean_t 2804 spa_is_root(spa_t *spa) 2805 { 2806 return (spa->spa_is_root); 2807 } 2808 2809 boolean_t 2810 spa_writeable(spa_t *spa) 2811 { 2812 return (!!(spa->spa_mode & SPA_MODE_WRITE) && spa->spa_trust_config); 2813 } 2814 2815 /* 2816 * Returns true if there is a pending sync task in any of the current 2817 * syncing txg, the current quiescing txg, or the current open txg. 2818 */ 2819 boolean_t 2820 spa_has_pending_synctask(spa_t *spa) 2821 { 2822 return (!txg_all_lists_empty(&spa->spa_dsl_pool->dp_sync_tasks) || 2823 !txg_all_lists_empty(&spa->spa_dsl_pool->dp_early_sync_tasks)); 2824 } 2825 2826 spa_mode_t 2827 spa_mode(spa_t *spa) 2828 { 2829 return (spa->spa_mode); 2830 } 2831 2832 uint64_t 2833 spa_get_last_scrubbed_txg(spa_t *spa) 2834 { 2835 return (spa->spa_scrubbed_last_txg); 2836 } 2837 2838 uint64_t 2839 spa_bootfs(spa_t *spa) 2840 { 2841 return (spa->spa_bootfs); 2842 } 2843 2844 uint64_t 2845 spa_delegation(spa_t *spa) 2846 { 2847 return (spa->spa_delegation); 2848 } 2849 2850 objset_t * 2851 spa_meta_objset(spa_t *spa) 2852 { 2853 return (spa->spa_meta_objset); 2854 } 2855 2856 enum zio_checksum 2857 spa_dedup_checksum(spa_t *spa) 2858 { 2859 return (spa->spa_dedup_checksum); 2860 } 2861 2862 /* 2863 * Reset pool scan stat per scan pass (or reboot). 2864 */ 2865 void 2866 spa_scan_stat_init(spa_t *spa) 2867 { 2868 /* data not stored on disk */ 2869 spa->spa_scan_pass_start = gethrestime_sec(); 2870 if (dsl_scan_is_paused_scrub(spa->spa_dsl_pool->dp_scan)) 2871 spa->spa_scan_pass_scrub_pause = spa->spa_scan_pass_start; 2872 else 2873 spa->spa_scan_pass_scrub_pause = 0; 2874 2875 if (dsl_errorscrub_is_paused(spa->spa_dsl_pool->dp_scan)) 2876 spa->spa_scan_pass_errorscrub_pause = spa->spa_scan_pass_start; 2877 else 2878 spa->spa_scan_pass_errorscrub_pause = 0; 2879 2880 spa->spa_scan_pass_scrub_spent_paused = 0; 2881 spa->spa_scan_pass_exam = 0; 2882 spa->spa_scan_pass_issued = 0; 2883 2884 // error scrub stats 2885 spa->spa_scan_pass_errorscrub_spent_paused = 0; 2886 } 2887 2888 /* 2889 * Get scan stats for zpool status reports 2890 */ 2891 int 2892 spa_scan_get_stats(spa_t *spa, pool_scan_stat_t *ps) 2893 { 2894 dsl_scan_t *scn = spa->spa_dsl_pool ? spa->spa_dsl_pool->dp_scan : NULL; 2895 2896 if (scn == NULL || (scn->scn_phys.scn_func == POOL_SCAN_NONE && 2897 scn->errorscrub_phys.dep_func == POOL_SCAN_NONE)) 2898 return (SET_ERROR(ENOENT)); 2899 2900 memset(ps, 0, sizeof (pool_scan_stat_t)); 2901 2902 /* data stored on disk */ 2903 ps->pss_func = scn->scn_phys.scn_func; 2904 ps->pss_state = scn->scn_phys.scn_state; 2905 ps->pss_start_time = scn->scn_phys.scn_start_time; 2906 ps->pss_end_time = scn->scn_phys.scn_end_time; 2907 ps->pss_to_examine = scn->scn_phys.scn_to_examine; 2908 ps->pss_examined = scn->scn_phys.scn_examined; 2909 ps->pss_skipped = scn->scn_phys.scn_skipped; 2910 ps->pss_processed = scn->scn_phys.scn_processed; 2911 ps->pss_errors = scn->scn_phys.scn_errors; 2912 2913 /* data not stored on disk */ 2914 ps->pss_pass_exam = spa->spa_scan_pass_exam; 2915 ps->pss_pass_start = spa->spa_scan_pass_start; 2916 ps->pss_pass_scrub_pause = spa->spa_scan_pass_scrub_pause; 2917 ps->pss_pass_scrub_spent_paused = spa->spa_scan_pass_scrub_spent_paused; 2918 ps->pss_pass_issued = spa->spa_scan_pass_issued; 2919 ps->pss_issued = 2920 scn->scn_issued_before_pass + spa->spa_scan_pass_issued; 2921 2922 /* error scrub data stored on disk */ 2923 ps->pss_error_scrub_func = scn->errorscrub_phys.dep_func; 2924 ps->pss_error_scrub_state = scn->errorscrub_phys.dep_state; 2925 ps->pss_error_scrub_start = scn->errorscrub_phys.dep_start_time; 2926 ps->pss_error_scrub_end = scn->errorscrub_phys.dep_end_time; 2927 ps->pss_error_scrub_examined = scn->errorscrub_phys.dep_examined; 2928 ps->pss_error_scrub_to_be_examined = 2929 scn->errorscrub_phys.dep_to_examine; 2930 2931 /* error scrub data not stored on disk */ 2932 ps->pss_pass_error_scrub_pause = spa->spa_scan_pass_errorscrub_pause; 2933 ps->pss_pass_scrub_flags = 0; 2934 if (scn->scn_phys.scn_flags & DSF_SCRUB_THOROUGH) 2935 ps->pss_pass_scrub_flags |= POOL_SCRUB_THOROUGH; 2936 2937 return (0); 2938 } 2939 2940 int 2941 spa_maxblocksize(spa_t *spa) 2942 { 2943 if (spa_feature_is_enabled(spa, SPA_FEATURE_LARGE_BLOCKS)) 2944 return (SPA_MAXBLOCKSIZE); 2945 else 2946 return (SPA_OLD_MAXBLOCKSIZE); 2947 } 2948 2949 2950 /* 2951 * Returns the txg that the last device removal completed. No indirect mappings 2952 * have been added since this txg. 2953 */ 2954 uint64_t 2955 spa_get_last_removal_txg(spa_t *spa) 2956 { 2957 uint64_t vdevid; 2958 uint64_t ret = -1ULL; 2959 2960 spa_config_enter(spa, SCL_VDEV, FTAG, RW_READER); 2961 /* 2962 * sr_prev_indirect_vdev is only modified while holding all the 2963 * config locks, so it is sufficient to hold SCL_VDEV as reader when 2964 * examining it. 2965 */ 2966 vdevid = spa->spa_removing_phys.sr_prev_indirect_vdev; 2967 2968 while (vdevid != -1ULL) { 2969 vdev_t *vd = vdev_lookup_top(spa, vdevid); 2970 vdev_indirect_births_t *vib = vd->vdev_indirect_births; 2971 2972 ASSERT3P(vd->vdev_ops, ==, &vdev_indirect_ops); 2973 2974 /* 2975 * If the removal did not remap any data, we don't care. 2976 */ 2977 if (vdev_indirect_births_count(vib) != 0) { 2978 ret = vdev_indirect_births_last_entry_txg(vib); 2979 break; 2980 } 2981 2982 vdevid = vd->vdev_indirect_config.vic_prev_indirect_vdev; 2983 } 2984 spa_config_exit(spa, SCL_VDEV, FTAG); 2985 2986 IMPLY(ret != -1ULL, 2987 spa_feature_is_active(spa, SPA_FEATURE_DEVICE_REMOVAL)); 2988 2989 return (ret); 2990 } 2991 2992 int 2993 spa_maxdnodesize(spa_t *spa) 2994 { 2995 if (spa_feature_is_enabled(spa, SPA_FEATURE_LARGE_DNODE)) 2996 return (DNODE_MAX_SIZE); 2997 else 2998 return (DNODE_MIN_SIZE); 2999 } 3000 3001 boolean_t 3002 spa_multihost(spa_t *spa) 3003 { 3004 return (spa->spa_multihost ? B_TRUE : B_FALSE); 3005 } 3006 3007 uint32_t 3008 spa_get_hostid(spa_t *spa) 3009 { 3010 return (spa->spa_hostid); 3011 } 3012 3013 boolean_t 3014 spa_trust_config(spa_t *spa) 3015 { 3016 return (spa->spa_trust_config); 3017 } 3018 3019 uint64_t 3020 spa_missing_tvds_allowed(spa_t *spa) 3021 { 3022 return (spa->spa_missing_tvds_allowed); 3023 } 3024 3025 space_map_t * 3026 spa_syncing_log_sm(spa_t *spa) 3027 { 3028 return (spa->spa_syncing_log_sm); 3029 } 3030 3031 void 3032 spa_set_missing_tvds(spa_t *spa, uint64_t missing) 3033 { 3034 spa->spa_missing_tvds = missing; 3035 } 3036 3037 /* 3038 * Return the pool state string ("ONLINE", "DEGRADED", "SUSPENDED", etc). 3039 */ 3040 const char * 3041 spa_state_to_name(spa_t *spa) 3042 { 3043 ASSERT3P(spa, !=, NULL); 3044 3045 /* 3046 * it is possible for the spa to exist, without root vdev 3047 * as the spa transitions during import/export 3048 */ 3049 vdev_t *rvd = spa->spa_root_vdev; 3050 if (rvd == NULL) { 3051 return ("TRANSITIONING"); 3052 } 3053 vdev_state_t state = rvd->vdev_state; 3054 vdev_aux_t aux = rvd->vdev_stat.vs_aux; 3055 3056 if (spa_suspended(spa)) 3057 return ("SUSPENDED"); 3058 3059 switch (state) { 3060 case VDEV_STATE_CLOSED: 3061 case VDEV_STATE_OFFLINE: 3062 return ("OFFLINE"); 3063 case VDEV_STATE_REMOVED: 3064 return ("REMOVED"); 3065 case VDEV_STATE_CANT_OPEN: 3066 if (aux == VDEV_AUX_CORRUPT_DATA || aux == VDEV_AUX_BAD_LOG) 3067 return ("FAULTED"); 3068 else if (aux == VDEV_AUX_SPLIT_POOL) 3069 return ("SPLIT"); 3070 else 3071 return ("UNAVAIL"); 3072 case VDEV_STATE_FAULTED: 3073 return ("FAULTED"); 3074 case VDEV_STATE_DEGRADED: 3075 return ("DEGRADED"); 3076 case VDEV_STATE_HEALTHY: 3077 return ("ONLINE"); 3078 default: 3079 break; 3080 } 3081 3082 return ("UNKNOWN"); 3083 } 3084 3085 boolean_t 3086 spa_top_vdevs_spacemap_addressable(spa_t *spa) 3087 { 3088 vdev_t *rvd = spa->spa_root_vdev; 3089 for (uint64_t c = 0; c < rvd->vdev_children; c++) { 3090 if (!vdev_is_spacemap_addressable(rvd->vdev_child[c])) 3091 return (B_FALSE); 3092 } 3093 return (B_TRUE); 3094 } 3095 3096 boolean_t 3097 spa_has_checkpoint(spa_t *spa) 3098 { 3099 return (spa->spa_checkpoint_txg != 0); 3100 } 3101 3102 boolean_t 3103 spa_importing_readonly_checkpoint(spa_t *spa) 3104 { 3105 return ((spa->spa_import_flags & ZFS_IMPORT_CHECKPOINT) && 3106 spa->spa_mode == SPA_MODE_READ); 3107 } 3108 3109 uint64_t 3110 spa_min_claim_txg(spa_t *spa) 3111 { 3112 uint64_t checkpoint_txg = spa->spa_uberblock.ub_checkpoint_txg; 3113 3114 if (checkpoint_txg != 0) 3115 return (checkpoint_txg + 1); 3116 3117 return (spa->spa_first_txg); 3118 } 3119 3120 /* 3121 * If there is a checkpoint, async destroys may consume more space from 3122 * the pool instead of freeing it. In an attempt to save the pool from 3123 * getting suspended when it is about to run out of space, we stop 3124 * processing async destroys. 3125 */ 3126 boolean_t 3127 spa_suspend_async_destroy(spa_t *spa) 3128 { 3129 dsl_pool_t *dp = spa_get_dsl(spa); 3130 3131 uint64_t unreserved = dsl_pool_unreserved_space(dp, 3132 ZFS_SPACE_CHECK_EXTRA_RESERVED); 3133 uint64_t used = dsl_dir_phys(dp->dp_root_dir)->dd_used_bytes; 3134 uint64_t avail = (unreserved > used) ? (unreserved - used) : 0; 3135 3136 if (spa_has_checkpoint(spa) && avail == 0) 3137 return (B_TRUE); 3138 3139 return (B_FALSE); 3140 } 3141 3142 #if defined(_KERNEL) 3143 3144 int 3145 param_set_deadman_failmode_common(const char *val) 3146 { 3147 spa_t *spa = NULL; 3148 char *p; 3149 3150 if (val == NULL) 3151 return (SET_ERROR(EINVAL)); 3152 3153 if ((p = strchr(val, '\n')) != NULL) 3154 *p = '\0'; 3155 3156 if (strcmp(val, "wait") != 0 && strcmp(val, "continue") != 0 && 3157 strcmp(val, "panic")) 3158 return (SET_ERROR(EINVAL)); 3159 3160 if (spa_mode_global != SPA_MODE_UNINIT) { 3161 spa_namespace_enter(FTAG); 3162 while ((spa = spa_next(spa)) != NULL) 3163 spa_set_deadman_failmode(spa, val); 3164 spa_namespace_exit(FTAG); 3165 } 3166 3167 return (0); 3168 } 3169 #endif 3170 3171 /* Namespace manipulation */ 3172 EXPORT_SYMBOL(spa_lookup); 3173 EXPORT_SYMBOL(spa_add); 3174 EXPORT_SYMBOL(spa_remove); 3175 EXPORT_SYMBOL(spa_next); 3176 3177 /* Refcount functions */ 3178 EXPORT_SYMBOL(spa_open_ref); 3179 EXPORT_SYMBOL(spa_close); 3180 EXPORT_SYMBOL(spa_refcount_zero); 3181 3182 /* Pool configuration lock */ 3183 EXPORT_SYMBOL(spa_config_tryenter); 3184 EXPORT_SYMBOL(spa_config_enter); 3185 EXPORT_SYMBOL(spa_config_exit); 3186 EXPORT_SYMBOL(spa_config_held); 3187 3188 /* Pool vdev add/remove lock */ 3189 EXPORT_SYMBOL(spa_vdev_enter); 3190 EXPORT_SYMBOL(spa_vdev_exit); 3191 3192 /* Pool vdev state change lock */ 3193 EXPORT_SYMBOL(spa_vdev_state_enter); 3194 EXPORT_SYMBOL(spa_vdev_state_exit); 3195 3196 /* Accessor functions */ 3197 EXPORT_SYMBOL(spa_shutting_down); 3198 EXPORT_SYMBOL(spa_get_dsl); 3199 EXPORT_SYMBOL(spa_get_rootblkptr); 3200 EXPORT_SYMBOL(spa_set_rootblkptr); 3201 EXPORT_SYMBOL(spa_altroot); 3202 EXPORT_SYMBOL(spa_sync_pass); 3203 EXPORT_SYMBOL(spa_name); 3204 EXPORT_SYMBOL(spa_load_name); 3205 EXPORT_SYMBOL(spa_guid); 3206 EXPORT_SYMBOL(spa_last_synced_txg); 3207 EXPORT_SYMBOL(spa_first_txg); 3208 EXPORT_SYMBOL(spa_syncing_txg); 3209 EXPORT_SYMBOL(spa_version); 3210 EXPORT_SYMBOL(spa_state); 3211 EXPORT_SYMBOL(spa_load_state); 3212 EXPORT_SYMBOL(spa_freeze_txg); 3213 EXPORT_SYMBOL(spa_get_min_alloc_range); /* for Lustre */ 3214 EXPORT_SYMBOL(spa_get_dspace); 3215 EXPORT_SYMBOL(spa_update_dspace); 3216 EXPORT_SYMBOL(spa_deflate); 3217 EXPORT_SYMBOL(spa_normal_class); 3218 EXPORT_SYMBOL(spa_log_class); 3219 EXPORT_SYMBOL(spa_special_class); 3220 EXPORT_SYMBOL(spa_preferred_class); 3221 EXPORT_SYMBOL(spa_max_replication); 3222 EXPORT_SYMBOL(spa_prev_software_version); 3223 EXPORT_SYMBOL(spa_get_failmode); 3224 EXPORT_SYMBOL(spa_suspended); 3225 EXPORT_SYMBOL(spa_bootfs); 3226 EXPORT_SYMBOL(spa_delegation); 3227 EXPORT_SYMBOL(spa_meta_objset); 3228 EXPORT_SYMBOL(spa_maxblocksize); 3229 EXPORT_SYMBOL(spa_maxdnodesize); 3230 3231 /* Miscellaneous support routines */ 3232 EXPORT_SYMBOL(spa_guid_exists); 3233 EXPORT_SYMBOL(spa_strdup); 3234 EXPORT_SYMBOL(spa_strfree); 3235 EXPORT_SYMBOL(spa_generate_guid); 3236 EXPORT_SYMBOL(snprintf_blkptr); 3237 EXPORT_SYMBOL(spa_freeze); 3238 EXPORT_SYMBOL(spa_upgrade); 3239 EXPORT_SYMBOL(spa_evict_all); 3240 EXPORT_SYMBOL(spa_lookup_by_guid); 3241 EXPORT_SYMBOL(spa_has_spare); 3242 EXPORT_SYMBOL(dva_get_dsize_sync); 3243 EXPORT_SYMBOL(bp_get_dsize_sync); 3244 EXPORT_SYMBOL(bp_get_dsize); 3245 EXPORT_SYMBOL(spa_has_slogs); 3246 EXPORT_SYMBOL(spa_is_root); 3247 EXPORT_SYMBOL(spa_writeable); 3248 EXPORT_SYMBOL(spa_mode); 3249 EXPORT_SYMBOL(spa_trust_config); 3250 EXPORT_SYMBOL(spa_missing_tvds_allowed); 3251 EXPORT_SYMBOL(spa_set_missing_tvds); 3252 EXPORT_SYMBOL(spa_state_to_name); 3253 EXPORT_SYMBOL(spa_importing_readonly_checkpoint); 3254 EXPORT_SYMBOL(spa_min_claim_txg); 3255 EXPORT_SYMBOL(spa_suspend_async_destroy); 3256 EXPORT_SYMBOL(spa_has_checkpoint); 3257 EXPORT_SYMBOL(spa_top_vdevs_spacemap_addressable); 3258 3259 ZFS_MODULE_PARAM(zfs, zfs_, flags, UINT, ZMOD_RW, 3260 "Set additional debugging flags"); 3261 3262 ZFS_MODULE_PARAM(zfs, zfs_, recover, INT, ZMOD_RW, 3263 "Set to attempt to recover from fatal errors"); 3264 3265 ZFS_MODULE_PARAM(zfs, zfs_, free_leak_on_eio, INT, ZMOD_RW, 3266 "Set to ignore IO errors during free and permanently leak the space"); 3267 3268 ZFS_MODULE_PARAM(zfs_deadman, zfs_deadman_, checktime_ms, U64, ZMOD_RW, 3269 "Dead I/O check interval in milliseconds"); 3270 3271 ZFS_MODULE_PARAM(zfs_deadman, zfs_deadman_, enabled, INT, ZMOD_RW, 3272 "Enable deadman timer"); 3273 3274 ZFS_MODULE_PARAM(zfs_spa, spa_, asize_inflation, UINT, ZMOD_RW, 3275 "SPA size estimate multiplication factor"); 3276 3277 ZFS_MODULE_PARAM(zfs, zfs_, ddt_data_is_special, INT, ZMOD_RW, 3278 "Place DDT data into the special class"); 3279 3280 ZFS_MODULE_PARAM(zfs, zfs_, user_indirect_is_special, INT, ZMOD_RW, 3281 "Place user data indirect blocks into the special class"); 3282 3283 ZFS_MODULE_PARAM_CALL(zfs_deadman, zfs_deadman_, failmode, 3284 param_set_deadman_failmode, param_get_charp, ZMOD_RW, 3285 "Failmode for deadman timer"); 3286 3287 ZFS_MODULE_PARAM_CALL(zfs_deadman, zfs_deadman_, synctime_ms, 3288 param_set_deadman_synctime, spl_param_get_u64, ZMOD_RW, 3289 "Pool sync expiration time in milliseconds"); 3290 3291 ZFS_MODULE_PARAM_CALL(zfs_deadman, zfs_deadman_, ziotime_ms, 3292 param_set_deadman_ziotime, spl_param_get_u64, ZMOD_RW, 3293 "IO expiration time in milliseconds"); 3294 3295 ZFS_MODULE_PARAM(zfs, zfs_, special_class_metadata_reserve_pct, UINT, ZMOD_RW, 3296 "Small file blocks in special vdevs depends on this much " 3297 "free space available"); 3298 3299 ZFS_MODULE_PARAM_CALL(zfs_spa, spa_, slop_shift, param_set_slop_shift, 3300 param_get_uint, ZMOD_RW, "Reserved free space in pool"); 3301 3302 ZFS_MODULE_PARAM(zfs, spa_, num_allocators, INT, ZMOD_RW, 3303 "Number of allocators per spa"); 3304 3305 ZFS_MODULE_PARAM(zfs, spa_, cpus_per_allocator, INT, ZMOD_RW, 3306 "Minimum number of CPUs per allocators"); 3307