1 // SPDX-License-Identifier: CDDL-1.0 2 /* 3 * This file and its contents are supplied under the terms of the 4 * Common Development and Distribution License ("CDDL"), version 1.0. 5 * You may only use this file in accordance with the terms of version 6 * 1.0 of the CDDL. 7 * 8 * A full copy of the text of the CDDL should have accompanied this 9 * source. A copy of the CDDL is also available via the Internet at 10 * https://opensource.org/license/CDDL-1.0. 11 */ 12 13 /* 14 * Copyright (c) 2005, 2010, Oracle and/or its affiliates. All rights reserved. 15 * Copyright (c) 2011, 2024 by Delphix. All rights reserved. 16 * Copyright (c) 2018, Nexenta Systems, Inc. All rights reserved. 17 * Copyright (c) 2014 Spectra Logic Corporation, All rights reserved. 18 * Copyright 2013 Saso Kiselkov. All rights reserved. 19 * Copyright (c) 2014 Integros [integros.com] 20 * Copyright 2016 Toomas Soome <tsoome@me.com> 21 * Copyright (c) 2016 Actifio, Inc. All rights reserved. 22 * Copyright 2018 Joyent, Inc. 23 * Copyright (c) 2017, 2019, Datto Inc. All rights reserved. 24 * Copyright 2017 Joyent, Inc. 25 * Copyright (c) 2017, Intel Corporation. 26 * Copyright (c) 2021, Colm Buckley <colm@tuatha.org> 27 * Copyright (c) 2023 Hewlett Packard Enterprise Development LP. 28 * Copyright (c) 2023-2026, Klara, Inc. 29 * Copyright (c) 2026, TrueNAS. 30 * Copyright 2026 Edgecast Cloud LLC. 31 */ 32 33 /* 34 * SPA: Storage Pool Allocator 35 * 36 * This file contains all the routines used when modifying on-disk SPA state. 37 * This includes opening, importing, destroying, exporting a pool, and syncing a 38 * pool. 39 */ 40 41 #include <sys/zfs_context.h> 42 #include <sys/fm/fs/zfs.h> 43 #include <sys/spa_impl.h> 44 #include <sys/zio.h> 45 #include <sys/zio_checksum.h> 46 #include <sys/dmu.h> 47 #include <sys/dmu_tx.h> 48 #include <sys/zap.h> 49 #include <sys/zil.h> 50 #include <sys/brt.h> 51 #include <sys/ddt.h> 52 #include <sys/vdev_impl.h> 53 #include <sys/vdev_removal.h> 54 #include <sys/vdev_indirect_mapping.h> 55 #include <sys/vdev_indirect_births.h> 56 #include <sys/vdev_initialize.h> 57 #include <sys/vdev_rebuild.h> 58 #include <sys/vdev_trim.h> 59 #include <sys/vdev_disk.h> 60 #include <sys/vdev_raidz.h> 61 #include <sys/vdev_draid.h> 62 #include <sys/metaslab.h> 63 #include <sys/metaslab_impl.h> 64 #include <sys/mmp.h> 65 #include <sys/uberblock_impl.h> 66 #include <sys/txg.h> 67 #include <sys/avl.h> 68 #include <sys/bpobj.h> 69 #include <sys/dmu_traverse.h> 70 #include <sys/dmu_objset.h> 71 #include <sys/unique.h> 72 #include <sys/dsl_pool.h> 73 #include <sys/dsl_dataset.h> 74 #include <sys/dsl_dir.h> 75 #include <sys/dsl_prop.h> 76 #include <sys/dsl_synctask.h> 77 #include <sys/fs/zfs.h> 78 #include <sys/arc.h> 79 #include <sys/callb.h> 80 #include <sys/systeminfo.h> 81 #include <sys/zfs_ioctl.h> 82 #include <sys/dsl_scan.h> 83 #include <sys/zfeature.h> 84 #include <sys/dsl_destroy.h> 85 #include <sys/zvol.h> 86 87 #ifdef _KERNEL 88 #include <sys/fm/protocol.h> 89 #include <sys/fm/util.h> 90 #include <sys/callb.h> 91 #include <sys/zone.h> 92 #include <sys/vmsystm.h> 93 #endif /* _KERNEL */ 94 95 #include "zfs_crrd.h" 96 #include "zfs_prop.h" 97 #include "zfs_comutil.h" 98 #include <cityhash.h> 99 100 /* 101 * spa_thread() existed on Illumos as a parent thread for the various worker 102 * threads that actually run the pool, as a way to both reference the entire 103 * pool work as a single object, and to share properties like scheduling 104 * options. It has not yet been adapted to Linux or FreeBSD. This define is 105 * used to mark related parts of the code to make things easier for the reader, 106 * and to compile this code out. It can be removed when someone implements it, 107 * moves it to some Illumos-specific place, or removes it entirely. 108 */ 109 #undef HAVE_SPA_THREAD 110 111 /* 112 * The "System Duty Cycle" scheduling class is an Illumos feature to help 113 * prevent CPU-intensive kernel threads from affecting latency on interactive 114 * threads. It doesn't exist on Linux or FreeBSD, so the supporting code is 115 * gated behind a define. On Illumos SDC depends on spa_thread(), but 116 * spa_thread() also has other uses, so this is a separate define. 117 */ 118 #undef HAVE_SYSDC 119 120 /* 121 * The interval, in seconds, at which failed configuration cache file writes 122 * should be retried. 123 */ 124 int zfs_ccw_retry_interval = 300; 125 126 typedef enum zti_modes { 127 ZTI_MODE_FIXED, /* value is # of threads (min 1) */ 128 ZTI_MODE_SCALE, /* Taskqs scale with CPUs. */ 129 ZTI_MODE_SYNC, /* sync thread assigned */ 130 ZTI_MODE_NULL, /* don't create a taskq */ 131 ZTI_NMODES 132 } zti_modes_t; 133 134 #define ZTI_P(n, q) { ZTI_MODE_FIXED, (n), (q) } 135 #define ZTI_PCT(n) { ZTI_MODE_ONLINE_PERCENT, (n), 1 } 136 #define ZTI_SCALE(min) { ZTI_MODE_SCALE, (min), 1 } 137 #define ZTI_SYNC { ZTI_MODE_SYNC, 0, 1 } 138 #define ZTI_NULL { ZTI_MODE_NULL, 0, 0 } 139 140 #define ZTI_N(n) ZTI_P(n, 1) 141 #define ZTI_ONE ZTI_N(1) 142 143 typedef struct zio_taskq_info { 144 zti_modes_t zti_mode; 145 uint_t zti_value; 146 uint_t zti_count; 147 } zio_taskq_info_t; 148 149 static const char *const zio_taskq_types[ZIO_TASKQ_TYPES] = { 150 "iss", "iss_h", "int", "int_h" 151 }; 152 153 /* 154 * This table defines the taskq settings for each ZFS I/O type. When 155 * initializing a pool, we use this table to create an appropriately sized 156 * taskq. Some operations are low volume and therefore have a small, static 157 * number of threads assigned to their taskqs using the ZTI_N(#) or ZTI_ONE 158 * macros. Other operations process a large amount of data; the ZTI_SCALE 159 * macro causes us to create a taskq oriented for throughput. Some operations 160 * are so high frequency and short-lived that the taskq itself can become a 161 * point of lock contention. The ZTI_P(#, #) macro indicates that we need an 162 * additional degree of parallelism specified by the number of threads per- 163 * taskq and the number of taskqs; when dispatching an event in this case, the 164 * particular taskq is chosen at random. ZTI_SCALE uses a number of taskqs 165 * that scales with the number of CPUs. 166 * 167 * The different taskq priorities are to handle the different contexts (issue 168 * and interrupt) and then to reserve threads for high priority I/Os that 169 * need to be handled with minimum delay. Illumos taskq has unfair TQ_FRONT 170 * implementation, so separate high priority threads are used there. 171 */ 172 static zio_taskq_info_t zio_taskqs[ZIO_TYPES][ZIO_TASKQ_TYPES] = { 173 /* ISSUE ISSUE_HIGH INTR INTR_HIGH */ 174 { ZTI_ONE, ZTI_NULL, ZTI_ONE, ZTI_NULL }, /* NULL */ 175 { ZTI_N(8), ZTI_NULL, ZTI_SCALE(0), ZTI_NULL }, /* READ */ 176 #ifdef illumos 177 { ZTI_SYNC, ZTI_N(5), ZTI_SCALE(0), ZTI_N(5) }, /* WRITE */ 178 #else 179 { ZTI_SYNC, ZTI_NULL, ZTI_SCALE(0), ZTI_NULL }, /* WRITE */ 180 #endif 181 { ZTI_SCALE(32), ZTI_NULL, ZTI_ONE, ZTI_NULL }, /* FREE */ 182 { ZTI_ONE, ZTI_NULL, ZTI_ONE, ZTI_NULL }, /* CLAIM */ 183 { ZTI_ONE, ZTI_NULL, ZTI_ONE, ZTI_NULL }, /* FLUSH */ 184 { ZTI_N(4), ZTI_NULL, ZTI_ONE, ZTI_NULL }, /* TRIM */ 185 }; 186 187 static void spa_sync_version(void *arg, dmu_tx_t *tx); 188 static void spa_sync_props(void *arg, dmu_tx_t *tx); 189 static boolean_t spa_has_active_shared_spare(spa_t *spa); 190 static int spa_load_impl(spa_t *spa, spa_import_type_t type, 191 const char **ereport); 192 static void spa_vdev_resilver_done(spa_t *spa); 193 194 /* 195 * Percentage of all CPUs that can be used by the metaslab preload taskq. 196 */ 197 static uint_t metaslab_preload_pct = 50; 198 199 static uint_t zio_taskq_batch_pct = 80; /* 1 thread per cpu in pset */ 200 static uint_t zio_taskq_batch_tpq; /* threads per taskq */ 201 202 #ifdef HAVE_SYSDC 203 static const boolean_t zio_taskq_sysdc = B_TRUE; /* use SDC scheduling class */ 204 static const uint_t zio_taskq_basedc = 80; /* base duty cycle */ 205 #endif 206 207 #ifdef HAVE_SPA_THREAD 208 static const boolean_t spa_create_process = B_TRUE; /* no process => no sysdc */ 209 #endif 210 211 static uint_t zio_taskq_write_tpq = 16; 212 213 /* 214 * Report any spa_load_verify errors found, but do not fail spa_load. 215 * This is used by zdb to analyze non-idle pools. 216 */ 217 boolean_t spa_load_verify_dryrun = B_FALSE; 218 219 /* 220 * Allow read spacemaps in case of readonly import (spa_mode == SPA_MODE_READ). 221 * This is used by zdb for spacemaps verification. 222 */ 223 boolean_t spa_mode_readable_spacemaps = B_FALSE; 224 225 /* 226 * This (illegal) pool name is used when temporarily importing a spa_t in order 227 * to get the vdev stats associated with the imported devices. 228 */ 229 #define TRYIMPORT_NAME "$import" 230 231 /* 232 * For debugging purposes: print out vdev tree during pool import. 233 */ 234 static int spa_load_print_vdev_tree = B_FALSE; 235 236 /* 237 * A non-zero value for zfs_max_missing_tvds means that we allow importing 238 * pools with missing top-level vdevs. This is strictly intended for advanced 239 * pool recovery cases since missing data is almost inevitable. Pools with 240 * missing devices can only be imported read-only for safety reasons, and their 241 * fail-mode will be automatically set to "continue". 242 * 243 * With 1 missing vdev we should be able to import the pool and mount all 244 * datasets. User data that was not modified after the missing device has been 245 * added should be recoverable. This means that snapshots created prior to the 246 * addition of that device should be completely intact. 247 * 248 * With 2 missing vdevs, some datasets may fail to mount since there are 249 * dataset statistics that are stored as regular metadata. Some data might be 250 * recoverable if those vdevs were added recently. 251 * 252 * With 3 or more missing vdevs, the pool is severely damaged and MOS entries 253 * may be missing entirely. Chances of data recovery are very low. Note that 254 * there are also risks of performing an inadvertent rewind as we might be 255 * missing all the vdevs with the latest uberblocks. 256 */ 257 uint64_t zfs_max_missing_tvds = 0; 258 259 /* 260 * The parameters below are similar to zfs_max_missing_tvds but are only 261 * intended for a preliminary open of the pool with an untrusted config which 262 * might be incomplete or out-dated. 263 * 264 * We are more tolerant for pools opened from a cachefile since we could have 265 * an out-dated cachefile where a device removal was not registered. 266 * We could have set the limit arbitrarily high but in the case where devices 267 * are really missing we would want to return the proper error codes; we chose 268 * SPA_DVAS_PER_BP - 1 so that some copies of the MOS would still be available 269 * and we get a chance to retrieve the trusted config. 270 */ 271 uint64_t zfs_max_missing_tvds_cachefile = SPA_DVAS_PER_BP - 1; 272 273 /* 274 * In the case where config was assembled by scanning device paths (/dev/dsks 275 * by default) we are less tolerant since all the existing devices should have 276 * been detected and we want spa_load to return the right error codes. 277 */ 278 uint64_t zfs_max_missing_tvds_scan = 0; 279 280 /* 281 * Debugging aid that pauses spa_sync() towards the end. 282 */ 283 static const boolean_t zfs_pause_spa_sync = B_FALSE; 284 285 /* 286 * Variables to indicate the livelist condense zthr func should wait at certain 287 * points for the livelist to be removed - used to test condense/destroy races 288 */ 289 static int zfs_livelist_condense_zthr_pause = 0; 290 static int zfs_livelist_condense_sync_pause = 0; 291 292 /* 293 * Variables to track whether or not condense cancellation has been 294 * triggered in testing. 295 */ 296 static int zfs_livelist_condense_sync_cancel = 0; 297 static int zfs_livelist_condense_zthr_cancel = 0; 298 299 /* 300 * Variable to track whether or not extra ALLOC blkptrs were added to a 301 * livelist entry while it was being condensed (caused by the way we track 302 * remapped blkptrs in dbuf_remap_impl) 303 */ 304 static int zfs_livelist_condense_new_alloc = 0; 305 306 /* 307 * Time variable to decide how often the txg should be added into the 308 * database (in seconds). 309 * The smallest available resolution is in minutes, which means an update occurs 310 * each time we reach `spa_note_txg_time` and the txg has changed. We provide 311 * a 256-slot ring buffer for minute-level resolution. The number is limited by 312 * the size of the structure we use and the maximum amount of bytes we can write 313 * into ZAP. Setting `spa_note_txg_time` to 10 minutes results in approximately 314 * 144 records per day. Given the 256 slots, this provides roughly 1.5 days of 315 * high-resolution data. 316 * 317 * The user can decrease `spa_note_txg_time` to increase resolution within 318 * a day, at the cost of retaining fewer days of data. Alternatively, increasing 319 * the interval allows storing data over a longer period, but with lower 320 * frequency. 321 * 322 * This parameter does not affect the daily or monthly databases, as those only 323 * store one record per day and per month, respectively. 324 */ 325 static uint_t spa_note_txg_time = 10 * 60; 326 327 /* 328 * How often flush txg database to a disk (in seconds). 329 * We flush data every time we write to it, making it the most reliable option. 330 * Since this happens every 10 minutes, it shouldn't introduce any noticeable 331 * overhead for the system. In case of failure, we will always have an 332 * up-to-date version of the database. 333 * 334 * The user can adjust the flush interval to a lower value, but it probably 335 * doesn't make sense to flush more often than the database is updated. 336 * The user can also increase the interval if they're concerned about the 337 * performance of writing the entire database to disk. 338 */ 339 static uint_t spa_flush_txg_time = 10 * 60; 340 341 /* 342 * ========================================================================== 343 * SPA properties routines 344 * ========================================================================== 345 */ 346 347 /* 348 * Add a (source=src, propname=propval) list to an nvlist. 349 */ 350 static void 351 spa_prop_add_list(nvlist_t *nvl, zpool_prop_t prop, const char *strval, 352 uint64_t intval, zprop_source_t src) 353 { 354 const char *propname = zpool_prop_to_name(prop); 355 nvlist_t *propval; 356 357 propval = fnvlist_alloc(); 358 fnvlist_add_uint64(propval, ZPROP_SOURCE, src); 359 360 if (strval != NULL) 361 fnvlist_add_string(propval, ZPROP_VALUE, strval); 362 else 363 fnvlist_add_uint64(propval, ZPROP_VALUE, intval); 364 365 fnvlist_add_nvlist(nvl, propname, propval); 366 nvlist_free(propval); 367 } 368 369 static int 370 spa_prop_add(spa_t *spa, const char *propname, nvlist_t *outnvl) 371 { 372 zpool_prop_t prop = zpool_name_to_prop(propname); 373 zprop_source_t src = ZPROP_SRC_NONE; 374 uint64_t intval; 375 int err; 376 377 /* 378 * NB: Not all properties lookups via this API require 379 * the spa props lock, so they must explicitly grab it here. 380 */ 381 switch (prop) { 382 case ZPOOL_PROP_DEDUPCACHED: 383 err = ddt_get_pool_dedup_cached(spa, &intval); 384 if (err != 0) 385 return (SET_ERROR(err)); 386 break; 387 default: 388 return (SET_ERROR(EINVAL)); 389 } 390 391 spa_prop_add_list(outnvl, prop, NULL, intval, src); 392 393 return (0); 394 } 395 396 int 397 spa_prop_get_nvlist(spa_t *spa, char **props, unsigned int n_props, 398 nvlist_t *outnvl) 399 { 400 int err = 0; 401 402 if (props == NULL) 403 return (0); 404 405 for (unsigned int i = 0; i < n_props && err == 0; i++) { 406 err = spa_prop_add(spa, props[i], outnvl); 407 } 408 409 return (err); 410 } 411 412 /* 413 * Add metaslab class properties to an nvlist. 414 */ 415 static void 416 spa_prop_add_metaslab_class(nvlist_t *nv, metaslab_class_t *mc, 417 zpool_mc_props_t mcp, uint64_t *sizep, uint64_t *allocp, uint64_t *usablep, 418 uint64_t *usedp) 419 { 420 uint64_t size = metaslab_class_get_space(mc); 421 uint64_t alloc = metaslab_class_get_alloc(mc); 422 uint64_t dsize = metaslab_class_get_dspace(mc); 423 uint64_t dalloc = metaslab_class_get_dalloc(mc); 424 uint64_t cap = (size == 0) ? 0 : (alloc * 100 / size); 425 const zprop_source_t src = ZPROP_SRC_NONE; 426 427 spa_prop_add_list(nv, mcp + ZPOOL_MC_PROP_SIZE, NULL, size, src); 428 spa_prop_add_list(nv, mcp + ZPOOL_MC_PROP_ALLOCATED, NULL, alloc, src); 429 spa_prop_add_list(nv, mcp + ZPOOL_MC_PROP_USABLE, NULL, dsize, src); 430 spa_prop_add_list(nv, mcp + ZPOOL_MC_PROP_USED, NULL, dalloc, src); 431 spa_prop_add_list(nv, mcp + ZPOOL_MC_PROP_FRAGMENTATION, NULL, 432 metaslab_class_fragmentation(mc), src); 433 spa_prop_add_list(nv, mcp + ZPOOL_MC_PROP_EXPANDSZ, NULL, 434 metaslab_class_expandable_space(mc), src); 435 spa_prop_add_list(nv, mcp + ZPOOL_MC_PROP_FREE, NULL, size - alloc, 436 src); 437 spa_prop_add_list(nv, mcp + ZPOOL_MC_PROP_AVAILABLE, NULL, 438 dsize - dalloc, src); 439 spa_prop_add_list(nv, mcp + ZPOOL_MC_PROP_CAPACITY, NULL, cap, src); 440 if (sizep != NULL) 441 *sizep += size; 442 if (allocp != NULL) 443 *allocp += alloc; 444 if (usablep != NULL) 445 *usablep += dsize; 446 if (usedp != NULL) 447 *usedp += dalloc; 448 } 449 450 /* 451 * Add a user property (source=src, propname=propval) to an nvlist. 452 */ 453 static void 454 spa_prop_add_user(nvlist_t *nvl, const char *propname, char *strval, 455 zprop_source_t src) 456 { 457 nvlist_t *propval; 458 459 VERIFY0(nvlist_alloc(&propval, NV_UNIQUE_NAME, KM_SLEEP)); 460 VERIFY0(nvlist_add_uint64(propval, ZPROP_SOURCE, src)); 461 VERIFY0(nvlist_add_string(propval, ZPROP_VALUE, strval)); 462 VERIFY0(nvlist_add_nvlist(nvl, propname, propval)); 463 nvlist_free(propval); 464 } 465 466 /* 467 * Get property values from the spa configuration. 468 */ 469 static void 470 spa_prop_get_config(spa_t *spa, nvlist_t *nv) 471 { 472 vdev_t *rvd = spa->spa_root_vdev; 473 dsl_pool_t *pool = spa->spa_dsl_pool; 474 uint64_t size, alloc, usable, used, cap, version; 475 const zprop_source_t src = ZPROP_SRC_NONE; 476 spa_config_dirent_t *dp; 477 metaslab_class_t *mc = spa_normal_class(spa); 478 479 ASSERT(MUTEX_HELD(&spa->spa_props_lock)); 480 481 if (rvd != NULL) { 482 spa_prop_add_list(nv, ZPOOL_PROP_NAME, spa_name(spa), 0, src); 483 484 size = alloc = usable = used = 0; 485 spa_prop_add_metaslab_class(nv, mc, ZPOOL_MC_PROPS_NORMAL, 486 &size, &alloc, &usable, &used); 487 spa_prop_add_metaslab_class(nv, spa_special_class(spa), 488 ZPOOL_MC_PROPS_SPECIAL, &size, &alloc, &usable, &used); 489 spa_prop_add_metaslab_class(nv, spa_dedup_class(spa), 490 ZPOOL_MC_PROPS_DEDUP, &size, &alloc, &usable, &used); 491 spa_prop_add_metaslab_class(nv, spa_log_class(spa), 492 ZPOOL_MC_PROPS_LOG, NULL, NULL, NULL, NULL); 493 spa_prop_add_metaslab_class(nv, spa_embedded_log_class(spa), 494 ZPOOL_MC_PROPS_ELOG, &size, &alloc, &usable, &used); 495 spa_prop_add_metaslab_class(nv, 496 spa_special_embedded_log_class(spa), ZPOOL_MC_PROPS_SELOG, 497 &size, &alloc, &usable, &used); 498 499 spa_prop_add_list(nv, ZPOOL_PROP_SIZE, NULL, size, src); 500 spa_prop_add_list(nv, ZPOOL_PROP_ALLOCATED, NULL, alloc, src); 501 spa_prop_add_list(nv, ZPOOL_PROP_FREE, NULL, 502 size - alloc, src); 503 spa_prop_add_list(nv, ZPOOL_PROP_FRAGMENTATION, NULL, 504 metaslab_class_fragmentation(mc), src); 505 spa_prop_add_list(nv, ZPOOL_PROP_EXPANDSZ, NULL, 506 metaslab_class_expandable_space(mc), src); 507 cap = (size == 0) ? 0 : (alloc * 100 / size); 508 spa_prop_add_list(nv, ZPOOL_PROP_CAPACITY, NULL, cap, src); 509 spa_prop_add_list(nv, ZPOOL_PROP_AVAILABLE, NULL, usable - used, 510 src); 511 spa_prop_add_list(nv, ZPOOL_PROP_USABLE, NULL, usable, src); 512 spa_prop_add_list(nv, ZPOOL_PROP_USED, NULL, used, src); 513 514 spa_prop_add_list(nv, ZPOOL_PROP_CHECKPOINT, NULL, 515 spa->spa_checkpoint_info.sci_dspace, src); 516 spa_prop_add_list(nv, ZPOOL_PROP_READONLY, NULL, 517 (spa_mode(spa) == SPA_MODE_READ), src); 518 519 spa_prop_add_list(nv, ZPOOL_PROP_DEDUPRATIO, NULL, 520 ddt_get_pool_dedup_ratio(spa), src); 521 spa_prop_add_list(nv, ZPOOL_PROP_DEDUPUSED, NULL, 522 ddt_get_dedup_used(spa), src); 523 spa_prop_add_list(nv, ZPOOL_PROP_DEDUPSAVED, NULL, 524 ddt_get_dedup_saved(spa), src); 525 spa_prop_add_list(nv, ZPOOL_PROP_BCLONEUSED, NULL, 526 brt_get_used(spa), src); 527 spa_prop_add_list(nv, ZPOOL_PROP_BCLONESAVED, NULL, 528 brt_get_saved(spa), src); 529 spa_prop_add_list(nv, ZPOOL_PROP_BCLONERATIO, NULL, 530 brt_get_ratio(spa), src); 531 532 spa_prop_add_list(nv, ZPOOL_PROP_DEDUP_TABLE_SIZE, NULL, 533 ddt_get_ddt_dsize(spa), src); 534 spa_prop_add_list(nv, ZPOOL_PROP_HEALTH, NULL, 535 rvd->vdev_state, src); 536 spa_prop_add_list(nv, ZPOOL_PROP_LAST_SCRUBBED_TXG, NULL, 537 spa_get_last_scrubbed_txg(spa), src); 538 539 version = spa_version(spa); 540 if (version == zpool_prop_default_numeric(ZPOOL_PROP_VERSION)) { 541 spa_prop_add_list(nv, ZPOOL_PROP_VERSION, NULL, 542 version, ZPROP_SRC_DEFAULT); 543 } else { 544 spa_prop_add_list(nv, ZPOOL_PROP_VERSION, NULL, 545 version, ZPROP_SRC_LOCAL); 546 } 547 spa_prop_add_list(nv, ZPOOL_PROP_LOAD_GUID, 548 NULL, spa_load_guid(spa), src); 549 } 550 551 if (pool != NULL) { 552 /* 553 * The $FREE directory was introduced in SPA_VERSION_DEADLISTS, 554 * when opening pools before this version freedir will be NULL. 555 */ 556 if (pool->dp_free_dir != NULL) { 557 spa_prop_add_list(nv, ZPOOL_PROP_FREEING, NULL, 558 dsl_dir_phys(pool->dp_free_dir)->dd_used_bytes, 559 src); 560 } else { 561 spa_prop_add_list(nv, ZPOOL_PROP_FREEING, 562 NULL, 0, src); 563 } 564 565 if (pool->dp_leak_dir != NULL) { 566 spa_prop_add_list(nv, ZPOOL_PROP_LEAKED, NULL, 567 dsl_dir_phys(pool->dp_leak_dir)->dd_used_bytes, 568 src); 569 } else { 570 spa_prop_add_list(nv, ZPOOL_PROP_LEAKED, 571 NULL, 0, src); 572 } 573 } 574 575 spa_prop_add_list(nv, ZPOOL_PROP_GUID, NULL, spa_guid(spa), src); 576 577 if (spa->spa_comment != NULL) { 578 spa_prop_add_list(nv, ZPOOL_PROP_COMMENT, spa->spa_comment, 579 0, ZPROP_SRC_LOCAL); 580 } 581 582 if (spa->spa_compatibility != NULL) { 583 spa_prop_add_list(nv, ZPOOL_PROP_COMPATIBILITY, 584 spa->spa_compatibility, 0, ZPROP_SRC_LOCAL); 585 } 586 587 if (spa->spa_root != NULL) 588 spa_prop_add_list(nv, ZPOOL_PROP_ALTROOT, spa->spa_root, 589 0, ZPROP_SRC_LOCAL); 590 591 if (spa_feature_is_enabled(spa, SPA_FEATURE_LARGE_BLOCKS)) { 592 spa_prop_add_list(nv, ZPOOL_PROP_MAXBLOCKSIZE, NULL, 593 MIN(zfs_max_recordsize, SPA_MAXBLOCKSIZE), ZPROP_SRC_NONE); 594 } else { 595 spa_prop_add_list(nv, ZPOOL_PROP_MAXBLOCKSIZE, NULL, 596 SPA_OLD_MAXBLOCKSIZE, ZPROP_SRC_NONE); 597 } 598 599 if (spa_feature_is_enabled(spa, SPA_FEATURE_LARGE_DNODE)) { 600 spa_prop_add_list(nv, ZPOOL_PROP_MAXDNODESIZE, NULL, 601 DNODE_MAX_SIZE, ZPROP_SRC_NONE); 602 } else { 603 spa_prop_add_list(nv, ZPOOL_PROP_MAXDNODESIZE, NULL, 604 DNODE_MIN_SIZE, ZPROP_SRC_NONE); 605 } 606 607 if ((dp = list_head(&spa->spa_config_list)) != NULL) { 608 if (dp->scd_path == NULL) { 609 spa_prop_add_list(nv, ZPOOL_PROP_CACHEFILE, 610 "none", 0, ZPROP_SRC_LOCAL); 611 } else if (strcmp(dp->scd_path, spa_config_path) != 0) { 612 spa_prop_add_list(nv, ZPOOL_PROP_CACHEFILE, 613 dp->scd_path, 0, ZPROP_SRC_LOCAL); 614 } 615 } 616 } 617 618 /* 619 * Get zpool property values. 620 */ 621 int 622 spa_prop_get(spa_t *spa, nvlist_t *nv) 623 { 624 objset_t *mos = spa->spa_meta_objset; 625 zap_cursor_t zc; 626 zap_attribute_t *za; 627 dsl_pool_t *dp; 628 int err = 0; 629 630 dp = spa_get_dsl(spa); 631 dsl_pool_config_enter(dp, FTAG); 632 za = zap_attribute_alloc(); 633 mutex_enter(&spa->spa_props_lock); 634 635 /* 636 * Get properties from the spa config. 637 */ 638 spa_prop_get_config(spa, nv); 639 640 /* If no pool property object, no more prop to get. */ 641 if (mos == NULL || spa->spa_pool_props_object == 0) 642 goto out; 643 644 /* 645 * Get properties from the MOS pool property object. 646 */ 647 for (zap_cursor_init(&zc, mos, spa->spa_pool_props_object); 648 (err = zap_cursor_retrieve(&zc, za)) == 0; 649 zap_cursor_advance(&zc)) { 650 uint64_t intval = 0; 651 char *strval = NULL; 652 zprop_source_t src = ZPROP_SRC_DEFAULT; 653 zpool_prop_t prop; 654 655 if ((prop = zpool_name_to_prop(za->za_name)) == 656 ZPOOL_PROP_INVAL && !zfs_prop_user(za->za_name)) 657 continue; 658 659 switch (za->za_integer_length) { 660 case 8: 661 /* integer property */ 662 if (za->za_first_integer != 663 zpool_prop_default_numeric(prop)) 664 src = ZPROP_SRC_LOCAL; 665 666 if (prop == ZPOOL_PROP_BOOTFS) { 667 dsl_dataset_t *ds = NULL; 668 669 err = dsl_dataset_hold_obj(dp, 670 za->za_first_integer, FTAG, &ds); 671 if (err != 0) 672 break; 673 674 strval = kmem_alloc(ZFS_MAX_DATASET_NAME_LEN, 675 KM_SLEEP); 676 dsl_dataset_name(ds, strval); 677 dsl_dataset_rele(ds, FTAG); 678 } else { 679 strval = NULL; 680 intval = za->za_first_integer; 681 } 682 683 spa_prop_add_list(nv, prop, strval, intval, src); 684 685 if (strval != NULL) 686 kmem_free(strval, ZFS_MAX_DATASET_NAME_LEN); 687 688 break; 689 690 case 1: 691 /* string property */ 692 strval = kmem_alloc(za->za_num_integers, KM_SLEEP); 693 err = zap_lookup(mos, spa->spa_pool_props_object, 694 za->za_name, 1, za->za_num_integers, strval); 695 if (err) { 696 kmem_free(strval, za->za_num_integers); 697 break; 698 } 699 if (prop != ZPOOL_PROP_INVAL) { 700 spa_prop_add_list(nv, prop, strval, 0, src); 701 } else { 702 src = ZPROP_SRC_LOCAL; 703 spa_prop_add_user(nv, za->za_name, strval, 704 src); 705 } 706 kmem_free(strval, za->za_num_integers); 707 break; 708 709 default: 710 break; 711 } 712 } 713 zap_cursor_fini(&zc); 714 out: 715 mutex_exit(&spa->spa_props_lock); 716 dsl_pool_config_exit(dp, FTAG); 717 zap_attribute_free(za); 718 719 if (err && err != ENOENT) 720 return (err); 721 722 return (0); 723 } 724 725 /* 726 * Validate the given pool properties nvlist and modify the list 727 * for the property values to be set. 728 */ 729 static int 730 spa_prop_validate(spa_t *spa, nvlist_t *props) 731 { 732 nvpair_t *elem; 733 int error = 0, reset_bootfs = 0; 734 uint64_t objnum = 0; 735 boolean_t has_feature = B_FALSE; 736 737 elem = NULL; 738 while ((elem = nvlist_next_nvpair(props, elem)) != NULL) { 739 uint64_t intval; 740 const char *strval, *slash, *check, *fname; 741 const char *propname = nvpair_name(elem); 742 zpool_prop_t prop = zpool_name_to_prop(propname); 743 744 switch (prop) { 745 case ZPOOL_PROP_INVAL: 746 /* 747 * Sanitize the input. 748 */ 749 if (zfs_prop_user(propname)) { 750 if (strlen(propname) >= ZAP_MAXNAMELEN) { 751 error = SET_ERROR(ENAMETOOLONG); 752 break; 753 } 754 755 if (strlen(fnvpair_value_string(elem)) >= 756 ZAP_MAXVALUELEN) { 757 error = SET_ERROR(E2BIG); 758 break; 759 } 760 } else if (zpool_prop_feature(propname)) { 761 if (nvpair_type(elem) != DATA_TYPE_UINT64) { 762 error = SET_ERROR(EINVAL); 763 break; 764 } 765 766 if (nvpair_value_uint64(elem, &intval) != 0) { 767 error = SET_ERROR(EINVAL); 768 break; 769 } 770 771 if (intval != 0) { 772 error = SET_ERROR(EINVAL); 773 break; 774 } 775 776 fname = strchr(propname, '@') + 1; 777 if (zfeature_lookup_name(fname, NULL) != 0) { 778 error = SET_ERROR(EINVAL); 779 break; 780 } 781 782 has_feature = B_TRUE; 783 } else { 784 error = SET_ERROR(EINVAL); 785 break; 786 } 787 break; 788 789 case ZPOOL_PROP_VERSION: 790 error = nvpair_value_uint64(elem, &intval); 791 if (!error && 792 (intval < spa_version(spa) || 793 intval > SPA_VERSION_BEFORE_FEATURES || 794 has_feature)) 795 error = SET_ERROR(EINVAL); 796 break; 797 798 case ZPOOL_PROP_DEDUP_TABLE_QUOTA: 799 error = nvpair_value_uint64(elem, &intval); 800 break; 801 802 case ZPOOL_PROP_DELEGATION: 803 case ZPOOL_PROP_AUTOREPLACE: 804 case ZPOOL_PROP_LISTSNAPS: 805 case ZPOOL_PROP_AUTOEXPAND: 806 case ZPOOL_PROP_AUTOTRIM: 807 error = nvpair_value_uint64(elem, &intval); 808 if (!error && intval > 1) 809 error = SET_ERROR(EINVAL); 810 break; 811 812 case ZPOOL_PROP_MULTIHOST: 813 error = nvpair_value_uint64(elem, &intval); 814 if (!error && intval > 1) 815 error = SET_ERROR(EINVAL); 816 817 if (!error) { 818 uint32_t hostid = zone_get_hostid(NULL); 819 if (hostid) 820 spa->spa_hostid = hostid; 821 else 822 error = SET_ERROR(ENOTSUP); 823 } 824 825 break; 826 827 case ZPOOL_PROP_BOOTFS: 828 /* 829 * If the pool version is less than SPA_VERSION_BOOTFS, 830 * or the pool is still being created (version == 0), 831 * the bootfs property cannot be set. 832 */ 833 if (spa_version(spa) < SPA_VERSION_BOOTFS) { 834 error = SET_ERROR(ENOTSUP); 835 break; 836 } 837 838 /* 839 * Make sure the vdev config is bootable 840 */ 841 if (!vdev_is_bootable(spa->spa_root_vdev)) { 842 error = SET_ERROR(ENOTSUP); 843 break; 844 } 845 846 reset_bootfs = 1; 847 848 error = nvpair_value_string(elem, &strval); 849 850 if (!error) { 851 objset_t *os; 852 853 if (strval == NULL || strval[0] == '\0') { 854 objnum = zpool_prop_default_numeric( 855 ZPOOL_PROP_BOOTFS); 856 break; 857 } 858 859 error = dmu_objset_hold(strval, FTAG, &os); 860 if (error != 0) 861 break; 862 863 /* Must be ZPL. */ 864 if (dmu_objset_type(os) != DMU_OST_ZFS) { 865 error = SET_ERROR(ENOTSUP); 866 } else { 867 objnum = dmu_objset_id(os); 868 } 869 dmu_objset_rele(os, FTAG); 870 } 871 break; 872 873 case ZPOOL_PROP_FAILUREMODE: 874 error = nvpair_value_uint64(elem, &intval); 875 if (!error && intval > ZIO_FAILURE_MODE_PANIC) 876 error = SET_ERROR(EINVAL); 877 878 /* 879 * This is a special case which only occurs when 880 * the pool has completely failed. This allows 881 * the user to change the in-core failmode property 882 * without syncing it out to disk (I/Os might 883 * currently be blocked). We do this by returning 884 * EIO to the caller (spa_prop_set) to trick it 885 * into thinking we encountered a property validation 886 * error. 887 */ 888 if (!error && spa_suspended(spa)) { 889 spa->spa_failmode = intval; 890 error = SET_ERROR(EIO); 891 } 892 break; 893 894 case ZPOOL_PROP_CACHEFILE: 895 if ((error = nvpair_value_string(elem, &strval)) != 0) 896 break; 897 898 if (strval[0] == '\0') 899 break; 900 901 if (strcmp(strval, "none") == 0) 902 break; 903 904 if (strval[0] != '/') { 905 error = SET_ERROR(EINVAL); 906 break; 907 } 908 909 slash = strrchr(strval, '/'); 910 ASSERT(slash != NULL); 911 912 if (slash[1] == '\0' || strcmp(slash, "/.") == 0 || 913 strcmp(slash, "/..") == 0) 914 error = SET_ERROR(EINVAL); 915 break; 916 917 case ZPOOL_PROP_COMMENT: 918 if ((error = nvpair_value_string(elem, &strval)) != 0) 919 break; 920 for (check = strval; *check != '\0'; check++) { 921 if (!isprint(*check)) { 922 error = SET_ERROR(EINVAL); 923 break; 924 } 925 } 926 if (strlen(strval) > ZPROP_MAX_COMMENT) 927 error = SET_ERROR(E2BIG); 928 break; 929 930 default: 931 break; 932 } 933 934 if (error) 935 break; 936 } 937 938 (void) nvlist_remove_all(props, 939 zpool_prop_to_name(ZPOOL_PROP_DEDUPDITTO)); 940 941 if (!error && reset_bootfs) { 942 error = nvlist_remove(props, 943 zpool_prop_to_name(ZPOOL_PROP_BOOTFS), DATA_TYPE_STRING); 944 945 if (!error) { 946 error = nvlist_add_uint64(props, 947 zpool_prop_to_name(ZPOOL_PROP_BOOTFS), objnum); 948 } 949 } 950 951 return (error); 952 } 953 954 void 955 spa_configfile_set(spa_t *spa, nvlist_t *nvp, boolean_t need_sync) 956 { 957 const char *cachefile; 958 spa_config_dirent_t *dp; 959 960 if (nvlist_lookup_string(nvp, zpool_prop_to_name(ZPOOL_PROP_CACHEFILE), 961 &cachefile) != 0) 962 return; 963 964 dp = kmem_alloc(sizeof (spa_config_dirent_t), 965 KM_SLEEP); 966 967 if (cachefile[0] == '\0') 968 dp->scd_path = spa_strdup(spa_config_path); 969 else if (strcmp(cachefile, "none") == 0) 970 dp->scd_path = NULL; 971 else 972 dp->scd_path = spa_strdup(cachefile); 973 974 list_insert_head(&spa->spa_config_list, dp); 975 if (need_sync) 976 spa_async_request(spa, SPA_ASYNC_CONFIG_UPDATE); 977 } 978 979 int 980 spa_prop_set(spa_t *spa, nvlist_t *nvp) 981 { 982 int error; 983 nvpair_t *elem = NULL; 984 boolean_t need_sync = B_FALSE; 985 986 if ((error = spa_prop_validate(spa, nvp)) != 0) 987 return (error); 988 989 while ((elem = nvlist_next_nvpair(nvp, elem)) != NULL) { 990 zpool_prop_t prop = zpool_name_to_prop(nvpair_name(elem)); 991 992 if (prop == ZPOOL_PROP_CACHEFILE || 993 prop == ZPOOL_PROP_ALTROOT || 994 prop == ZPOOL_PROP_READONLY) 995 continue; 996 997 if (prop == ZPOOL_PROP_INVAL && 998 zfs_prop_user(nvpair_name(elem))) { 999 need_sync = B_TRUE; 1000 break; 1001 } 1002 1003 if (prop == ZPOOL_PROP_VERSION || prop == ZPOOL_PROP_INVAL) { 1004 uint64_t ver = 0; 1005 1006 if (prop == ZPOOL_PROP_VERSION) { 1007 VERIFY0(nvpair_value_uint64(elem, &ver)); 1008 } else { 1009 ASSERT(zpool_prop_feature(nvpair_name(elem))); 1010 ver = SPA_VERSION_FEATURES; 1011 need_sync = B_TRUE; 1012 } 1013 1014 /* Save time if the version is already set. */ 1015 if (ver == spa_version(spa)) 1016 continue; 1017 1018 /* 1019 * In addition to the pool directory object, we might 1020 * create the pool properties object, the features for 1021 * read object, the features for write object, or the 1022 * feature descriptions object. 1023 */ 1024 error = dsl_sync_task(spa->spa_name, NULL, 1025 spa_sync_version, &ver, 1026 6, ZFS_SPACE_CHECK_RESERVED); 1027 if (error) 1028 return (error); 1029 continue; 1030 } 1031 1032 need_sync = B_TRUE; 1033 break; 1034 } 1035 1036 if (need_sync) { 1037 return (dsl_sync_task(spa->spa_name, NULL, spa_sync_props, 1038 nvp, 6, ZFS_SPACE_CHECK_RESERVED)); 1039 } 1040 1041 return (0); 1042 } 1043 1044 /* 1045 * If the bootfs property value is dsobj, clear it. 1046 */ 1047 void 1048 spa_prop_clear_bootfs(spa_t *spa, uint64_t dsobj, dmu_tx_t *tx) 1049 { 1050 if (spa->spa_bootfs == dsobj && spa->spa_pool_props_object != 0) { 1051 VERIFY(zap_remove(spa->spa_meta_objset, 1052 spa->spa_pool_props_object, 1053 zpool_prop_to_name(ZPOOL_PROP_BOOTFS), tx) == 0); 1054 spa->spa_bootfs = 0; 1055 } 1056 } 1057 1058 static int 1059 spa_change_guid_check(void *arg, dmu_tx_t *tx) 1060 { 1061 uint64_t *newguid __maybe_unused = arg; 1062 spa_t *spa = dmu_tx_pool(tx)->dp_spa; 1063 vdev_t *rvd = spa->spa_root_vdev; 1064 uint64_t vdev_state; 1065 1066 if (spa_feature_is_active(spa, SPA_FEATURE_POOL_CHECKPOINT)) { 1067 int error = (spa_has_checkpoint(spa)) ? 1068 ZFS_ERR_CHECKPOINT_EXISTS : ZFS_ERR_DISCARDING_CHECKPOINT; 1069 return (SET_ERROR(error)); 1070 } 1071 1072 spa_config_enter(spa, SCL_STATE, FTAG, RW_READER); 1073 vdev_state = rvd->vdev_state; 1074 spa_config_exit(spa, SCL_STATE, FTAG); 1075 1076 if (vdev_state != VDEV_STATE_HEALTHY) 1077 return (SET_ERROR(ENXIO)); 1078 1079 ASSERT3U(spa_guid(spa), !=, *newguid); 1080 1081 return (0); 1082 } 1083 1084 static void 1085 spa_change_guid_sync(void *arg, dmu_tx_t *tx) 1086 { 1087 uint64_t *newguid = arg; 1088 spa_t *spa = dmu_tx_pool(tx)->dp_spa; 1089 uint64_t oldguid; 1090 vdev_t *rvd = spa->spa_root_vdev; 1091 1092 oldguid = spa_guid(spa); 1093 1094 spa_config_enter(spa, SCL_STATE, FTAG, RW_READER); 1095 rvd->vdev_guid = *newguid; 1096 rvd->vdev_guid_sum += (*newguid - oldguid); 1097 vdev_config_dirty(rvd); 1098 spa_config_exit(spa, SCL_STATE, FTAG); 1099 1100 spa_history_log_internal(spa, "guid change", tx, "old=%llu new=%llu", 1101 (u_longlong_t)oldguid, (u_longlong_t)*newguid); 1102 } 1103 1104 /* 1105 * Change the GUID for the pool. This is done so that we can later 1106 * re-import a pool built from a clone of our own vdevs. We will modify 1107 * the root vdev's guid, our own pool guid, and then mark all of our 1108 * vdevs dirty. Note that we must make sure that all our vdevs are 1109 * online when we do this, or else any vdevs that weren't present 1110 * would be orphaned from our pool. We are also going to issue a 1111 * sysevent to update any watchers. 1112 * 1113 * The GUID of the pool will be changed to the value pointed to by guidp. 1114 * The GUID may not be set to the reserverd value of 0. 1115 * The new GUID will be generated if guidp is NULL. 1116 */ 1117 int 1118 spa_change_guid(spa_t *spa, const uint64_t *guidp) 1119 { 1120 uint64_t guid; 1121 int error; 1122 1123 mutex_enter(&spa->spa_vdev_top_lock); 1124 spa_namespace_enter(FTAG); 1125 1126 if (guidp != NULL) { 1127 guid = *guidp; 1128 if (guid == 0) { 1129 error = SET_ERROR(EINVAL); 1130 goto out; 1131 } 1132 1133 if (spa_guid_exists(guid, 0)) { 1134 error = SET_ERROR(EEXIST); 1135 goto out; 1136 } 1137 } else { 1138 guid = spa_generate_guid(NULL); 1139 } 1140 1141 error = dsl_sync_task(spa->spa_name, spa_change_guid_check, 1142 spa_change_guid_sync, &guid, 5, ZFS_SPACE_CHECK_RESERVED); 1143 1144 if (error == 0) { 1145 /* 1146 * Clear the kobj flag from all the vdevs to allow 1147 * vdev_cache_process_kobj_evt() to post events to all the 1148 * vdevs since GUID is updated. 1149 */ 1150 vdev_clear_kobj_evt(spa->spa_root_vdev); 1151 for (int i = 0; i < spa->spa_l2cache.sav_count; i++) 1152 vdev_clear_kobj_evt(spa->spa_l2cache.sav_vdevs[i]); 1153 1154 spa_write_cachefile(spa, B_FALSE, B_TRUE, B_TRUE); 1155 spa_event_notify(spa, NULL, NULL, ESC_ZFS_POOL_REGUID); 1156 } 1157 1158 out: 1159 spa_namespace_exit(FTAG); 1160 mutex_exit(&spa->spa_vdev_top_lock); 1161 1162 return (error); 1163 } 1164 1165 /* 1166 * ========================================================================== 1167 * SPA state manipulation (open/create/destroy/import/export) 1168 * ========================================================================== 1169 */ 1170 1171 static int 1172 spa_error_entry_compare(const void *a, const void *b) 1173 { 1174 const spa_error_entry_t *sa = (const spa_error_entry_t *)a; 1175 const spa_error_entry_t *sb = (const spa_error_entry_t *)b; 1176 int ret; 1177 1178 ret = memcmp(&sa->se_bookmark, &sb->se_bookmark, 1179 sizeof (zbookmark_phys_t)); 1180 1181 return (TREE_ISIGN(ret)); 1182 } 1183 1184 /* 1185 * Utility function which retrieves copies of the current logs and 1186 * re-initializes them in the process. 1187 */ 1188 void 1189 spa_get_errlists(spa_t *spa, avl_tree_t *last, avl_tree_t *scrub) 1190 { 1191 ASSERT(MUTEX_HELD(&spa->spa_errlist_lock)); 1192 1193 memcpy(last, &spa->spa_errlist_last, sizeof (avl_tree_t)); 1194 memcpy(scrub, &spa->spa_errlist_scrub, sizeof (avl_tree_t)); 1195 1196 avl_create(&spa->spa_errlist_scrub, 1197 spa_error_entry_compare, sizeof (spa_error_entry_t), 1198 offsetof(spa_error_entry_t, se_avl)); 1199 avl_create(&spa->spa_errlist_last, 1200 spa_error_entry_compare, sizeof (spa_error_entry_t), 1201 offsetof(spa_error_entry_t, se_avl)); 1202 } 1203 1204 static void 1205 spa_taskqs_init(spa_t *spa, zio_type_t t, zio_taskq_type_t q) 1206 { 1207 const zio_taskq_info_t *ztip = &zio_taskqs[t][q]; 1208 enum zti_modes mode = ztip->zti_mode; 1209 uint_t value = ztip->zti_value; 1210 uint_t count = ztip->zti_count; 1211 spa_taskqs_t *tqs = &spa->spa_zio_taskq[t][q]; 1212 uint_t cpus, threads, flags = TASKQ_DYNAMIC; 1213 1214 switch (mode) { 1215 case ZTI_MODE_FIXED: 1216 ASSERT3U(value, >, 0); 1217 break; 1218 1219 case ZTI_MODE_SYNC: 1220 1221 /* 1222 * Create one wr_iss taskq for every 'zio_taskq_write_tpq' CPUs, 1223 * not to exceed the number of spa allocators, and align to it. 1224 */ 1225 threads = MAX(1, boot_ncpus * zio_taskq_batch_pct / 100); 1226 count = MAX(1, threads / MAX(1, zio_taskq_write_tpq)); 1227 count = MAX(count, (zio_taskq_batch_pct + 99) / 100); 1228 count = MIN(count, spa->spa_alloc_count); 1229 while (spa->spa_alloc_count % count != 0 && 1230 spa->spa_alloc_count < count * 2) 1231 count--; 1232 1233 /* 1234 * zio_taskq_batch_pct is unbounded and may exceed 100%, but no 1235 * single taskq may have more threads than 100% of online cpus. 1236 */ 1237 value = (zio_taskq_batch_pct + count / 2) / count; 1238 value = MIN(value, 100); 1239 flags |= TASKQ_THREADS_CPU_PCT; 1240 break; 1241 1242 case ZTI_MODE_SCALE: 1243 /* 1244 * We want more taskqs to reduce lock contention, but we want 1245 * less for better request ordering and CPU utilization. 1246 */ 1247 threads = MAX(1, boot_ncpus * zio_taskq_batch_pct / 100); 1248 threads = MAX(threads, value); 1249 if (zio_taskq_batch_tpq > 0) { 1250 count = MAX(1, (threads + zio_taskq_batch_tpq / 2) / 1251 zio_taskq_batch_tpq); 1252 } else { 1253 /* 1254 * Prefer 6 threads per taskq, but no more taskqs 1255 * than threads in them on large systems. For 80%: 1256 * 1257 * taskq taskq total 1258 * cpus taskqs percent threads threads 1259 * ------- ------- ------- ------- ------- 1260 * 1 1 80% 1 1 1261 * 2 1 80% 1 1 1262 * 4 1 80% 3 3 1263 * 8 2 40% 3 6 1264 * 16 3 27% 4 12 1265 * 32 5 16% 5 25 1266 * 64 7 11% 7 49 1267 * 128 10 8% 10 100 1268 * 256 14 6% 15 210 1269 */ 1270 cpus = MIN(threads, boot_ncpus); 1271 count = 1 + threads / 6; 1272 while (count * count > cpus) 1273 count--; 1274 } 1275 1276 /* 1277 * Try to represent the number of threads per taskq as percent 1278 * of online CPUs to allow scaling with later online/offline. 1279 * Fall back to absolute numbers if can't. 1280 */ 1281 value = (threads * 100 + boot_ncpus * count / 2) / 1282 (boot_ncpus * count); 1283 if (value < 5 || value > 100) 1284 value = MAX(1, (threads + count / 2) / count); 1285 else 1286 flags |= TASKQ_THREADS_CPU_PCT; 1287 break; 1288 1289 case ZTI_MODE_NULL: 1290 tqs->stqs_count = 0; 1291 tqs->stqs_taskq = NULL; 1292 return; 1293 1294 default: 1295 panic("unrecognized mode for %s_%s taskq (%u:%u) in " 1296 "spa_taskqs_init()", 1297 zio_type_name[t], zio_taskq_types[q], mode, value); 1298 break; 1299 } 1300 1301 ASSERT3U(count, >, 0); 1302 tqs->stqs_count = count; 1303 tqs->stqs_taskq = kmem_alloc(count * sizeof (taskq_t *), KM_SLEEP); 1304 1305 for (uint_t i = 0; i < count; i++) { 1306 taskq_t *tq; 1307 char name[32]; 1308 1309 if (count > 1) 1310 (void) snprintf(name, sizeof (name), "%s_%s_%u", 1311 zio_type_name[t], zio_taskq_types[q], i); 1312 else 1313 (void) snprintf(name, sizeof (name), "%s_%s", 1314 zio_type_name[t], zio_taskq_types[q]); 1315 1316 #ifdef HAVE_SYSDC 1317 if (zio_taskq_sysdc && spa->spa_proc != &p0) { 1318 (void) zio_taskq_basedc; 1319 tq = taskq_create_sysdc(name, value, 50, INT_MAX, 1320 spa->spa_proc, zio_taskq_basedc, flags); 1321 } else { 1322 #endif 1323 /* 1324 * The write issue taskq can be extremely CPU 1325 * intensive. Run it at slightly less important 1326 * priority than the other taskqs. 1327 */ 1328 const pri_t pri = (t == ZIO_TYPE_WRITE && 1329 q == ZIO_TASKQ_ISSUE) ? 1330 wtqclsyspri : maxclsyspri; 1331 tq = taskq_create_proc(name, value, pri, 50, 1332 INT_MAX, spa->spa_proc, flags); 1333 #ifdef HAVE_SYSDC 1334 } 1335 #endif 1336 1337 tqs->stqs_taskq[i] = tq; 1338 } 1339 } 1340 1341 static void 1342 spa_taskqs_fini(spa_t *spa, zio_type_t t, zio_taskq_type_t q) 1343 { 1344 spa_taskqs_t *tqs = &spa->spa_zio_taskq[t][q]; 1345 1346 if (tqs->stqs_taskq == NULL) { 1347 ASSERT0(tqs->stqs_count); 1348 return; 1349 } 1350 1351 for (uint_t i = 0; i < tqs->stqs_count; i++) { 1352 ASSERT3P(tqs->stqs_taskq[i], !=, NULL); 1353 taskq_destroy(tqs->stqs_taskq[i]); 1354 } 1355 1356 kmem_free(tqs->stqs_taskq, tqs->stqs_count * sizeof (taskq_t *)); 1357 tqs->stqs_taskq = NULL; 1358 } 1359 1360 #ifdef _KERNEL 1361 /* 1362 * The READ and WRITE rows of zio_taskqs are configurable at module load time 1363 * by setting zio_taskq_read or zio_taskq_write. 1364 * 1365 * Example (the defaults for READ and WRITE) 1366 * zio_taskq_read='fixed,1,8 null scale null' 1367 * zio_taskq_write='sync null scale null' 1368 * 1369 * Each sets the entire row at a time. 1370 * 1371 * 'fixed' is parameterised: fixed,Q,T where Q is number of taskqs, T is number 1372 * of threads per taskq. 1373 * 1374 * 'null' can only be set on the high-priority queues (queue selection for 1375 * high-priority queues will fall back to the regular queue if the high-pri 1376 * is NULL. 1377 */ 1378 static const char *const modes[ZTI_NMODES] = { 1379 "fixed", "scale", "sync", "null" 1380 }; 1381 1382 /* Parse the incoming config string. Modifies cfg */ 1383 static int 1384 spa_taskq_param_set(zio_type_t t, char *cfg) 1385 { 1386 int err = 0; 1387 1388 zio_taskq_info_t row[ZIO_TASKQ_TYPES] = {{0}}; 1389 1390 char *next = cfg, *tok, *c; 1391 1392 /* 1393 * Parse out each element from the string and fill `row`. The entire 1394 * row has to be set at once, so any errors are flagged by just 1395 * breaking out of this loop early. 1396 */ 1397 uint_t q; 1398 for (q = 0; q < ZIO_TASKQ_TYPES; q++) { 1399 /* `next` is the start of the config */ 1400 if (next == NULL) 1401 break; 1402 1403 /* Eat up leading space */ 1404 while (isspace(*next)) 1405 next++; 1406 if (*next == '\0') 1407 break; 1408 1409 /* Mode ends at space or end of string */ 1410 tok = next; 1411 next = strchr(tok, ' '); 1412 if (next != NULL) *next++ = '\0'; 1413 1414 /* Parameters start after a comma */ 1415 c = strchr(tok, ','); 1416 if (c != NULL) *c++ = '\0'; 1417 1418 /* Match mode string */ 1419 uint_t mode; 1420 for (mode = 0; mode < ZTI_NMODES; mode++) 1421 if (strcmp(tok, modes[mode]) == 0) 1422 break; 1423 if (mode == ZTI_NMODES) 1424 break; 1425 1426 /* Invalid canary */ 1427 row[q].zti_mode = ZTI_NMODES; 1428 1429 /* Per-mode setup */ 1430 switch (mode) { 1431 1432 /* 1433 * FIXED is parameterised: number of queues, and number of 1434 * threads per queue. 1435 */ 1436 case ZTI_MODE_FIXED: { 1437 /* No parameters? */ 1438 if (c == NULL || *c == '\0') 1439 break; 1440 1441 /* Find next parameter */ 1442 tok = c; 1443 c = strchr(tok, ','); 1444 if (c == NULL) 1445 break; 1446 1447 /* Take digits and convert */ 1448 unsigned long long nq; 1449 if (!(isdigit(*tok))) 1450 break; 1451 err = ddi_strtoull(tok, &tok, 10, &nq); 1452 /* Must succeed and also end at the next param sep */ 1453 if (err != 0 || tok != c) 1454 break; 1455 1456 /* Move past the comma */ 1457 tok++; 1458 /* Need another number */ 1459 if (!(isdigit(*tok))) 1460 break; 1461 /* Remember start to make sure we moved */ 1462 c = tok; 1463 1464 /* Take digits */ 1465 unsigned long long ntpq; 1466 err = ddi_strtoull(tok, &tok, 10, &ntpq); 1467 /* Must succeed, and moved forward */ 1468 if (err != 0 || tok == c || *tok != '\0') 1469 break; 1470 1471 /* 1472 * sanity; zero queues/threads make no sense, and 1473 * 16K is almost certainly more than anyone will ever 1474 * need and avoids silly numbers like UINT32_MAX 1475 */ 1476 if (nq == 0 || nq >= 16384 || 1477 ntpq == 0 || ntpq >= 16384) 1478 break; 1479 1480 const zio_taskq_info_t zti = ZTI_P(ntpq, nq); 1481 row[q] = zti; 1482 break; 1483 } 1484 1485 /* 1486 * SCALE is optionally parameterised by minimum number of 1487 * threads. 1488 */ 1489 case ZTI_MODE_SCALE: { 1490 unsigned long long mint = 0; 1491 if (c != NULL && *c != '\0') { 1492 /* Need a number */ 1493 if (!(isdigit(*c))) 1494 break; 1495 tok = c; 1496 1497 /* Take digits */ 1498 err = ddi_strtoull(tok, &tok, 10, &mint); 1499 /* Must succeed, and moved forward */ 1500 if (err != 0 || tok == c || *tok != '\0') 1501 break; 1502 1503 /* Sanity check */ 1504 if (mint >= 16384) 1505 break; 1506 } 1507 1508 const zio_taskq_info_t zti = ZTI_SCALE(mint); 1509 row[q] = zti; 1510 break; 1511 } 1512 1513 case ZTI_MODE_SYNC: { 1514 const zio_taskq_info_t zti = ZTI_SYNC; 1515 row[q] = zti; 1516 break; 1517 } 1518 1519 case ZTI_MODE_NULL: { 1520 /* 1521 * Can only null the high-priority queues; the general- 1522 * purpose ones have to exist. 1523 */ 1524 if (q != ZIO_TASKQ_ISSUE_HIGH && 1525 q != ZIO_TASKQ_INTERRUPT_HIGH) 1526 break; 1527 1528 const zio_taskq_info_t zti = ZTI_NULL; 1529 row[q] = zti; 1530 break; 1531 } 1532 1533 default: 1534 break; 1535 } 1536 1537 /* Ensure we set a mode */ 1538 if (row[q].zti_mode == ZTI_NMODES) 1539 break; 1540 } 1541 1542 /* Didn't get a full row, fail */ 1543 if (q < ZIO_TASKQ_TYPES) 1544 return (SET_ERROR(EINVAL)); 1545 1546 /* Eat trailing space */ 1547 if (next != NULL) 1548 while (isspace(*next)) 1549 next++; 1550 1551 /* If there's anything left over then fail */ 1552 if (next != NULL && *next != '\0') 1553 return (SET_ERROR(EINVAL)); 1554 1555 /* Success! Copy it into the real config */ 1556 for (q = 0; q < ZIO_TASKQ_TYPES; q++) 1557 zio_taskqs[t][q] = row[q]; 1558 1559 return (0); 1560 } 1561 1562 static int 1563 spa_taskq_param_get(zio_type_t t, char *buf, boolean_t add_newline) 1564 { 1565 int pos = 0; 1566 1567 /* Build paramater string from live config */ 1568 const char *sep = ""; 1569 for (uint_t q = 0; q < ZIO_TASKQ_TYPES; q++) { 1570 const zio_taskq_info_t *zti = &zio_taskqs[t][q]; 1571 if (zti->zti_mode == ZTI_MODE_FIXED) 1572 pos += sprintf(&buf[pos], "%s%s,%u,%u", sep, 1573 modes[zti->zti_mode], zti->zti_count, 1574 zti->zti_value); 1575 else if (zti->zti_mode == ZTI_MODE_SCALE && zti->zti_value > 0) 1576 pos += sprintf(&buf[pos], "%s%s,%u", sep, 1577 modes[zti->zti_mode], zti->zti_value); 1578 else 1579 pos += sprintf(&buf[pos], "%s%s", sep, 1580 modes[zti->zti_mode]); 1581 sep = " "; 1582 } 1583 1584 if (add_newline) 1585 buf[pos++] = '\n'; 1586 buf[pos] = '\0'; 1587 1588 return (pos); 1589 } 1590 1591 #ifdef __linux__ 1592 static int 1593 spa_taskq_read_param_set(const char *val, zfs_kernel_param_t *kp) 1594 { 1595 char *cfg = kmem_strdup(val); 1596 int err = spa_taskq_param_set(ZIO_TYPE_READ, cfg); 1597 kmem_strfree(cfg); 1598 return (-err); 1599 } 1600 1601 static int 1602 spa_taskq_read_param_get(char *buf, zfs_kernel_param_t *kp) 1603 { 1604 return (spa_taskq_param_get(ZIO_TYPE_READ, buf, TRUE)); 1605 } 1606 1607 static int 1608 spa_taskq_write_param_set(const char *val, zfs_kernel_param_t *kp) 1609 { 1610 char *cfg = kmem_strdup(val); 1611 int err = spa_taskq_param_set(ZIO_TYPE_WRITE, cfg); 1612 kmem_strfree(cfg); 1613 return (-err); 1614 } 1615 1616 static int 1617 spa_taskq_write_param_get(char *buf, zfs_kernel_param_t *kp) 1618 { 1619 return (spa_taskq_param_get(ZIO_TYPE_WRITE, buf, TRUE)); 1620 } 1621 1622 static int 1623 spa_taskq_free_param_set(const char *val, zfs_kernel_param_t *kp) 1624 { 1625 char *cfg = kmem_strdup(val); 1626 int err = spa_taskq_param_set(ZIO_TYPE_FREE, cfg); 1627 kmem_strfree(cfg); 1628 return (-err); 1629 } 1630 1631 static int 1632 spa_taskq_free_param_get(char *buf, zfs_kernel_param_t *kp) 1633 { 1634 return (spa_taskq_param_get(ZIO_TYPE_FREE, buf, TRUE)); 1635 } 1636 #else 1637 /* 1638 * On FreeBSD load-time parameters can be set up before malloc() is available, 1639 * so we have to do all the parsing work on the stack. 1640 */ 1641 #define SPA_TASKQ_PARAM_MAX (128) 1642 1643 static int 1644 spa_taskq_read_param(ZFS_MODULE_PARAM_ARGS) 1645 { 1646 char buf[SPA_TASKQ_PARAM_MAX]; 1647 int err; 1648 1649 (void) spa_taskq_param_get(ZIO_TYPE_READ, buf, FALSE); 1650 err = sysctl_handle_string(oidp, buf, sizeof (buf), req); 1651 if (err || req->newptr == NULL) 1652 return (err); 1653 return (spa_taskq_param_set(ZIO_TYPE_READ, buf)); 1654 } 1655 1656 static int 1657 spa_taskq_write_param(ZFS_MODULE_PARAM_ARGS) 1658 { 1659 char buf[SPA_TASKQ_PARAM_MAX]; 1660 int err; 1661 1662 (void) spa_taskq_param_get(ZIO_TYPE_WRITE, buf, FALSE); 1663 err = sysctl_handle_string(oidp, buf, sizeof (buf), req); 1664 if (err || req->newptr == NULL) 1665 return (err); 1666 return (spa_taskq_param_set(ZIO_TYPE_WRITE, buf)); 1667 } 1668 1669 static int 1670 spa_taskq_free_param(ZFS_MODULE_PARAM_ARGS) 1671 { 1672 char buf[SPA_TASKQ_PARAM_MAX]; 1673 int err; 1674 1675 (void) spa_taskq_param_get(ZIO_TYPE_FREE, buf, FALSE); 1676 err = sysctl_handle_string(oidp, buf, sizeof (buf), req); 1677 if (err || req->newptr == NULL) 1678 return (err); 1679 return (spa_taskq_param_set(ZIO_TYPE_FREE, buf)); 1680 } 1681 #endif 1682 #endif /* _KERNEL */ 1683 1684 /* 1685 * Dispatch a task to the appropriate taskq for the ZFS I/O type and priority. 1686 * Note that a type may have multiple discrete taskqs to avoid lock contention 1687 * on the taskq itself. 1688 */ 1689 void 1690 spa_taskq_dispatch(spa_t *spa, zio_type_t t, zio_taskq_type_t q, 1691 task_func_t *func, zio_t *zio, boolean_t cutinline) 1692 { 1693 spa_taskqs_t *tqs = &spa->spa_zio_taskq[t][q]; 1694 taskq_t *tq; 1695 1696 ASSERT3P(tqs->stqs_taskq, !=, NULL); 1697 ASSERT3U(tqs->stqs_count, !=, 0); 1698 1699 /* 1700 * NB: We are assuming that the zio can only be dispatched 1701 * to a single taskq at a time. It would be a grievous error 1702 * to dispatch the zio to another taskq at the same time. 1703 */ 1704 ASSERT(zio); 1705 ASSERT(taskq_empty_ent(&zio->io_tqent)); 1706 1707 if (tqs->stqs_count == 1) { 1708 tq = tqs->stqs_taskq[0]; 1709 } else if ((t == ZIO_TYPE_WRITE) && (q == ZIO_TASKQ_ISSUE) && 1710 ZIO_HAS_ALLOCATOR(zio)) { 1711 tq = tqs->stqs_taskq[zio->io_allocator % tqs->stqs_count]; 1712 } else { 1713 tq = tqs->stqs_taskq[((uint64_t)gethrtime()) % tqs->stqs_count]; 1714 } 1715 1716 taskq_dispatch_ent(tq, func, zio, cutinline ? TQ_FRONT : 0, 1717 &zio->io_tqent); 1718 } 1719 1720 static void 1721 spa_create_zio_taskqs(spa_t *spa) 1722 { 1723 for (int t = 0; t < ZIO_TYPES; t++) { 1724 for (int q = 0; q < ZIO_TASKQ_TYPES; q++) { 1725 spa_taskqs_init(spa, t, q); 1726 } 1727 } 1728 } 1729 1730 #ifdef HAVE_SPA_THREAD 1731 static void 1732 spa_thread(void *arg) 1733 { 1734 psetid_t zio_taskq_psrset_bind = PS_NONE; 1735 callb_cpr_t cprinfo; 1736 1737 spa_t *spa = arg; 1738 user_t *pu = PTOU(curproc); 1739 1740 CALLB_CPR_INIT(&cprinfo, &spa->spa_proc_lock, callb_generic_cpr, 1741 spa->spa_name); 1742 1743 ASSERT(curproc != &p0); 1744 (void) snprintf(pu->u_psargs, sizeof (pu->u_psargs), 1745 "zpool-%s", spa->spa_name); 1746 (void) strlcpy(pu->u_comm, pu->u_psargs, sizeof (pu->u_comm)); 1747 1748 /* bind this thread to the requested psrset */ 1749 if (zio_taskq_psrset_bind != PS_NONE) { 1750 pool_lock(); 1751 mutex_enter(&cpu_lock); 1752 mutex_enter(&pidlock); 1753 mutex_enter(&curproc->p_lock); 1754 1755 if (cpupart_bind_thread(curthread, zio_taskq_psrset_bind, 1756 0, NULL, NULL) == 0) { 1757 curthread->t_bind_pset = zio_taskq_psrset_bind; 1758 } else { 1759 cmn_err(CE_WARN, 1760 "Couldn't bind process for zfs pool \"%s\" to " 1761 "pset %d\n", spa->spa_name, zio_taskq_psrset_bind); 1762 } 1763 1764 mutex_exit(&curproc->p_lock); 1765 mutex_exit(&pidlock); 1766 mutex_exit(&cpu_lock); 1767 pool_unlock(); 1768 } 1769 1770 #ifdef HAVE_SYSDC 1771 if (zio_taskq_sysdc) { 1772 sysdc_thread_enter(curthread, 100, 0); 1773 } 1774 #endif 1775 1776 spa->spa_proc = curproc; 1777 spa->spa_did = curthread->t_did; 1778 1779 spa_create_zio_taskqs(spa); 1780 1781 mutex_enter(&spa->spa_proc_lock); 1782 ASSERT(spa->spa_proc_state == SPA_PROC_CREATED); 1783 1784 spa->spa_proc_state = SPA_PROC_ACTIVE; 1785 cv_broadcast(&spa->spa_proc_cv); 1786 1787 CALLB_CPR_SAFE_BEGIN(&cprinfo); 1788 while (spa->spa_proc_state == SPA_PROC_ACTIVE) 1789 cv_wait(&spa->spa_proc_cv, &spa->spa_proc_lock); 1790 CALLB_CPR_SAFE_END(&cprinfo, &spa->spa_proc_lock); 1791 1792 ASSERT(spa->spa_proc_state == SPA_PROC_DEACTIVATE); 1793 spa->spa_proc_state = SPA_PROC_GONE; 1794 spa->spa_proc = &p0; 1795 cv_broadcast(&spa->spa_proc_cv); 1796 CALLB_CPR_EXIT(&cprinfo); /* drops spa_proc_lock */ 1797 1798 mutex_enter(&curproc->p_lock); 1799 lwp_exit(); 1800 } 1801 #endif 1802 1803 extern metaslab_ops_t *metaslab_allocator(spa_t *spa); 1804 1805 /* 1806 * Activate an uninitialized pool. 1807 */ 1808 static void 1809 spa_activate(spa_t *spa, spa_mode_t mode) 1810 { 1811 metaslab_ops_t *msp = metaslab_allocator(spa); 1812 ASSERT(spa->spa_state == POOL_STATE_UNINITIALIZED); 1813 1814 spa->spa_state = POOL_STATE_ACTIVE; 1815 spa->spa_final_txg = UINT64_MAX; 1816 spa->spa_mode = mode; 1817 spa->spa_read_spacemaps = spa_mode_readable_spacemaps; 1818 1819 spa->spa_normal_class = metaslab_class_create(spa, "normal", 1820 msp, B_FALSE); 1821 spa->spa_log_class = metaslab_class_create(spa, "log", msp, B_TRUE); 1822 spa->spa_embedded_log_class = metaslab_class_create(spa, 1823 "embedded_log", msp, B_TRUE); 1824 spa->spa_special_class = metaslab_class_create(spa, "special", 1825 msp, B_FALSE); 1826 spa->spa_special_embedded_log_class = metaslab_class_create(spa, 1827 "special_embedded_log", msp, B_TRUE); 1828 spa->spa_dedup_class = metaslab_class_create(spa, "dedup", 1829 msp, B_FALSE); 1830 1831 /* Try to create a covering process */ 1832 mutex_enter(&spa->spa_proc_lock); 1833 ASSERT(spa->spa_proc_state == SPA_PROC_NONE); 1834 ASSERT(spa->spa_proc == &p0); 1835 spa->spa_did = 0; 1836 1837 #ifdef HAVE_SPA_THREAD 1838 /* Only create a process if we're going to be around a while. */ 1839 if (spa_create_process && strcmp(spa->spa_name, TRYIMPORT_NAME) != 0) { 1840 if (newproc(spa_thread, (caddr_t)spa, syscid, maxclsyspri, 1841 NULL, 0) == 0) { 1842 spa->spa_proc_state = SPA_PROC_CREATED; 1843 while (spa->spa_proc_state == SPA_PROC_CREATED) { 1844 cv_wait(&spa->spa_proc_cv, 1845 &spa->spa_proc_lock); 1846 } 1847 ASSERT(spa->spa_proc_state == SPA_PROC_ACTIVE); 1848 ASSERT(spa->spa_proc != &p0); 1849 ASSERT(spa->spa_did != 0); 1850 } else { 1851 cmn_err(CE_WARN, 1852 "Couldn't create process for zfs pool \"%s\"\n", 1853 spa->spa_name); 1854 } 1855 } 1856 #endif /* HAVE_SPA_THREAD */ 1857 mutex_exit(&spa->spa_proc_lock); 1858 1859 /* If we didn't create a process, we need to create our taskqs. */ 1860 if (spa->spa_proc == &p0) { 1861 spa_create_zio_taskqs(spa); 1862 } 1863 1864 for (size_t i = 0; i < TXG_SIZE; i++) { 1865 spa->spa_txg_zio[i] = zio_root(spa, NULL, NULL, 1866 ZIO_FLAG_CANFAIL); 1867 } 1868 1869 list_create(&spa->spa_config_dirty_list, sizeof (vdev_t), 1870 offsetof(vdev_t, vdev_config_dirty_node)); 1871 list_create(&spa->spa_evicting_os_list, sizeof (objset_t), 1872 offsetof(objset_t, os_evicting_node)); 1873 list_create(&spa->spa_state_dirty_list, sizeof (vdev_t), 1874 offsetof(vdev_t, vdev_state_dirty_node)); 1875 1876 txg_list_create(&spa->spa_vdev_txg_list, spa, 1877 offsetof(struct vdev, vdev_txg_node)); 1878 1879 avl_create(&spa->spa_errlist_scrub, 1880 spa_error_entry_compare, sizeof (spa_error_entry_t), 1881 offsetof(spa_error_entry_t, se_avl)); 1882 avl_create(&spa->spa_errlist_last, 1883 spa_error_entry_compare, sizeof (spa_error_entry_t), 1884 offsetof(spa_error_entry_t, se_avl)); 1885 avl_create(&spa->spa_errlist_healed, 1886 spa_error_entry_compare, sizeof (spa_error_entry_t), 1887 offsetof(spa_error_entry_t, se_avl)); 1888 1889 spa_activate_os(spa); 1890 1891 spa_keystore_init(&spa->spa_keystore); 1892 1893 /* 1894 * This taskq is used to perform zvol-minor-related tasks 1895 * asynchronously. This has several advantages, including easy 1896 * resolution of various deadlocks. 1897 * 1898 * The taskq must be single threaded to ensure tasks are always 1899 * processed in the order in which they were dispatched. 1900 * 1901 * A taskq per pool allows one to keep the pools independent. 1902 * This way if one pool is suspended, it will not impact another. 1903 * 1904 * The preferred location to dispatch a zvol minor task is a sync 1905 * task. In this context, there is easy access to the spa_t and minimal 1906 * error handling is required because the sync task must succeed. 1907 */ 1908 spa->spa_zvol_taskq = taskq_create("z_zvol", 1, defclsyspri, 1909 1, INT_MAX, 0); 1910 1911 /* 1912 * The taskq to preload metaslabs. 1913 */ 1914 spa->spa_metaslab_taskq = taskq_create("z_metaslab", 1915 metaslab_preload_pct, maxclsyspri, 1, INT_MAX, 1916 TASKQ_DYNAMIC | TASKQ_THREADS_CPU_PCT); 1917 1918 /* 1919 * Taskq dedicated to prefetcher threads: this is used to prevent the 1920 * pool traverse code from monopolizing the global (and limited) 1921 * system_taskq by inappropriately scheduling long running tasks on it. 1922 */ 1923 spa->spa_prefetch_taskq = taskq_create("z_prefetch", 100, 1924 defclsyspri, 1, INT_MAX, TASKQ_DYNAMIC | TASKQ_THREADS_CPU_PCT); 1925 1926 /* 1927 * The taskq to upgrade datasets in this pool. Currently used by 1928 * feature SPA_FEATURE_USEROBJ_ACCOUNTING/SPA_FEATURE_PROJECT_QUOTA. 1929 */ 1930 spa->spa_upgrade_taskq = taskq_create("z_upgrade", 100, 1931 defclsyspri, 1, INT_MAX, TASKQ_DYNAMIC | TASKQ_THREADS_CPU_PCT); 1932 } 1933 1934 /* 1935 * Opposite of spa_activate(). 1936 */ 1937 static void 1938 spa_deactivate(spa_t *spa) 1939 { 1940 if (spa->spa_create_info != NULL) { 1941 nvlist_free(spa->spa_create_info); 1942 spa->spa_create_info = NULL; 1943 } 1944 ASSERT(spa->spa_sync_on == B_FALSE); 1945 ASSERT0P(spa->spa_dsl_pool); 1946 ASSERT0P(spa->spa_root_vdev); 1947 ASSERT0P(spa->spa_async_zio_root); 1948 ASSERT(spa->spa_state != POOL_STATE_UNINITIALIZED); 1949 1950 spa_evicting_os_wait(spa); 1951 1952 if (spa->spa_zvol_taskq) { 1953 taskq_destroy(spa->spa_zvol_taskq); 1954 spa->spa_zvol_taskq = NULL; 1955 } 1956 1957 if (spa->spa_metaslab_taskq) { 1958 taskq_destroy(spa->spa_metaslab_taskq); 1959 spa->spa_metaslab_taskq = NULL; 1960 } 1961 1962 if (spa->spa_prefetch_taskq) { 1963 taskq_destroy(spa->spa_prefetch_taskq); 1964 spa->spa_prefetch_taskq = NULL; 1965 } 1966 1967 if (spa->spa_upgrade_taskq) { 1968 taskq_destroy(spa->spa_upgrade_taskq); 1969 spa->spa_upgrade_taskq = NULL; 1970 } 1971 1972 txg_list_destroy(&spa->spa_vdev_txg_list); 1973 1974 list_destroy(&spa->spa_config_dirty_list); 1975 list_destroy(&spa->spa_evicting_os_list); 1976 list_destroy(&spa->spa_state_dirty_list); 1977 1978 taskq_cancel_id(system_delay_taskq, spa->spa_deadman_tqid, B_TRUE); 1979 1980 for (int t = 0; t < ZIO_TYPES; t++) { 1981 for (int q = 0; q < ZIO_TASKQ_TYPES; q++) { 1982 spa_taskqs_fini(spa, t, q); 1983 } 1984 } 1985 1986 for (size_t i = 0; i < TXG_SIZE; i++) { 1987 ASSERT3P(spa->spa_txg_zio[i], !=, NULL); 1988 VERIFY0(zio_wait(spa->spa_txg_zio[i])); 1989 spa->spa_txg_zio[i] = NULL; 1990 } 1991 1992 metaslab_class_destroy(spa->spa_normal_class); 1993 spa->spa_normal_class = NULL; 1994 1995 metaslab_class_destroy(spa->spa_log_class); 1996 spa->spa_log_class = NULL; 1997 1998 metaslab_class_destroy(spa->spa_embedded_log_class); 1999 spa->spa_embedded_log_class = NULL; 2000 2001 metaslab_class_destroy(spa->spa_special_class); 2002 spa->spa_special_class = NULL; 2003 2004 metaslab_class_destroy(spa->spa_special_embedded_log_class); 2005 spa->spa_special_embedded_log_class = NULL; 2006 2007 metaslab_class_destroy(spa->spa_dedup_class); 2008 spa->spa_dedup_class = NULL; 2009 2010 /* 2011 * If this was part of an import or the open otherwise failed, we may 2012 * still have errors left in the queues. Empty them just in case. 2013 */ 2014 spa_errlog_drain(spa); 2015 avl_destroy(&spa->spa_errlist_scrub); 2016 avl_destroy(&spa->spa_errlist_last); 2017 avl_destroy(&spa->spa_errlist_healed); 2018 2019 spa_keystore_fini(&spa->spa_keystore); 2020 2021 spa->spa_state = POOL_STATE_UNINITIALIZED; 2022 2023 mutex_enter(&spa->spa_proc_lock); 2024 if (spa->spa_proc_state != SPA_PROC_NONE) { 2025 ASSERT(spa->spa_proc_state == SPA_PROC_ACTIVE); 2026 spa->spa_proc_state = SPA_PROC_DEACTIVATE; 2027 cv_broadcast(&spa->spa_proc_cv); 2028 while (spa->spa_proc_state == SPA_PROC_DEACTIVATE) { 2029 ASSERT(spa->spa_proc != &p0); 2030 cv_wait(&spa->spa_proc_cv, &spa->spa_proc_lock); 2031 } 2032 ASSERT(spa->spa_proc_state == SPA_PROC_GONE); 2033 spa->spa_proc_state = SPA_PROC_NONE; 2034 } 2035 ASSERT(spa->spa_proc == &p0); 2036 mutex_exit(&spa->spa_proc_lock); 2037 2038 /* 2039 * We want to make sure spa_thread() has actually exited the ZFS 2040 * module, so that the module can't be unloaded out from underneath 2041 * it. 2042 */ 2043 if (spa->spa_did != 0) { 2044 thread_join(spa->spa_did); 2045 spa->spa_did = 0; 2046 } 2047 2048 spa_deactivate_os(spa); 2049 2050 } 2051 2052 /* 2053 * Verify a pool configuration, and construct the vdev tree appropriately. This 2054 * will create all the necessary vdevs in the appropriate layout, with each vdev 2055 * in the CLOSED state. This will prep the pool before open/creation/import. 2056 * All vdev validation is done by the vdev_alloc() routine. 2057 */ 2058 int 2059 spa_config_parse(spa_t *spa, vdev_t **vdp, nvlist_t *nv, vdev_t *parent, 2060 uint_t id, int atype) 2061 { 2062 nvlist_t **child; 2063 uint_t children; 2064 int error; 2065 2066 if ((error = vdev_alloc(spa, vdp, nv, parent, id, atype)) != 0) 2067 return (error); 2068 2069 if ((*vdp)->vdev_ops->vdev_op_leaf) 2070 return (0); 2071 2072 error = nvlist_lookup_nvlist_array(nv, ZPOOL_CONFIG_CHILDREN, 2073 &child, &children); 2074 2075 if (error == ENOENT) 2076 return (0); 2077 2078 if (error) { 2079 vdev_free(*vdp); 2080 *vdp = NULL; 2081 return (SET_ERROR(EINVAL)); 2082 } 2083 2084 for (int c = 0; c < children; c++) { 2085 vdev_t *vd; 2086 if ((error = spa_config_parse(spa, &vd, child[c], *vdp, c, 2087 atype)) != 0) { 2088 vdev_free(*vdp); 2089 *vdp = NULL; 2090 return (error); 2091 } 2092 } 2093 2094 ASSERT(*vdp != NULL); 2095 2096 return (0); 2097 } 2098 2099 static boolean_t 2100 spa_should_flush_logs_on_unload(spa_t *spa) 2101 { 2102 if (!spa_feature_is_active(spa, SPA_FEATURE_LOG_SPACEMAP)) 2103 return (B_FALSE); 2104 2105 if (!spa_writeable(spa)) 2106 return (B_FALSE); 2107 2108 if (!spa->spa_sync_on) 2109 return (B_FALSE); 2110 2111 if (spa_state(spa) != POOL_STATE_EXPORTED) 2112 return (B_FALSE); 2113 2114 if (zfs_keep_log_spacemaps_at_export) 2115 return (B_FALSE); 2116 2117 return (B_TRUE); 2118 } 2119 2120 /* 2121 * Opens a transaction that will set the flag that will instruct 2122 * spa_sync to attempt to flush all the metaslabs for that txg. 2123 */ 2124 static void 2125 spa_unload_log_sm_flush_all(spa_t *spa) 2126 { 2127 dmu_tx_t *tx = dmu_tx_create_dd(spa_get_dsl(spa)->dp_mos_dir); 2128 VERIFY0(dmu_tx_assign(tx, DMU_TX_WAIT | DMU_TX_SUSPEND)); 2129 2130 spa_log_flushall_start(spa, SPA_LOG_FLUSHALL_EXPORT, 2131 dmu_tx_get_txg(tx)); 2132 2133 dmu_tx_commit(tx); 2134 txg_wait_synced(spa_get_dsl(spa), spa->spa_log_flushall_txg); 2135 } 2136 2137 static void 2138 spa_unload_log_sm_metadata(spa_t *spa) 2139 { 2140 void *cookie = NULL; 2141 spa_log_sm_t *sls; 2142 log_summary_entry_t *e; 2143 2144 while ((sls = avl_destroy_nodes(&spa->spa_sm_logs_by_txg, 2145 &cookie)) != NULL) { 2146 VERIFY0(sls->sls_mscount); 2147 kmem_free(sls, sizeof (spa_log_sm_t)); 2148 } 2149 2150 while ((e = list_remove_head(&spa->spa_log_summary)) != NULL) { 2151 VERIFY0(e->lse_mscount); 2152 kmem_free(e, sizeof (log_summary_entry_t)); 2153 } 2154 2155 spa->spa_unflushed_stats.sus_nblocks = 0; 2156 spa->spa_unflushed_stats.sus_memused = 0; 2157 spa->spa_unflushed_stats.sus_blocklimit = 0; 2158 spa->spa_unflushed_stats.sus_nmetaslabs = 0; 2159 2160 spa_log_sm_stats_update(spa); 2161 } 2162 2163 static void 2164 spa_destroy_aux_threads(spa_t *spa) 2165 { 2166 if (spa->spa_condense_zthr != NULL) { 2167 zthr_destroy(spa->spa_condense_zthr); 2168 spa->spa_condense_zthr = NULL; 2169 } 2170 if (spa->spa_checkpoint_discard_zthr != NULL) { 2171 zthr_destroy(spa->spa_checkpoint_discard_zthr); 2172 spa->spa_checkpoint_discard_zthr = NULL; 2173 } 2174 if (spa->spa_livelist_delete_zthr != NULL) { 2175 zthr_destroy(spa->spa_livelist_delete_zthr); 2176 spa->spa_livelist_delete_zthr = NULL; 2177 } 2178 if (spa->spa_livelist_condense_zthr != NULL) { 2179 zthr_destroy(spa->spa_livelist_condense_zthr); 2180 spa->spa_livelist_condense_zthr = NULL; 2181 } 2182 if (spa->spa_raidz_expand_zthr != NULL) { 2183 zthr_destroy(spa->spa_raidz_expand_zthr); 2184 spa->spa_raidz_expand_zthr = NULL; 2185 } 2186 } 2187 2188 static void 2189 spa_sync_time_logger(spa_t *spa, uint64_t txg, boolean_t force) 2190 { 2191 uint64_t curtime, dirty; 2192 dmu_tx_t *tx; 2193 dsl_pool_t *dp = spa->spa_dsl_pool; 2194 uint64_t idx = txg & TXG_MASK; 2195 2196 if (!spa_writeable(spa)) { 2197 return; 2198 } 2199 2200 curtime = gethrestime_sec(); 2201 if (txg > spa->spa_last_noted_txg && 2202 (force || 2203 curtime >= spa->spa_last_noted_txg_time + spa_note_txg_time)) { 2204 spa->spa_last_noted_txg_time = curtime; 2205 spa->spa_last_noted_txg = txg; 2206 2207 mutex_enter(&spa->spa_txg_log_time_lock); 2208 dbrrd_add(&spa->spa_txg_log_time, curtime, txg); 2209 mutex_exit(&spa->spa_txg_log_time_lock); 2210 } 2211 2212 if (!force && 2213 curtime < spa->spa_last_flush_txg_time + spa_flush_txg_time) { 2214 return; 2215 } 2216 if (txg > spa_final_dirty_txg(spa)) { 2217 return; 2218 } 2219 spa->spa_last_flush_txg_time = curtime; 2220 2221 dirty = dp->dp_dirty_pertxg[idx]; 2222 if (!force && dirty == 0) { 2223 return; 2224 } 2225 2226 spa->spa_last_flush_txg_time = curtime; 2227 tx = dmu_tx_create_assigned(spa_get_dsl(spa), txg); 2228 2229 VERIFY0(zap_update(spa_meta_objset(spa), DMU_POOL_DIRECTORY_OBJECT, 2230 DMU_POOL_TXG_LOG_TIME_MINUTES, RRD_ENTRY_SIZE, RRD_STRUCT_ELEM, 2231 &spa->spa_txg_log_time.dbr_minutes, tx)); 2232 VERIFY0(zap_update(spa_meta_objset(spa), DMU_POOL_DIRECTORY_OBJECT, 2233 DMU_POOL_TXG_LOG_TIME_DAYS, RRD_ENTRY_SIZE, RRD_STRUCT_ELEM, 2234 &spa->spa_txg_log_time.dbr_days, tx)); 2235 VERIFY0(zap_update(spa_meta_objset(spa), DMU_POOL_DIRECTORY_OBJECT, 2236 DMU_POOL_TXG_LOG_TIME_MONTHS, RRD_ENTRY_SIZE, RRD_STRUCT_ELEM, 2237 &spa->spa_txg_log_time.dbr_months, tx)); 2238 dmu_tx_commit(tx); 2239 } 2240 2241 static void 2242 spa_unload_sync_time_logger(spa_t *spa) 2243 { 2244 uint64_t txg; 2245 dmu_tx_t *tx = dmu_tx_create_dd(spa_get_dsl(spa)->dp_mos_dir); 2246 VERIFY0(dmu_tx_assign(tx, DMU_TX_WAIT)); 2247 2248 txg = dmu_tx_get_txg(tx); 2249 spa_sync_time_logger(spa, txg, B_TRUE); 2250 2251 dmu_tx_commit(tx); 2252 } 2253 2254 static void 2255 spa_load_txg_log_time(spa_t *spa) 2256 { 2257 int error; 2258 2259 error = zap_lookup(spa->spa_meta_objset, DMU_POOL_DIRECTORY_OBJECT, 2260 DMU_POOL_TXG_LOG_TIME_MINUTES, RRD_ENTRY_SIZE, RRD_STRUCT_ELEM, 2261 &spa->spa_txg_log_time.dbr_minutes); 2262 if (error != 0 && error != ENOENT) { 2263 spa_load_note(spa, "unable to load a txg time database with " 2264 "minute resolution [error=%d]", error); 2265 } 2266 error = zap_lookup(spa->spa_meta_objset, DMU_POOL_DIRECTORY_OBJECT, 2267 DMU_POOL_TXG_LOG_TIME_DAYS, RRD_ENTRY_SIZE, RRD_STRUCT_ELEM, 2268 &spa->spa_txg_log_time.dbr_days); 2269 if (error != 0 && error != ENOENT) { 2270 spa_load_note(spa, "unable to load a txg time database with " 2271 "day resolution [error=%d]", error); 2272 } 2273 error = zap_lookup(spa->spa_meta_objset, DMU_POOL_DIRECTORY_OBJECT, 2274 DMU_POOL_TXG_LOG_TIME_MONTHS, RRD_ENTRY_SIZE, RRD_STRUCT_ELEM, 2275 &spa->spa_txg_log_time.dbr_months); 2276 if (error != 0 && error != ENOENT) { 2277 spa_load_note(spa, "unable to load a txg time database with " 2278 "month resolution [error=%d]", error); 2279 } 2280 } 2281 2282 static boolean_t 2283 spa_should_sync_time_logger_on_unload(spa_t *spa) 2284 { 2285 2286 if (!spa_writeable(spa)) 2287 return (B_FALSE); 2288 2289 if (!spa->spa_sync_on) 2290 return (B_FALSE); 2291 2292 if (spa_state(spa) != POOL_STATE_EXPORTED) 2293 return (B_FALSE); 2294 2295 if (spa->spa_last_noted_txg == 0) 2296 return (B_FALSE); 2297 2298 return (B_TRUE); 2299 } 2300 2301 2302 /* 2303 * Opposite of spa_load(). 2304 */ 2305 static void 2306 spa_unload(spa_t *spa) 2307 { 2308 ASSERT(spa_namespace_held() || 2309 spa->spa_export_thread == curthread); 2310 ASSERT(spa_state(spa) != POOL_STATE_UNINITIALIZED); 2311 2312 spa_import_progress_remove(spa_guid(spa)); 2313 spa_load_note(spa, "UNLOADING"); 2314 2315 spa_wake_waiters(spa); 2316 2317 /* 2318 * If we have set the spa_final_txg, we have already performed the 2319 * tasks below in spa_export_common(). We should not redo it here since 2320 * we delay the final TXGs beyond what spa_final_txg is set at. 2321 */ 2322 if (spa->spa_final_txg == UINT64_MAX) { 2323 if (spa_should_sync_time_logger_on_unload(spa)) 2324 spa_unload_sync_time_logger(spa); 2325 2326 /* 2327 * If the log space map feature is enabled and the pool is 2328 * getting exported (but not destroyed), we want to spend some 2329 * time flushing as many metaslabs as we can in an attempt to 2330 * destroy log space maps and save import time. 2331 */ 2332 if (spa_should_flush_logs_on_unload(spa)) 2333 spa_unload_log_sm_flush_all(spa); 2334 else 2335 spa_log_flushall_done(spa); 2336 2337 /* 2338 * Stop async tasks. 2339 */ 2340 spa_async_suspend(spa); 2341 2342 if (spa->spa_root_vdev) { 2343 vdev_t *root_vdev = spa->spa_root_vdev; 2344 vdev_initialize_stop_all(root_vdev, 2345 VDEV_INITIALIZE_ACTIVE); 2346 vdev_trim_stop_all(root_vdev, VDEV_TRIM_ACTIVE); 2347 vdev_autotrim_stop_all(spa); 2348 vdev_rebuild_stop_all(spa); 2349 l2arc_spa_rebuild_stop(spa); 2350 } 2351 2352 spa_config_enter(spa, SCL_ALL, FTAG, RW_WRITER); 2353 spa->spa_final_txg = spa_last_synced_txg(spa) + 2354 TXG_DEFER_SIZE + 1; 2355 spa_config_exit(spa, SCL_ALL, FTAG); 2356 } 2357 2358 /* 2359 * Stop syncing. 2360 */ 2361 if (spa->spa_sync_on) { 2362 txg_sync_stop(spa->spa_dsl_pool); 2363 spa->spa_sync_on = B_FALSE; 2364 } 2365 2366 /* 2367 * This ensures that there is no async metaslab prefetching 2368 * while we attempt to unload the spa. 2369 */ 2370 taskq_wait(spa->spa_metaslab_taskq); 2371 2372 if (spa->spa_mmp.mmp_thread) 2373 mmp_thread_stop(spa); 2374 2375 /* 2376 * Wait for any outstanding async I/O to complete. 2377 */ 2378 if (spa->spa_async_zio_root != NULL) { 2379 for (int i = 0; i < max_ncpus; i++) 2380 (void) zio_wait(spa->spa_async_zio_root[i]); 2381 kmem_free(spa->spa_async_zio_root, max_ncpus * sizeof (void *)); 2382 spa->spa_async_zio_root = NULL; 2383 } 2384 2385 if (spa->spa_vdev_removal != NULL) { 2386 spa_vdev_removal_destroy(spa->spa_vdev_removal); 2387 spa->spa_vdev_removal = NULL; 2388 } 2389 2390 spa_destroy_aux_threads(spa); 2391 2392 spa_condense_fini(spa); 2393 2394 bpobj_close(&spa->spa_deferred_bpobj); 2395 2396 spa_config_enter(spa, SCL_ALL, spa, RW_WRITER); 2397 2398 /* 2399 * Close all vdevs. 2400 */ 2401 if (spa->spa_root_vdev) 2402 vdev_free(spa->spa_root_vdev); 2403 ASSERT0P(spa->spa_root_vdev); 2404 2405 /* 2406 * Close the dsl pool. 2407 */ 2408 if (spa->spa_dsl_pool) { 2409 dsl_pool_close(spa->spa_dsl_pool); 2410 spa->spa_dsl_pool = NULL; 2411 spa->spa_meta_objset = NULL; 2412 } 2413 2414 ddt_unload(spa); 2415 brt_unload(spa); 2416 spa_unload_log_sm_metadata(spa); 2417 2418 /* 2419 * Drop and purge level 2 cache 2420 */ 2421 spa_l2cache_drop(spa); 2422 2423 if (spa->spa_spares.sav_vdevs) { 2424 for (int i = 0; i < spa->spa_spares.sav_count; i++) 2425 vdev_free(spa->spa_spares.sav_vdevs[i]); 2426 kmem_free(spa->spa_spares.sav_vdevs, 2427 spa->spa_spares.sav_count * sizeof (void *)); 2428 spa->spa_spares.sav_vdevs = NULL; 2429 } 2430 if (spa->spa_spares.sav_config) { 2431 nvlist_free(spa->spa_spares.sav_config); 2432 spa->spa_spares.sav_config = NULL; 2433 } 2434 spa->spa_spares.sav_count = 0; 2435 2436 if (spa->spa_l2cache.sav_vdevs) { 2437 for (int i = 0; i < spa->spa_l2cache.sav_count; i++) { 2438 vdev_clear_stats(spa->spa_l2cache.sav_vdevs[i]); 2439 vdev_free(spa->spa_l2cache.sav_vdevs[i]); 2440 } 2441 kmem_free(spa->spa_l2cache.sav_vdevs, 2442 spa->spa_l2cache.sav_count * sizeof (void *)); 2443 spa->spa_l2cache.sav_vdevs = NULL; 2444 } 2445 if (spa->spa_l2cache.sav_config) { 2446 nvlist_free(spa->spa_l2cache.sav_config); 2447 spa->spa_l2cache.sav_config = NULL; 2448 } 2449 spa->spa_l2cache.sav_count = 0; 2450 2451 spa->spa_async_suspended = 0; 2452 2453 spa->spa_indirect_vdevs_loaded = B_FALSE; 2454 2455 if (spa->spa_comment != NULL) { 2456 spa_strfree(spa->spa_comment); 2457 spa->spa_comment = NULL; 2458 } 2459 if (spa->spa_compatibility != NULL) { 2460 spa_strfree(spa->spa_compatibility); 2461 spa->spa_compatibility = NULL; 2462 } 2463 2464 spa->spa_raidz_expand = NULL; 2465 spa->spa_checkpoint_txg = 0; 2466 2467 spa_config_exit(spa, SCL_ALL, spa); 2468 } 2469 2470 /* 2471 * Load (or re-load) the current list of vdevs describing the active spares for 2472 * this pool. When this is called, we have some form of basic information in 2473 * 'spa_spares.sav_config'. We parse this into vdevs, try to open them, and 2474 * then re-generate a more complete list including status information. 2475 */ 2476 void 2477 spa_load_spares(spa_t *spa) 2478 { 2479 nvlist_t **spares; 2480 uint_t nspares; 2481 int i; 2482 vdev_t *vd, *tvd; 2483 2484 #ifndef _KERNEL 2485 /* 2486 * zdb opens both the current state of the pool and the 2487 * checkpointed state (if present), with a different spa_t. 2488 * 2489 * As spare vdevs are shared among open pools, we skip loading 2490 * them when we load the checkpointed state of the pool. 2491 */ 2492 if (!spa_writeable(spa)) 2493 return; 2494 #endif 2495 2496 ASSERT(spa_config_held(spa, SCL_ALL, RW_WRITER) == SCL_ALL); 2497 2498 /* 2499 * First, close and free any existing spare vdevs. 2500 */ 2501 if (spa->spa_spares.sav_vdevs) { 2502 for (i = 0; i < spa->spa_spares.sav_count; i++) { 2503 vd = spa->spa_spares.sav_vdevs[i]; 2504 2505 /* Undo the call to spa_activate() below */ 2506 if ((tvd = spa_lookup_by_guid(spa, vd->vdev_guid, 2507 B_FALSE)) != NULL && tvd->vdev_isspare) 2508 spa_spare_remove(tvd); 2509 vdev_close(vd); 2510 vdev_free(vd); 2511 } 2512 2513 kmem_free(spa->spa_spares.sav_vdevs, 2514 spa->spa_spares.sav_count * sizeof (void *)); 2515 } 2516 2517 if (spa->spa_spares.sav_config == NULL) 2518 nspares = 0; 2519 else 2520 VERIFY0(nvlist_lookup_nvlist_array(spa->spa_spares.sav_config, 2521 ZPOOL_CONFIG_SPARES, &spares, &nspares)); 2522 2523 spa->spa_spares.sav_count = (int)nspares; 2524 spa->spa_spares.sav_vdevs = NULL; 2525 2526 if (nspares == 0) 2527 return; 2528 2529 /* 2530 * Construct the array of vdevs, opening them to get status in the 2531 * process. For each spare, there is potentially two different vdev_t 2532 * structures associated with it: one in the list of spares (used only 2533 * for basic validation purposes) and one in the active vdev 2534 * configuration (if it's spared in). During this phase we open and 2535 * validate each vdev on the spare list. If the vdev also exists in the 2536 * active configuration, then we also mark this vdev as an active spare. 2537 */ 2538 spa->spa_spares.sav_vdevs = kmem_zalloc(nspares * sizeof (void *), 2539 KM_SLEEP); 2540 for (i = 0; i < spa->spa_spares.sav_count; i++) { 2541 VERIFY0(spa_config_parse(spa, &vd, spares[i], NULL, 0, 2542 VDEV_ALLOC_SPARE)); 2543 ASSERT(vd != NULL); 2544 2545 spa->spa_spares.sav_vdevs[i] = vd; 2546 2547 if ((tvd = spa_lookup_by_guid(spa, vd->vdev_guid, 2548 B_FALSE)) != NULL) { 2549 if (!tvd->vdev_isspare) 2550 spa_spare_add(tvd); 2551 2552 /* 2553 * We only mark the spare active if we were successfully 2554 * able to load the vdev. Otherwise, importing a pool 2555 * with a bad active spare would result in strange 2556 * behavior, because multiple pool would think the spare 2557 * is actively in use. 2558 * 2559 * There is a vulnerability here to an equally bizarre 2560 * circumstance, where a dead active spare is later 2561 * brought back to life (onlined or otherwise). Given 2562 * the rarity of this scenario, and the extra complexity 2563 * it adds, we ignore the possibility. 2564 */ 2565 if (!vdev_is_dead(tvd)) 2566 spa_spare_activate(tvd); 2567 } 2568 2569 vd->vdev_top = vd; 2570 vd->vdev_aux = &spa->spa_spares; 2571 2572 if (vdev_open(vd, CRED()) != 0) 2573 continue; 2574 2575 if (vdev_validate_aux(vd) == 0) 2576 spa_spare_add(vd); 2577 } 2578 2579 /* 2580 * Recompute the stashed list of spares, with status information 2581 * this time. 2582 */ 2583 fnvlist_remove(spa->spa_spares.sav_config, ZPOOL_CONFIG_SPARES); 2584 2585 spares = kmem_alloc(spa->spa_spares.sav_count * sizeof (void *), 2586 KM_SLEEP); 2587 for (i = 0; i < spa->spa_spares.sav_count; i++) 2588 spares[i] = vdev_config_generate(spa, 2589 spa->spa_spares.sav_vdevs[i], B_TRUE, VDEV_CONFIG_SPARE); 2590 fnvlist_add_nvlist_array(spa->spa_spares.sav_config, 2591 ZPOOL_CONFIG_SPARES, (const nvlist_t * const *)spares, 2592 spa->spa_spares.sav_count); 2593 for (i = 0; i < spa->spa_spares.sav_count; i++) 2594 nvlist_free(spares[i]); 2595 kmem_free(spares, spa->spa_spares.sav_count * sizeof (void *)); 2596 } 2597 2598 /* 2599 * Load (or re-load) the current list of vdevs describing the active l2cache for 2600 * this pool. When this is called, we have some form of basic information in 2601 * 'spa_l2cache.sav_config'. We parse this into vdevs, try to open them, and 2602 * then re-generate a more complete list including status information. 2603 * Devices which are already active have their details maintained, and are 2604 * not re-opened. 2605 */ 2606 void 2607 spa_load_l2cache(spa_t *spa) 2608 { 2609 nvlist_t **l2cache = NULL; 2610 uint_t nl2cache; 2611 int i, j, oldnvdevs; 2612 uint64_t guid; 2613 vdev_t *vd, **oldvdevs, **newvdevs; 2614 spa_aux_vdev_t *sav = &spa->spa_l2cache; 2615 2616 #ifndef _KERNEL 2617 /* 2618 * zdb opens both the current state of the pool and the 2619 * checkpointed state (if present), with a different spa_t. 2620 * 2621 * As L2 caches are part of the ARC which is shared among open 2622 * pools, we skip loading them when we load the checkpointed 2623 * state of the pool. 2624 */ 2625 if (!spa_writeable(spa)) 2626 return; 2627 #endif 2628 2629 ASSERT(spa_config_held(spa, SCL_ALL, RW_WRITER) == SCL_ALL); 2630 2631 oldvdevs = sav->sav_vdevs; 2632 oldnvdevs = sav->sav_count; 2633 sav->sav_vdevs = NULL; 2634 sav->sav_count = 0; 2635 2636 if (sav->sav_config == NULL) { 2637 nl2cache = 0; 2638 newvdevs = NULL; 2639 goto out; 2640 } 2641 2642 VERIFY0(nvlist_lookup_nvlist_array(sav->sav_config, 2643 ZPOOL_CONFIG_L2CACHE, &l2cache, &nl2cache)); 2644 newvdevs = kmem_alloc(nl2cache * sizeof (void *), KM_SLEEP); 2645 2646 /* 2647 * Process new nvlist of vdevs. 2648 */ 2649 for (i = 0; i < nl2cache; i++) { 2650 guid = fnvlist_lookup_uint64(l2cache[i], ZPOOL_CONFIG_GUID); 2651 2652 newvdevs[i] = NULL; 2653 for (j = 0; j < oldnvdevs; j++) { 2654 vd = oldvdevs[j]; 2655 if (vd != NULL && guid == vd->vdev_guid) { 2656 /* 2657 * Retain previous vdev for add/remove ops. 2658 */ 2659 newvdevs[i] = vd; 2660 oldvdevs[j] = NULL; 2661 break; 2662 } 2663 } 2664 2665 if (newvdevs[i] == NULL) { 2666 /* 2667 * Create new vdev 2668 */ 2669 VERIFY0(spa_config_parse(spa, &vd, l2cache[i], NULL, 0, 2670 VDEV_ALLOC_L2CACHE)); 2671 ASSERT(vd != NULL); 2672 newvdevs[i] = vd; 2673 2674 /* 2675 * Commit this vdev as an l2cache device, 2676 * even if it fails to open. 2677 */ 2678 spa_l2cache_add(vd); 2679 2680 vd->vdev_top = vd; 2681 vd->vdev_aux = sav; 2682 2683 spa_l2cache_activate(vd); 2684 2685 if (vdev_open(vd, CRED()) != 0) 2686 continue; 2687 2688 (void) vdev_validate_aux(vd); 2689 2690 if (!vdev_is_dead(vd)) 2691 l2arc_add_vdev(spa, vd); 2692 2693 /* 2694 * Upon cache device addition to a pool or pool 2695 * creation with a cache device or if the header 2696 * of the device is invalid we issue an async 2697 * TRIM command for the whole device which will 2698 * execute if l2arc_trim_ahead > 0. 2699 */ 2700 spa_async_request(spa, SPA_ASYNC_L2CACHE_TRIM); 2701 } 2702 } 2703 2704 sav->sav_vdevs = newvdevs; 2705 sav->sav_count = (int)nl2cache; 2706 2707 /* 2708 * Recompute the stashed list of l2cache devices, with status 2709 * information this time. 2710 */ 2711 fnvlist_remove(sav->sav_config, ZPOOL_CONFIG_L2CACHE); 2712 2713 if (sav->sav_count > 0) 2714 l2cache = kmem_alloc(sav->sav_count * sizeof (void *), 2715 KM_SLEEP); 2716 for (i = 0; i < sav->sav_count; i++) 2717 l2cache[i] = vdev_config_generate(spa, 2718 sav->sav_vdevs[i], B_TRUE, VDEV_CONFIG_L2CACHE); 2719 fnvlist_add_nvlist_array(sav->sav_config, ZPOOL_CONFIG_L2CACHE, 2720 (const nvlist_t * const *)l2cache, sav->sav_count); 2721 2722 out: 2723 /* 2724 * Purge vdevs that were dropped 2725 */ 2726 if (oldvdevs) { 2727 for (i = 0; i < oldnvdevs; i++) { 2728 uint64_t pool; 2729 2730 vd = oldvdevs[i]; 2731 if (vd != NULL) { 2732 ASSERT(vd->vdev_isl2cache); 2733 2734 if (spa_l2cache_exists(vd->vdev_guid, &pool) && 2735 pool != 0ULL && l2arc_vdev_present(vd)) 2736 l2arc_remove_vdev(vd); 2737 vdev_clear_stats(vd); 2738 vdev_free(vd); 2739 } 2740 } 2741 2742 kmem_free(oldvdevs, oldnvdevs * sizeof (void *)); 2743 } 2744 2745 for (i = 0; i < sav->sav_count; i++) 2746 nvlist_free(l2cache[i]); 2747 if (sav->sav_count) 2748 kmem_free(l2cache, sav->sav_count * sizeof (void *)); 2749 } 2750 2751 static int 2752 load_nvlist(spa_t *spa, uint64_t obj, nvlist_t **value) 2753 { 2754 dmu_buf_t *db; 2755 char *packed = NULL; 2756 size_t nvsize = 0; 2757 int error; 2758 *value = NULL; 2759 2760 error = dmu_bonus_hold(spa->spa_meta_objset, obj, FTAG, &db); 2761 if (error) 2762 return (error); 2763 2764 nvsize = *(uint64_t *)db->db_data; 2765 dmu_buf_rele(db, FTAG); 2766 2767 packed = vmem_alloc(nvsize, KM_SLEEP); 2768 error = dmu_read(spa->spa_meta_objset, obj, 0, nvsize, packed, 2769 DMU_READ_PREFETCH); 2770 if (error == 0) 2771 error = nvlist_unpack(packed, nvsize, value, 0); 2772 vmem_free(packed, nvsize); 2773 2774 return (error); 2775 } 2776 2777 /* 2778 * Concrete top-level vdevs that are not missing and are not logs. At every 2779 * spa_sync we write new uberblocks to at least SPA_SYNC_MIN_VDEVS core tvds. 2780 */ 2781 static uint64_t 2782 spa_healthy_core_tvds(spa_t *spa) 2783 { 2784 vdev_t *rvd = spa->spa_root_vdev; 2785 uint64_t tvds = 0; 2786 2787 for (uint64_t i = 0; i < rvd->vdev_children; i++) { 2788 vdev_t *vd = rvd->vdev_child[i]; 2789 if (vd->vdev_islog) 2790 continue; 2791 if (vdev_is_concrete(vd) && !vdev_is_dead(vd)) 2792 tvds++; 2793 } 2794 2795 return (tvds); 2796 } 2797 2798 /* 2799 * Checks to see if the given vdev could not be opened, in which case we post a 2800 * sysevent to notify the autoreplace code that the device has been removed. 2801 */ 2802 static void 2803 spa_check_removed(vdev_t *vd) 2804 { 2805 for (uint64_t c = 0; c < vd->vdev_children; c++) 2806 spa_check_removed(vd->vdev_child[c]); 2807 2808 if (vd->vdev_ops->vdev_op_leaf && vdev_is_dead(vd) && 2809 vdev_is_concrete(vd)) { 2810 zfs_post_autoreplace(vd->vdev_spa, vd); 2811 spa_event_notify(vd->vdev_spa, vd, NULL, ESC_ZFS_VDEV_CHECK); 2812 } 2813 } 2814 2815 static int 2816 spa_check_for_missing_logs(spa_t *spa) 2817 { 2818 vdev_t *rvd = spa->spa_root_vdev; 2819 2820 /* 2821 * If we're doing a normal import, then build up any additional 2822 * diagnostic information about missing log devices. 2823 * We'll pass this up to the user for further processing. 2824 */ 2825 if (!(spa->spa_import_flags & ZFS_IMPORT_MISSING_LOG)) { 2826 nvlist_t **child, *nv; 2827 uint64_t idx = 0; 2828 2829 child = kmem_alloc(rvd->vdev_children * sizeof (nvlist_t *), 2830 KM_SLEEP); 2831 nv = fnvlist_alloc(); 2832 2833 for (uint64_t c = 0; c < rvd->vdev_children; c++) { 2834 vdev_t *tvd = rvd->vdev_child[c]; 2835 2836 /* 2837 * We consider a device as missing only if it failed 2838 * to open (i.e. offline or faulted is not considered 2839 * as missing). 2840 */ 2841 if (tvd->vdev_islog && 2842 tvd->vdev_state == VDEV_STATE_CANT_OPEN) { 2843 child[idx++] = vdev_config_generate(spa, tvd, 2844 B_FALSE, VDEV_CONFIG_MISSING); 2845 } 2846 } 2847 2848 if (idx > 0) { 2849 fnvlist_add_nvlist_array(nv, ZPOOL_CONFIG_CHILDREN, 2850 (const nvlist_t * const *)child, idx); 2851 fnvlist_add_nvlist(spa->spa_load_info, 2852 ZPOOL_CONFIG_MISSING_DEVICES, nv); 2853 2854 for (uint64_t i = 0; i < idx; i++) 2855 nvlist_free(child[i]); 2856 } 2857 nvlist_free(nv); 2858 kmem_free(child, rvd->vdev_children * sizeof (char **)); 2859 2860 if (idx > 0) { 2861 spa_load_failed(spa, "some log devices are missing"); 2862 vdev_dbgmsg_print_tree(rvd, 2); 2863 return (SET_ERROR(ENXIO)); 2864 } 2865 } else { 2866 for (uint64_t c = 0; c < rvd->vdev_children; c++) { 2867 vdev_t *tvd = rvd->vdev_child[c]; 2868 2869 if (tvd->vdev_islog && 2870 tvd->vdev_state == VDEV_STATE_CANT_OPEN) { 2871 spa_set_log_state(spa, SPA_LOG_CLEAR); 2872 spa_load_note(spa, "some log devices are " 2873 "missing, ZIL is dropped."); 2874 vdev_dbgmsg_print_tree(rvd, 2); 2875 break; 2876 } 2877 } 2878 } 2879 2880 return (0); 2881 } 2882 2883 /* 2884 * Check for missing log devices 2885 */ 2886 static boolean_t 2887 spa_check_logs(spa_t *spa) 2888 { 2889 boolean_t rv = B_FALSE; 2890 dsl_pool_t *dp = spa_get_dsl(spa); 2891 2892 switch (spa->spa_log_state) { 2893 default: 2894 break; 2895 case SPA_LOG_MISSING: 2896 /* need to recheck in case slog has been restored */ 2897 case SPA_LOG_UNKNOWN: 2898 rv = (dmu_objset_find_dp(dp, dp->dp_root_dir_obj, 2899 zil_check_log_chain, NULL, DS_FIND_CHILDREN) != 0); 2900 if (rv) 2901 spa_set_log_state(spa, SPA_LOG_MISSING); 2902 break; 2903 } 2904 return (rv); 2905 } 2906 2907 /* 2908 * Passivate any log vdevs (note, does not apply to embedded log metaslabs). 2909 */ 2910 static boolean_t 2911 spa_passivate_log(spa_t *spa) 2912 { 2913 vdev_t *rvd = spa->spa_root_vdev; 2914 boolean_t slog_found = B_FALSE; 2915 2916 ASSERT(spa_config_held(spa, SCL_ALLOC, RW_WRITER)); 2917 2918 for (int c = 0; c < rvd->vdev_children; c++) { 2919 vdev_t *tvd = rvd->vdev_child[c]; 2920 2921 if (tvd->vdev_islog) { 2922 ASSERT0P(tvd->vdev_log_mg); 2923 metaslab_group_passivate(tvd->vdev_mg); 2924 slog_found = B_TRUE; 2925 } 2926 } 2927 2928 return (slog_found); 2929 } 2930 2931 /* 2932 * Activate any log vdevs (note, does not apply to embedded log metaslabs). 2933 */ 2934 static void 2935 spa_activate_log(spa_t *spa) 2936 { 2937 vdev_t *rvd = spa->spa_root_vdev; 2938 2939 ASSERT(spa_config_held(spa, SCL_ALLOC, RW_WRITER)); 2940 2941 for (int c = 0; c < rvd->vdev_children; c++) { 2942 vdev_t *tvd = rvd->vdev_child[c]; 2943 2944 if (tvd->vdev_islog) { 2945 ASSERT0P(tvd->vdev_log_mg); 2946 metaslab_group_activate(tvd->vdev_mg); 2947 } 2948 } 2949 } 2950 2951 int 2952 spa_reset_logs(spa_t *spa) 2953 { 2954 int error; 2955 2956 error = dmu_objset_find(spa_name(spa), zil_reset, 2957 NULL, DS_FIND_CHILDREN); 2958 if (error == 0) { 2959 /* 2960 * We successfully offlined the log device, sync out the 2961 * current txg so that the "stubby" block can be removed 2962 * by zil_sync(). 2963 */ 2964 txg_wait_synced(spa->spa_dsl_pool, 0); 2965 } 2966 return (error); 2967 } 2968 2969 static void 2970 spa_aux_check_removed(spa_aux_vdev_t *sav) 2971 { 2972 for (int i = 0; i < sav->sav_count; i++) 2973 spa_check_removed(sav->sav_vdevs[i]); 2974 } 2975 2976 void 2977 spa_claim_notify(zio_t *zio) 2978 { 2979 spa_t *spa = zio->io_spa; 2980 2981 if (zio->io_error) 2982 return; 2983 2984 mutex_enter(&spa->spa_props_lock); /* any mutex will do */ 2985 if (spa->spa_claim_max_txg < BP_GET_BIRTH(zio->io_bp)) 2986 spa->spa_claim_max_txg = BP_GET_BIRTH(zio->io_bp); 2987 mutex_exit(&spa->spa_props_lock); 2988 } 2989 2990 typedef struct spa_load_error { 2991 boolean_t sle_verify_data; 2992 boolean_t sle_relaxmeta; /* tolerate non-critical meta-data */ 2993 uint64_t sle_maxmeta; /* max acceptable meta-data errors */ 2994 uint64_t sle_maxdata; /* max acceptable data errors */ 2995 uint64_t sle_meta_count; 2996 uint64_t sle_data_count; 2997 } spa_load_error_t; 2998 2999 static void 3000 spa_load_verify_done(zio_t *zio) 3001 { 3002 blkptr_t *bp = zio->io_bp; 3003 spa_load_error_t *sle = zio->io_private; 3004 dmu_object_type_t type = BP_GET_TYPE(bp); 3005 int error = zio->io_error; 3006 spa_t *spa = zio->io_spa; 3007 3008 abd_free(zio->io_abd); 3009 if (error) { 3010 boolean_t meta; 3011 3012 if (type == DMU_OT_INTENT_LOG) { 3013 meta = B_FALSE; 3014 } else if (zio->io_bookmark.zb_objset == DMU_META_OBJSET) { 3015 meta = B_TRUE; 3016 } else if (sle->sle_relaxmeta) { 3017 /* 3018 * Losing a file or a directory costs us the affected 3019 * objects, but the pool as a whole remains operable. 3020 */ 3021 meta = DMU_OT_IS_CRITICAL(type, BP_GET_LEVEL(bp)); 3022 } else { 3023 meta = BP_GET_LEVEL(bp) != 0 || 3024 DMU_OT_IS_METADATA(type); 3025 } 3026 if (meta) 3027 atomic_inc_64(&sle->sle_meta_count); 3028 else 3029 atomic_inc_64(&sle->sle_data_count); 3030 } 3031 3032 mutex_enter(&spa->spa_scrub_lock); 3033 spa->spa_load_verify_bytes -= BP_GET_PSIZE(bp); 3034 cv_broadcast(&spa->spa_scrub_io_cv); 3035 mutex_exit(&spa->spa_scrub_lock); 3036 } 3037 3038 /* 3039 * Maximum number of inflight bytes is the log2 fraction of the arc size. 3040 * By default, we set it to 1/16th of the arc. 3041 */ 3042 static uint_t spa_load_verify_shift = 4; 3043 static int spa_load_verify_metadata = B_TRUE; 3044 static int spa_load_verify_data = B_TRUE; 3045 3046 static int 3047 spa_load_verify_cb(spa_t *spa, zilog_t *zilog, const blkptr_t *bp, 3048 const zbookmark_phys_t *zb, const dnode_phys_t *dnp, void *arg) 3049 { 3050 zio_t *rio = arg; 3051 spa_load_error_t *sle = rio->io_private; 3052 3053 (void) zilog, (void) dnp; 3054 3055 /* 3056 * Note: normally this routine will not be called if 3057 * spa_load_verify_metadata is not set. However, it may be useful 3058 * to manually set the flag after the traversal has begun. 3059 */ 3060 if (!spa_load_verify_metadata) 3061 return (0); 3062 3063 /* 3064 * Stop the traversal as soon as the verdict is known, there is no 3065 * point in counting the errors we are not going to tolerate anyway. 3066 */ 3067 if (sle->sle_meta_count > sle->sle_maxmeta || 3068 sle->sle_data_count > sle->sle_maxdata) 3069 return (SET_ERROR(ECANCELED)); 3070 3071 /* 3072 * Sanity check the block pointer in order to detect obvious damage 3073 * before using the contents in subsequent checks or in zio_read(). 3074 * When damaged consider it to be a metadata error since we cannot 3075 * trust the BP_GET_TYPE and BP_GET_LEVEL values. 3076 */ 3077 if (zfs_blkptr_verify(spa, bp, BLK_CONFIG_NEEDED, BLK_VERIFY_LOG)) { 3078 atomic_inc_64(&sle->sle_meta_count); 3079 return (0); 3080 } 3081 3082 if (zb->zb_level == ZB_DNODE_LEVEL || BP_IS_HOLE(bp) || 3083 BP_IS_EMBEDDED(bp) || BP_IS_REDACTED(bp)) 3084 return (0); 3085 3086 if (!BP_IS_METADATA(bp) && 3087 (!spa_load_verify_data || !sle->sle_verify_data)) 3088 return (0); 3089 3090 uint64_t maxinflight_bytes = 3091 arc_target_bytes() >> spa_load_verify_shift; 3092 size_t size = BP_GET_PSIZE(bp); 3093 3094 mutex_enter(&spa->spa_scrub_lock); 3095 while (spa->spa_load_verify_bytes >= maxinflight_bytes) 3096 cv_wait(&spa->spa_scrub_io_cv, &spa->spa_scrub_lock); 3097 spa->spa_load_verify_bytes += size; 3098 mutex_exit(&spa->spa_scrub_lock); 3099 3100 zio_nowait(zio_read(rio, spa, bp, abd_alloc_for_io(size, B_FALSE), size, 3101 spa_load_verify_done, rio->io_private, ZIO_PRIORITY_SCRUB, 3102 ZIO_FLAG_SPECULATIVE | ZIO_FLAG_CANFAIL | 3103 ZIO_FLAG_SCRUB | ZIO_FLAG_RAW, zb)); 3104 return (0); 3105 } 3106 3107 static int 3108 verify_dataset_name_len(dsl_pool_t *dp, dsl_dataset_t *ds, void *arg) 3109 { 3110 (void) dp, (void) arg; 3111 3112 if (dsl_dataset_namelen(ds) >= ZFS_MAX_DATASET_NAME_LEN) 3113 return (SET_ERROR(ENAMETOOLONG)); 3114 3115 return (0); 3116 } 3117 3118 static int 3119 spa_load_verify(spa_t *spa) 3120 { 3121 zio_t *rio; 3122 spa_load_error_t sle = { 0 }; 3123 zpool_load_policy_t policy; 3124 boolean_t verify_ok = B_FALSE, aborted = B_FALSE; 3125 int error = 0; 3126 3127 zpool_get_load_policy(spa->spa_config, &policy); 3128 3129 if (policy.zlp_rewind & ZPOOL_NEVER_REWIND || 3130 policy.zlp_maxmeta == UINT64_MAX) 3131 return (0); 3132 3133 dsl_pool_config_enter(spa->spa_dsl_pool, FTAG); 3134 error = dmu_objset_find_dp(spa->spa_dsl_pool, 3135 spa->spa_dsl_pool->dp_root_dir_obj, verify_dataset_name_len, NULL, 3136 DS_FIND_CHILDREN); 3137 dsl_pool_config_exit(spa->spa_dsl_pool, FTAG); 3138 if (error != 0) 3139 return (error); 3140 3141 /* 3142 * Verify data only if somebody is going to look at the error count: 3143 * either the caller set a limit for it, or we are only searching for 3144 * the best txg without rewinding to it (zpool import -nF), which 3145 * reports the count back to the user. 3146 */ 3147 sle.sle_verify_data = policy.zlp_maxdata < UINT64_MAX || 3148 ((policy.zlp_rewind & ZPOOL_REWIND_MASK) && 3149 (spa_load_verify_dryrun || 3150 spa->spa_load_state != SPA_LOAD_RECOVER)); 3151 3152 sle.sle_relaxmeta = policy.zlp_relaxmeta; 3153 3154 /* 3155 * Dry run reports the errors instead of acting on them, so it needs 3156 * the complete counts. Otherwise stop counting once the thresholds 3157 * are exceeded, since the result can not change after that. 3158 */ 3159 if (spa_load_verify_dryrun) { 3160 sle.sle_maxmeta = sle.sle_maxdata = UINT64_MAX; 3161 } else { 3162 sle.sle_maxmeta = policy.zlp_maxmeta; 3163 sle.sle_maxdata = policy.zlp_maxdata; 3164 } 3165 3166 rio = zio_root(spa, NULL, &sle, 3167 ZIO_FLAG_CANFAIL | ZIO_FLAG_SPECULATIVE); 3168 3169 if (spa_load_verify_metadata) { 3170 if (spa->spa_extreme_rewind) { 3171 spa_load_note(spa, "performing a complete scan of the " 3172 "pool since extreme rewind is on. This may take " 3173 "a very long time.\n (verifying metadata=%u, " 3174 "data=%u)", spa_load_verify_metadata, 3175 spa_load_verify_data && sle.sle_verify_data); 3176 } 3177 3178 error = traverse_pool(spa, spa->spa_verify_min_txg, 3179 TRAVERSE_PRE | TRAVERSE_PREFETCH_METADATA | 3180 TRAVERSE_NO_DECRYPT | TRAVERSE_HARD, 3181 spa_load_verify_cb, rio); 3182 3183 /* 3184 * We aborted the traversal ourselves, so this is not a real 3185 * error, only the error counts below are now lower bounds. 3186 */ 3187 if (error == ECANCELED) { 3188 error = 0; 3189 aborted = B_TRUE; 3190 } 3191 } 3192 3193 (void) zio_wait(rio); 3194 ASSERT0(spa->spa_load_verify_bytes); 3195 3196 spa->spa_load_meta_errors = sle.sle_meta_count; 3197 spa->spa_load_data_errors = sle.sle_data_count; 3198 3199 if (sle.sle_meta_count != 0 || sle.sle_data_count != 0) { 3200 spa_load_note(spa, "spa_load_verify found %s%llu metadata " 3201 "errors and %llu data errors", 3202 aborted ? "at least " : "", 3203 (u_longlong_t)sle.sle_meta_count, 3204 (u_longlong_t)sle.sle_data_count); 3205 } 3206 3207 if (spa_load_verify_dryrun || 3208 (!error && sle.sle_meta_count <= policy.zlp_maxmeta && 3209 sle.sle_data_count <= policy.zlp_maxdata)) { 3210 verify_ok = B_TRUE; 3211 spa->spa_load_txg = spa->spa_uberblock.ub_txg; 3212 spa->spa_load_txg_ts = spa->spa_uberblock.ub_timestamp; 3213 3214 fnvlist_add_uint64(spa->spa_load_info, ZPOOL_CONFIG_LOAD_TXG, 3215 spa->spa_load_txg); 3216 fnvlist_add_uint64(spa->spa_load_info, ZPOOL_CONFIG_LOAD_TIME, 3217 spa->spa_load_txg_ts); 3218 /* 3219 * The loss makes sense only for a fallback to an older 3220 * uberblock, which is the only case we know the newest one in. 3221 */ 3222 if (spa->spa_last_ubsync_txg_ts != 0) { 3223 fnvlist_add_int64(spa->spa_load_info, 3224 ZPOOL_CONFIG_REWIND_TIME, 3225 spa->spa_last_ubsync_txg_ts - 3226 spa->spa_load_txg_ts); 3227 } 3228 fnvlist_add_uint64(spa->spa_load_info, 3229 ZPOOL_CONFIG_LOAD_META_ERRORS, sle.sle_meta_count); 3230 fnvlist_add_uint64(spa->spa_load_info, 3231 ZPOOL_CONFIG_LOAD_DATA_ERRORS, sle.sle_data_count); 3232 } else { 3233 spa->spa_load_max_txg = spa->spa_uberblock.ub_txg; 3234 } 3235 3236 if (spa_load_verify_dryrun) 3237 return (0); 3238 3239 if (error) { 3240 if (error != ENXIO && error != EIO) 3241 error = SET_ERROR(EIO); 3242 return (error); 3243 } 3244 3245 return (verify_ok ? 0 : EIO); 3246 } 3247 3248 /* 3249 * Find a value in the pool props object. 3250 */ 3251 static void 3252 spa_prop_find(spa_t *spa, zpool_prop_t prop, uint64_t *val) 3253 { 3254 (void) zap_lookup(spa->spa_meta_objset, spa->spa_pool_props_object, 3255 zpool_prop_to_name(prop), sizeof (uint64_t), 1, val); 3256 } 3257 3258 /* 3259 * Find a value in the pool directory object. 3260 */ 3261 static int 3262 spa_dir_prop(spa_t *spa, const char *name, uint64_t *val, boolean_t log_enoent) 3263 { 3264 int error = zap_lookup(spa->spa_meta_objset, DMU_POOL_DIRECTORY_OBJECT, 3265 name, sizeof (uint64_t), 1, val); 3266 3267 if (error != 0 && (error != ENOENT || log_enoent)) { 3268 spa_load_failed(spa, "couldn't get '%s' value in MOS directory " 3269 "[error=%d]", name, error); 3270 } 3271 3272 return (error); 3273 } 3274 3275 static int 3276 spa_vdev_err(vdev_t *vdev, vdev_aux_t aux, int err) 3277 { 3278 vdev_set_state(vdev, B_TRUE, VDEV_STATE_CANT_OPEN, aux); 3279 return (SET_ERROR(err)); 3280 } 3281 3282 boolean_t 3283 spa_livelist_delete_check(spa_t *spa) 3284 { 3285 return (spa->spa_livelists_to_delete != 0); 3286 } 3287 3288 static boolean_t 3289 spa_livelist_delete_cb_check(void *arg, zthr_t *z) 3290 { 3291 (void) z; 3292 spa_t *spa = arg; 3293 return (spa_livelist_delete_check(spa)); 3294 } 3295 3296 static int 3297 delete_blkptr_cb(void *arg, const blkptr_t *bp, dmu_tx_t *tx) 3298 { 3299 spa_t *spa = arg; 3300 zio_free(spa, tx->tx_txg, bp); 3301 dsl_dir_diduse_space(tx->tx_pool->dp_free_dir, DD_USED_HEAD, 3302 -bp_get_dsize_sync(spa, bp), 3303 -BP_GET_PSIZE(bp), -BP_GET_UCSIZE(bp), tx); 3304 return (0); 3305 } 3306 3307 static int 3308 dsl_get_next_livelist_obj(objset_t *os, uint64_t zap_obj, uint64_t *llp) 3309 { 3310 int err; 3311 zap_cursor_t zc; 3312 zap_attribute_t *za = zap_attribute_alloc(); 3313 zap_cursor_init(&zc, os, zap_obj); 3314 err = zap_cursor_retrieve(&zc, za); 3315 zap_cursor_fini(&zc); 3316 if (err == 0) 3317 *llp = za->za_first_integer; 3318 zap_attribute_free(za); 3319 return (err); 3320 } 3321 3322 /* 3323 * Components of livelist deletion that must be performed in syncing 3324 * context: freeing block pointers and updating the pool-wide data 3325 * structures to indicate how much work is left to do 3326 */ 3327 typedef struct sublist_delete_arg { 3328 spa_t *spa; 3329 dsl_deadlist_t *ll; 3330 uint64_t key; 3331 bplist_t *to_free; 3332 } sublist_delete_arg_t; 3333 3334 static void 3335 sublist_delete_sync(void *arg, dmu_tx_t *tx) 3336 { 3337 sublist_delete_arg_t *sda = arg; 3338 spa_t *spa = sda->spa; 3339 dsl_deadlist_t *ll = sda->ll; 3340 uint64_t key = sda->key; 3341 bplist_t *to_free = sda->to_free; 3342 3343 bplist_iterate(to_free, delete_blkptr_cb, spa, tx); 3344 dsl_deadlist_remove_entry(ll, key, tx); 3345 } 3346 3347 typedef struct livelist_delete_arg { 3348 spa_t *spa; 3349 uint64_t ll_obj; 3350 uint64_t zap_obj; 3351 } livelist_delete_arg_t; 3352 3353 static void 3354 livelist_delete_sync(void *arg, dmu_tx_t *tx) 3355 { 3356 livelist_delete_arg_t *lda = arg; 3357 spa_t *spa = lda->spa; 3358 uint64_t ll_obj = lda->ll_obj; 3359 uint64_t zap_obj = lda->zap_obj; 3360 objset_t *mos = spa->spa_meta_objset; 3361 uint64_t count; 3362 3363 /* free the livelist and decrement the feature count */ 3364 VERIFY0(zap_remove_int(mos, zap_obj, ll_obj, tx)); 3365 dsl_deadlist_free(mos, ll_obj, tx); 3366 spa_feature_decr(spa, SPA_FEATURE_LIVELIST, tx); 3367 VERIFY0(zap_count(mos, zap_obj, &count)); 3368 if (count == 0) { 3369 /* no more livelists to delete */ 3370 VERIFY0(zap_remove(mos, DMU_POOL_DIRECTORY_OBJECT, 3371 DMU_POOL_DELETED_CLONES, tx)); 3372 VERIFY0(zap_destroy(mos, zap_obj, tx)); 3373 spa->spa_livelists_to_delete = 0; 3374 spa_notify_waiters(spa); 3375 } 3376 } 3377 3378 /* 3379 * Load in the value for the livelist to be removed and open it. Then, 3380 * load its first sublist and determine which block pointers should actually 3381 * be freed. Then, call a synctask which performs the actual frees and updates 3382 * the pool-wide livelist data. 3383 */ 3384 static void 3385 spa_livelist_delete_cb(void *arg, zthr_t *z) 3386 { 3387 spa_t *spa = arg; 3388 uint64_t ll_obj = 0, count; 3389 objset_t *mos = spa->spa_meta_objset; 3390 uint64_t zap_obj = spa->spa_livelists_to_delete; 3391 /* 3392 * Determine the next livelist to delete. This function should only 3393 * be called if there is at least one deleted clone. 3394 */ 3395 VERIFY0(dsl_get_next_livelist_obj(mos, zap_obj, &ll_obj)); 3396 VERIFY0(zap_count(mos, ll_obj, &count)); 3397 if (count > 0) { 3398 dsl_deadlist_t *ll; 3399 dsl_deadlist_entry_t *dle; 3400 bplist_t to_free; 3401 ll = kmem_zalloc(sizeof (dsl_deadlist_t), KM_SLEEP); 3402 VERIFY0(dsl_deadlist_open(ll, mos, ll_obj)); 3403 dle = dsl_deadlist_first(ll); 3404 ASSERT3P(dle, !=, NULL); 3405 bplist_create(&to_free); 3406 int err = dsl_process_sub_livelist(&dle->dle_bpobj, &to_free, 3407 z, NULL); 3408 if (err == 0) { 3409 sublist_delete_arg_t sync_arg = { 3410 .spa = spa, 3411 .ll = ll, 3412 .key = dle->dle_mintxg, 3413 .to_free = &to_free 3414 }; 3415 zfs_dbgmsg("deleting sublist (id %llu) from" 3416 " livelist %llu, %lld remaining", 3417 (u_longlong_t)dle->dle_bpobj.bpo_object, 3418 (u_longlong_t)ll_obj, (longlong_t)count - 1); 3419 VERIFY0(dsl_sync_task(spa_name(spa), NULL, 3420 sublist_delete_sync, &sync_arg, 0, 3421 ZFS_SPACE_CHECK_DESTROY)); 3422 } else { 3423 VERIFY3U(err, ==, EINTR); 3424 } 3425 bplist_clear(&to_free); 3426 bplist_destroy(&to_free); 3427 dsl_deadlist_close(ll); 3428 kmem_free(ll, sizeof (dsl_deadlist_t)); 3429 } else { 3430 livelist_delete_arg_t sync_arg = { 3431 .spa = spa, 3432 .ll_obj = ll_obj, 3433 .zap_obj = zap_obj 3434 }; 3435 zfs_dbgmsg("deletion of livelist %llu completed", 3436 (u_longlong_t)ll_obj); 3437 VERIFY0(dsl_sync_task(spa_name(spa), NULL, livelist_delete_sync, 3438 &sync_arg, 0, ZFS_SPACE_CHECK_DESTROY)); 3439 } 3440 } 3441 3442 static void 3443 spa_start_livelist_destroy_thread(spa_t *spa) 3444 { 3445 ASSERT0P(spa->spa_livelist_delete_zthr); 3446 spa->spa_livelist_delete_zthr = 3447 zthr_create("z_livelist_destroy", 3448 spa_livelist_delete_cb_check, spa_livelist_delete_cb, spa, 3449 minclsyspri); 3450 } 3451 3452 typedef struct livelist_new_arg { 3453 bplist_t *allocs; 3454 bplist_t *frees; 3455 } livelist_new_arg_t; 3456 3457 static int 3458 livelist_track_new_cb(void *arg, const blkptr_t *bp, boolean_t bp_freed, 3459 dmu_tx_t *tx) 3460 { 3461 ASSERT0P(tx); 3462 livelist_new_arg_t *lna = arg; 3463 if (bp_freed) { 3464 bplist_append(lna->frees, bp); 3465 } else { 3466 bplist_append(lna->allocs, bp); 3467 zfs_livelist_condense_new_alloc++; 3468 } 3469 return (0); 3470 } 3471 3472 typedef struct livelist_condense_arg { 3473 spa_t *spa; 3474 bplist_t to_keep; 3475 uint64_t first_size; 3476 uint64_t next_size; 3477 } livelist_condense_arg_t; 3478 3479 static void 3480 spa_livelist_condense_sync(void *arg, dmu_tx_t *tx) 3481 { 3482 livelist_condense_arg_t *lca = arg; 3483 spa_t *spa = lca->spa; 3484 bplist_t new_frees; 3485 dsl_dataset_t *ds = spa->spa_to_condense.ds; 3486 3487 /* Have we been cancelled? */ 3488 if (spa->spa_to_condense.cancelled) { 3489 zfs_livelist_condense_sync_cancel++; 3490 goto out; 3491 } 3492 3493 dsl_deadlist_entry_t *first = spa->spa_to_condense.first; 3494 dsl_deadlist_entry_t *next = spa->spa_to_condense.next; 3495 dsl_deadlist_t *ll = &ds->ds_dir->dd_livelist; 3496 3497 /* 3498 * It's possible that the livelist was changed while the zthr was 3499 * running. Therefore, we need to check for new blkptrs in the two 3500 * entries being condensed and continue to track them in the livelist. 3501 * Because of the way we handle remapped blkptrs (see dbuf_remap_impl), 3502 * it's possible that the newly added blkptrs are FREEs or ALLOCs so 3503 * we need to sort them into two different bplists. 3504 */ 3505 uint64_t first_obj = first->dle_bpobj.bpo_object; 3506 uint64_t next_obj = next->dle_bpobj.bpo_object; 3507 uint64_t cur_first_size = first->dle_bpobj.bpo_phys->bpo_num_blkptrs; 3508 uint64_t cur_next_size = next->dle_bpobj.bpo_phys->bpo_num_blkptrs; 3509 3510 bplist_create(&new_frees); 3511 livelist_new_arg_t new_bps = { 3512 .allocs = &lca->to_keep, 3513 .frees = &new_frees, 3514 }; 3515 3516 if (cur_first_size > lca->first_size) { 3517 VERIFY0(livelist_bpobj_iterate_from_nofree(&first->dle_bpobj, 3518 livelist_track_new_cb, &new_bps, lca->first_size)); 3519 } 3520 if (cur_next_size > lca->next_size) { 3521 VERIFY0(livelist_bpobj_iterate_from_nofree(&next->dle_bpobj, 3522 livelist_track_new_cb, &new_bps, lca->next_size)); 3523 } 3524 3525 dsl_deadlist_clear_entry(first, ll, tx); 3526 ASSERT(bpobj_is_empty(&first->dle_bpobj)); 3527 dsl_deadlist_remove_entry(ll, next->dle_mintxg, tx); 3528 3529 bplist_iterate(&lca->to_keep, dsl_deadlist_insert_alloc_cb, ll, tx); 3530 bplist_iterate(&new_frees, dsl_deadlist_insert_free_cb, ll, tx); 3531 bplist_destroy(&new_frees); 3532 3533 char dsname[ZFS_MAX_DATASET_NAME_LEN]; 3534 dsl_dataset_name(ds, dsname); 3535 zfs_dbgmsg("txg %llu condensing livelist of %s (id %llu), bpobj %llu " 3536 "(%llu blkptrs) and bpobj %llu (%llu blkptrs) -> bpobj %llu " 3537 "(%llu blkptrs)", (u_longlong_t)tx->tx_txg, dsname, 3538 (u_longlong_t)ds->ds_object, (u_longlong_t)first_obj, 3539 (u_longlong_t)cur_first_size, (u_longlong_t)next_obj, 3540 (u_longlong_t)cur_next_size, 3541 (u_longlong_t)first->dle_bpobj.bpo_object, 3542 (u_longlong_t)first->dle_bpobj.bpo_phys->bpo_num_blkptrs); 3543 out: 3544 dmu_buf_rele(ds->ds_dbuf, spa); 3545 spa->spa_to_condense.ds = NULL; 3546 bplist_clear(&lca->to_keep); 3547 bplist_destroy(&lca->to_keep); 3548 kmem_free(lca, sizeof (livelist_condense_arg_t)); 3549 spa->spa_to_condense.syncing = B_FALSE; 3550 } 3551 3552 static void 3553 spa_livelist_condense_cb(void *arg, zthr_t *t) 3554 { 3555 while (zfs_livelist_condense_zthr_pause && 3556 !(zthr_has_waiters(t) || zthr_iscancelled(t))) 3557 delay(1); 3558 3559 spa_t *spa = arg; 3560 dsl_deadlist_entry_t *first = spa->spa_to_condense.first; 3561 dsl_deadlist_entry_t *next = spa->spa_to_condense.next; 3562 uint64_t first_size, next_size; 3563 3564 livelist_condense_arg_t *lca = 3565 kmem_alloc(sizeof (livelist_condense_arg_t), KM_SLEEP); 3566 bplist_create(&lca->to_keep); 3567 3568 /* 3569 * Process the livelists (matching FREEs and ALLOCs) in open context 3570 * so we have minimal work in syncing context to condense. 3571 * 3572 * We save bpobj sizes (first_size and next_size) to use later in 3573 * syncing context to determine if entries were added to these sublists 3574 * while in open context. This is possible because the clone is still 3575 * active and open for normal writes and we want to make sure the new, 3576 * unprocessed blockpointers are inserted into the livelist normally. 3577 * 3578 * Note that dsl_process_sub_livelist() both stores the size number of 3579 * blockpointers and iterates over them while the bpobj's lock held, so 3580 * the sizes returned to us are consistent which what was actually 3581 * processed. 3582 */ 3583 int err = dsl_process_sub_livelist(&first->dle_bpobj, &lca->to_keep, t, 3584 &first_size); 3585 if (err == 0) 3586 err = dsl_process_sub_livelist(&next->dle_bpobj, &lca->to_keep, 3587 t, &next_size); 3588 3589 if (err == 0) { 3590 while (zfs_livelist_condense_sync_pause && 3591 !(zthr_has_waiters(t) || zthr_iscancelled(t))) 3592 delay(1); 3593 3594 dmu_tx_t *tx = dmu_tx_create_dd(spa_get_dsl(spa)->dp_mos_dir); 3595 dmu_tx_mark_netfree(tx); 3596 dmu_tx_hold_space(tx, 1); 3597 err = dmu_tx_assign(tx, DMU_TX_NOWAIT | DMU_TX_NOTHROTTLE); 3598 if (err == 0) { 3599 /* 3600 * Prevent the condense zthr restarting before 3601 * the synctask completes. 3602 */ 3603 spa->spa_to_condense.syncing = B_TRUE; 3604 lca->spa = spa; 3605 lca->first_size = first_size; 3606 lca->next_size = next_size; 3607 dsl_sync_task_nowait(spa_get_dsl(spa), 3608 spa_livelist_condense_sync, lca, tx); 3609 dmu_tx_commit(tx); 3610 return; 3611 } 3612 } 3613 /* 3614 * Condensing can not continue: either it was externally stopped or 3615 * we were unable to assign to a tx because the pool has run out of 3616 * space. In the second case, we'll just end up trying to condense 3617 * again in a later txg. 3618 */ 3619 ASSERT(err != 0); 3620 bplist_clear(&lca->to_keep); 3621 bplist_destroy(&lca->to_keep); 3622 kmem_free(lca, sizeof (livelist_condense_arg_t)); 3623 dmu_buf_rele(spa->spa_to_condense.ds->ds_dbuf, spa); 3624 spa->spa_to_condense.ds = NULL; 3625 if (err == EINTR) 3626 zfs_livelist_condense_zthr_cancel++; 3627 } 3628 3629 /* 3630 * Check that there is something to condense but that a condense is not 3631 * already in progress and that condensing has not been cancelled. 3632 */ 3633 static boolean_t 3634 spa_livelist_condense_cb_check(void *arg, zthr_t *z) 3635 { 3636 (void) z; 3637 spa_t *spa = arg; 3638 if ((spa->spa_to_condense.ds != NULL) && 3639 (spa->spa_to_condense.syncing == B_FALSE) && 3640 (spa->spa_to_condense.cancelled == B_FALSE)) { 3641 return (B_TRUE); 3642 } 3643 return (B_FALSE); 3644 } 3645 3646 static void 3647 spa_start_livelist_condensing_thread(spa_t *spa) 3648 { 3649 spa->spa_to_condense.ds = NULL; 3650 spa->spa_to_condense.first = NULL; 3651 spa->spa_to_condense.next = NULL; 3652 spa->spa_to_condense.syncing = B_FALSE; 3653 spa->spa_to_condense.cancelled = B_FALSE; 3654 3655 ASSERT0P(spa->spa_livelist_condense_zthr); 3656 spa->spa_livelist_condense_zthr = 3657 zthr_create("z_livelist_condense", 3658 spa_livelist_condense_cb_check, 3659 spa_livelist_condense_cb, spa, minclsyspri); 3660 } 3661 3662 static void 3663 spa_spawn_aux_threads(spa_t *spa) 3664 { 3665 ASSERT(spa_writeable(spa)); 3666 3667 spa_start_raidz_expansion_thread(spa); 3668 spa_start_indirect_condensing_thread(spa); 3669 spa_start_livelist_destroy_thread(spa); 3670 spa_start_livelist_condensing_thread(spa); 3671 3672 ASSERT0P(spa->spa_checkpoint_discard_zthr); 3673 spa->spa_checkpoint_discard_zthr = 3674 zthr_create("z_checkpoint_discard", 3675 spa_checkpoint_discard_thread_check, 3676 spa_checkpoint_discard_thread, spa, minclsyspri); 3677 } 3678 3679 /* 3680 * Fix up config after a partly-completed split. This is done with the 3681 * ZPOOL_CONFIG_SPLIT nvlist. Both the splitting pool and the split-off 3682 * pool have that entry in their config, but only the splitting one contains 3683 * a list of all the guids of the vdevs that are being split off. 3684 * 3685 * This function determines what to do with that list: either rejoin 3686 * all the disks to the pool, or complete the splitting process. To attempt 3687 * the rejoin, each disk that is offlined is marked online again, and 3688 * we do a reopen() call. If the vdev label for every disk that was 3689 * marked online indicates it was successfully split off (VDEV_AUX_SPLIT_POOL) 3690 * then we call vdev_split() on each disk, and complete the split. 3691 * 3692 * Otherwise we leave the config alone, with all the vdevs in place in 3693 * the original pool. 3694 */ 3695 static void 3696 spa_try_repair(spa_t *spa, nvlist_t *config) 3697 { 3698 uint_t extracted; 3699 uint64_t *glist; 3700 uint_t i, gcount; 3701 nvlist_t *nvl; 3702 vdev_t **vd; 3703 boolean_t attempt_reopen; 3704 3705 if (nvlist_lookup_nvlist(config, ZPOOL_CONFIG_SPLIT, &nvl) != 0) 3706 return; 3707 3708 /* check that the config is complete */ 3709 if (nvlist_lookup_uint64_array(nvl, ZPOOL_CONFIG_SPLIT_LIST, 3710 &glist, &gcount) != 0) 3711 return; 3712 3713 vd = kmem_zalloc(gcount * sizeof (vdev_t *), KM_SLEEP); 3714 3715 /* attempt to online all the vdevs & validate */ 3716 attempt_reopen = B_TRUE; 3717 for (i = 0; i < gcount; i++) { 3718 if (glist[i] == 0) /* vdev is hole */ 3719 continue; 3720 3721 vd[i] = spa_lookup_by_guid(spa, glist[i], B_FALSE); 3722 if (vd[i] == NULL) { 3723 /* 3724 * Don't bother attempting to reopen the disks; 3725 * just do the split. 3726 */ 3727 attempt_reopen = B_FALSE; 3728 } else { 3729 /* attempt to re-online it */ 3730 vd[i]->vdev_offline = B_FALSE; 3731 } 3732 } 3733 3734 if (attempt_reopen) { 3735 vdev_reopen(spa->spa_root_vdev); 3736 3737 /* check each device to see what state it's in */ 3738 for (extracted = 0, i = 0; i < gcount; i++) { 3739 if (vd[i] != NULL && 3740 vd[i]->vdev_stat.vs_aux != VDEV_AUX_SPLIT_POOL) 3741 break; 3742 ++extracted; 3743 } 3744 } 3745 3746 /* 3747 * If every disk has been moved to the new pool, or if we never 3748 * even attempted to look at them, then we split them off for 3749 * good. 3750 */ 3751 if (!attempt_reopen || gcount == extracted) { 3752 for (i = 0; i < gcount; i++) 3753 if (vd[i] != NULL) 3754 vdev_split(vd[i]); 3755 vdev_reopen(spa->spa_root_vdev); 3756 } 3757 3758 kmem_free(vd, gcount * sizeof (vdev_t *)); 3759 } 3760 3761 static int 3762 spa_load(spa_t *spa, spa_load_state_t state, spa_import_type_t type) 3763 { 3764 const char *ereport = FM_EREPORT_ZFS_POOL; 3765 int error; 3766 3767 spa->spa_load_state = state; 3768 (void) spa_import_progress_set_state(spa_guid(spa), 3769 spa_load_state(spa)); 3770 spa_import_progress_set_notes(spa, "spa_load()"); 3771 3772 gethrestime(&spa->spa_loaded_ts); 3773 error = spa_load_impl(spa, type, &ereport); 3774 3775 /* 3776 * Don't count references from objsets that are already closed 3777 * and are making their way through the eviction process. 3778 */ 3779 spa_evicting_os_wait(spa); 3780 spa->spa_minref = zfs_refcount_count(&spa->spa_refcount); 3781 if (error) { 3782 if (error != EEXIST) { 3783 spa->spa_loaded_ts.tv_sec = 0; 3784 spa->spa_loaded_ts.tv_nsec = 0; 3785 } 3786 if (error != EBADF) { 3787 (void) zfs_ereport_post(ereport, spa, 3788 NULL, NULL, NULL, 0); 3789 } 3790 } 3791 spa->spa_load_state = error ? SPA_LOAD_ERROR : SPA_LOAD_NONE; 3792 spa->spa_ena = 0; 3793 3794 (void) spa_import_progress_set_state(spa_guid(spa), 3795 spa_load_state(spa)); 3796 3797 return (error); 3798 } 3799 3800 #ifdef ZFS_DEBUG 3801 /* 3802 * Count the number of per-vdev ZAPs associated with all of the vdevs in the 3803 * vdev tree rooted in the given vd, and ensure that each ZAP is present in the 3804 * spa's per-vdev ZAP list. 3805 */ 3806 static uint64_t 3807 vdev_count_verify_zaps(vdev_t *vd) 3808 { 3809 spa_t *spa = vd->vdev_spa; 3810 uint64_t total = 0; 3811 3812 if (spa_feature_is_active(vd->vdev_spa, SPA_FEATURE_AVZ_V2) && 3813 vd->vdev_root_zap != 0) { 3814 total++; 3815 ASSERT0(zap_lookup_int(spa->spa_meta_objset, 3816 spa->spa_all_vdev_zaps, vd->vdev_root_zap)); 3817 } 3818 if (vd->vdev_top_zap != 0) { 3819 total++; 3820 ASSERT0(zap_lookup_int(spa->spa_meta_objset, 3821 spa->spa_all_vdev_zaps, vd->vdev_top_zap)); 3822 } 3823 if (vd->vdev_leaf_zap != 0) { 3824 total++; 3825 ASSERT0(zap_lookup_int(spa->spa_meta_objset, 3826 spa->spa_all_vdev_zaps, vd->vdev_leaf_zap)); 3827 } 3828 3829 for (uint64_t i = 0; i < vd->vdev_children; i++) { 3830 total += vdev_count_verify_zaps(vd->vdev_child[i]); 3831 } 3832 3833 return (total); 3834 } 3835 #else 3836 #define vdev_count_verify_zaps(vd) ((void) sizeof (vd), 0) 3837 #endif 3838 3839 /* 3840 * Check the results load_info results from previous tryimport. 3841 * 3842 * error results: 3843 * 0 - Pool remains in an idle state 3844 * EREMOTEIO - Pool was known to be active on the other host 3845 * ENOENT - The config does not contain complete tryimport info 3846 */ 3847 static int 3848 spa_activity_verify_config(spa_t *spa, uberblock_t *ub) 3849 { 3850 uint64_t tryconfig_mmp_state = MMP_STATE_ACTIVE; 3851 uint64_t tryconfig_txg = 0; 3852 uint64_t tryconfig_timestamp = 0; 3853 uint16_t tryconfig_mmp_seq = 0; 3854 nvlist_t *nvinfo, *config = spa->spa_config; 3855 int error; 3856 3857 /* Simply a non-zero value to indicate the verify was done. */ 3858 spa->spa_mmp.mmp_import_ns = 1000; 3859 3860 error = nvlist_lookup_nvlist(config, ZPOOL_CONFIG_LOAD_INFO, &nvinfo); 3861 if (error) 3862 return (SET_ERROR(ENOENT)); 3863 3864 /* 3865 * If ZPOOL_CONFIG_MMP_STATE is present an activity check was performed 3866 * during the earlier tryimport. If the state recorded there isn't 3867 * MMP_STATE_INACTIVE the pool is known to be active on another host. 3868 */ 3869 error = nvlist_lookup_uint64(nvinfo, ZPOOL_CONFIG_MMP_STATE, 3870 &tryconfig_mmp_state); 3871 if (error) 3872 return (SET_ERROR(ENOENT)); 3873 3874 if (tryconfig_mmp_state != MMP_STATE_INACTIVE) { 3875 spa_load_failed(spa, "mmp: pool is active on remote host, " 3876 "state=%llu", (u_longlong_t)tryconfig_mmp_state); 3877 return (SET_ERROR(EREMOTEIO)); 3878 } 3879 3880 /* 3881 * If ZPOOL_CONFIG_MMP_TXG is present an activity check was performed 3882 * during the earlier tryimport. If the txg recorded there is 0 then 3883 * the pool is known to be active on another host. 3884 */ 3885 error = nvlist_lookup_uint64(nvinfo, ZPOOL_CONFIG_MMP_TXG, 3886 &tryconfig_txg); 3887 if (error) 3888 return (SET_ERROR(ENOENT)); 3889 3890 if (tryconfig_txg == 0) { 3891 spa_load_failed(spa, "mmp: pool is active on remote host, " 3892 "tryconfig_txg=%llu", (u_longlong_t)tryconfig_txg); 3893 return (SET_ERROR(EREMOTEIO)); 3894 } 3895 3896 error = nvlist_lookup_uint64(config, ZPOOL_CONFIG_TIMESTAMP, 3897 &tryconfig_timestamp); 3898 if (error) 3899 return (SET_ERROR(ENOENT)); 3900 3901 error = nvlist_lookup_uint16(nvinfo, ZPOOL_CONFIG_MMP_SEQ, 3902 &tryconfig_mmp_seq); 3903 if (error) 3904 return (SET_ERROR(ENOENT)); 3905 3906 if (tryconfig_timestamp == ub->ub_timestamp && 3907 tryconfig_txg == ub->ub_txg && 3908 MMP_SEQ_VALID(ub) && tryconfig_mmp_seq == MMP_SEQ(ub)) { 3909 zfs_dbgmsg("mmp: verified pool mmp tryimport config, " 3910 "spa=%s", spa_load_name(spa)); 3911 return (0); 3912 } 3913 3914 spa_load_failed(spa, "mmp: pool is active on remote host, " 3915 "tc_timestamp=%llu ub_timestamp=%llu " 3916 "tc_txg=%llu ub_txg=%llu tc_seq=%llu ub_seq=%llu", 3917 (u_longlong_t)tryconfig_timestamp, (u_longlong_t)ub->ub_timestamp, 3918 (u_longlong_t)tryconfig_txg, (u_longlong_t)ub->ub_txg, 3919 (u_longlong_t)tryconfig_mmp_seq, (u_longlong_t)MMP_SEQ(ub)); 3920 3921 return (SET_ERROR(EREMOTEIO)); 3922 } 3923 3924 /* 3925 * Determine whether the activity check is required. 3926 */ 3927 static boolean_t 3928 spa_activity_check_required(spa_t *spa, uberblock_t *ub, nvlist_t *label) 3929 { 3930 nvlist_t *config = spa->spa_config; 3931 uint64_t state = POOL_STATE_ACTIVE; 3932 uint64_t hostid = 0; 3933 3934 /* 3935 * Disable the MMP activity check - This is used by zdb which 3936 * is always read-only and intended to be used on potentially 3937 * active pools. 3938 */ 3939 if (spa->spa_import_flags & ZFS_IMPORT_SKIP_MMP) { 3940 zfs_dbgmsg("mmp: skipping check ZFS_IMPORT_SKIP_MMP is set, " 3941 "spa=%s", spa_load_name(spa)); 3942 return (B_FALSE); 3943 } 3944 3945 /* 3946 * Skip the activity check when the MMP feature is disabled. 3947 * - MMP_MAGIC not set - Legacy pool predates the MMP feature, or 3948 * - MMP_MAGIC set && mmp_delay == 0 - MMP feature is disabled. 3949 */ 3950 if ((ub->ub_mmp_magic != MMP_MAGIC) || 3951 (ub->ub_mmp_magic == MMP_MAGIC && ub->ub_mmp_delay == 0)) { 3952 zfs_dbgmsg("mmp: skipping check: feature is disabled, " 3953 "spa=%s", spa_load_name(spa)); 3954 return (B_FALSE); 3955 } 3956 3957 /* 3958 * Allow the activity check to be skipped when importing a cleanly 3959 * exported pool on the same host which last imported it. Since the 3960 * hostid from configuration may be stale use the one read from the 3961 * label. Imports from other hostids must perform the activity check. 3962 */ 3963 if (label != NULL) { 3964 if (nvlist_exists(label, ZPOOL_CONFIG_HOSTID)) 3965 hostid = fnvlist_lookup_uint64(label, 3966 ZPOOL_CONFIG_HOSTID); 3967 3968 if (nvlist_exists(config, ZPOOL_CONFIG_POOL_STATE)) 3969 state = fnvlist_lookup_uint64(config, 3970 ZPOOL_CONFIG_POOL_STATE); 3971 3972 if (spa_get_hostid(spa) && hostid == spa_get_hostid(spa) && 3973 state == POOL_STATE_EXPORTED) { 3974 zfs_dbgmsg("mmp: skipping check: hostid matches " 3975 "and pool is exported, spa=%s, hostid=%llx", 3976 spa_load_name(spa), (u_longlong_t)hostid); 3977 return (B_FALSE); 3978 } 3979 3980 if (state == POOL_STATE_DESTROYED) { 3981 zfs_dbgmsg("mmp: skipping check: intentionally " 3982 "destroyed pool, spa=%s", spa_load_name(spa)); 3983 return (B_FALSE); 3984 } 3985 } 3986 3987 return (B_TRUE); 3988 } 3989 3990 /* 3991 * Nanoseconds the activity check must watch for changes on-disk. 3992 */ 3993 static uint64_t 3994 spa_activity_check_duration(spa_t *spa, uberblock_t *ub) 3995 { 3996 uint64_t import_intervals = MAX(zfs_multihost_import_intervals, 1); 3997 uint64_t multihost_interval = MSEC2NSEC( 3998 MMP_INTERVAL_OK(zfs_multihost_interval)); 3999 uint64_t import_delay = MAX(NANOSEC, import_intervals * 4000 multihost_interval); 4001 4002 /* 4003 * Local tunables determine a minimum duration except for the case 4004 * where we know when the remote host will suspend the pool if MMP 4005 * writes do not land. 4006 * 4007 * See Big Theory comment at the top of mmp.c for the reasoning behind 4008 * these cases and times. 4009 */ 4010 4011 ASSERT(MMP_IMPORT_SAFETY_FACTOR >= 100); 4012 4013 if (MMP_INTERVAL_VALID(ub) && MMP_FAIL_INT_VALID(ub) && 4014 MMP_FAIL_INT(ub) > 0) { 4015 4016 /* MMP on remote host will suspend pool after failed writes */ 4017 import_delay = MMP_FAIL_INT(ub) * MSEC2NSEC(MMP_INTERVAL(ub)) * 4018 MMP_IMPORT_SAFETY_FACTOR / 100; 4019 4020 zfs_dbgmsg("mmp: settings spa=%s fail_intvals>0 " 4021 "import_delay=%llu mmp_fails=%llu mmp_interval=%llu " 4022 "import_intervals=%llu", spa_load_name(spa), 4023 (u_longlong_t)import_delay, 4024 (u_longlong_t)MMP_FAIL_INT(ub), 4025 (u_longlong_t)MMP_INTERVAL(ub), 4026 (u_longlong_t)import_intervals); 4027 4028 } else if (MMP_INTERVAL_VALID(ub) && MMP_FAIL_INT_VALID(ub) && 4029 MMP_FAIL_INT(ub) == 0) { 4030 4031 /* MMP on remote host will never suspend pool */ 4032 import_delay = MAX(import_delay, (MSEC2NSEC(MMP_INTERVAL(ub)) + 4033 ub->ub_mmp_delay) * import_intervals); 4034 4035 zfs_dbgmsg("mmp: settings spa=%s fail_intvals=0 " 4036 "import_delay=%llu mmp_interval=%llu ub_mmp_delay=%llu " 4037 "import_intervals=%llu", spa_load_name(spa), 4038 (u_longlong_t)import_delay, 4039 (u_longlong_t)MMP_INTERVAL(ub), 4040 (u_longlong_t)ub->ub_mmp_delay, 4041 (u_longlong_t)import_intervals); 4042 4043 } else if (MMP_VALID(ub)) { 4044 /* 4045 * zfs-0.7 compatibility case 4046 */ 4047 4048 import_delay = MAX(import_delay, (multihost_interval + 4049 ub->ub_mmp_delay) * import_intervals); 4050 4051 zfs_dbgmsg("mmp: settings spa=%s import_delay=%llu " 4052 "ub_mmp_delay=%llu import_intervals=%llu leaves=%u", 4053 spa_load_name(spa), (u_longlong_t)import_delay, 4054 (u_longlong_t)ub->ub_mmp_delay, 4055 (u_longlong_t)import_intervals, 4056 vdev_count_leaves(spa)); 4057 } else { 4058 /* Using local tunings is the only reasonable option */ 4059 zfs_dbgmsg("mmp: pool last imported on non-MMP aware " 4060 "host using settings spa=%s import_delay=%llu " 4061 "multihost_interval=%llu import_intervals=%llu", 4062 spa_load_name(spa), (u_longlong_t)import_delay, 4063 (u_longlong_t)multihost_interval, 4064 (u_longlong_t)import_intervals); 4065 } 4066 4067 return (import_delay); 4068 } 4069 4070 /* 4071 * Store the observed pool status in spa->spa_load_info nvlist. If the 4072 * remote hostname or hostid are available from configuration read from 4073 * disk store them as well. Additionally, provide some diagnostic info 4074 * for which activity checks were run and their duration. This allows 4075 * 'zpool import' to generate a more useful message. 4076 * 4077 * Mandatory observed pool status 4078 * - ZPOOL_CONFIG_MMP_STATE - observed pool status (active/inactive) 4079 * - ZPOOL_CONFIG_MMP_TXG - observed pool txg number 4080 * - ZPOOL_CONFIG_MMP_SEQ - observed pool sequence id 4081 * 4082 * Optional information for detailed reporting 4083 * - ZPOOL_CONFIG_MMP_HOSTNAME - hostname from the active pool 4084 * - ZPOOL_CONFIG_MMP_HOSTID - hostid from the active pool 4085 * - ZPOOL_CONFIG_MMP_RESULT - set to result of activity check 4086 * - ZPOOL_CONFIG_MMP_TRYIMPORT_NS - tryimport duration in nanosec 4087 * - ZPOOL_CONFIG_MMP_IMPORT_NS - import duration in nanosec 4088 * - ZPOOL_CONFIG_MMP_CLAIM_NS - claim duration in nanosec 4089 * 4090 * ZPOOL_CONFIG_MMP_RESULT can be set to: 4091 * - ENXIO - system hostid not set 4092 * - ESRCH - activity check skipped 4093 * - EREMOTEIO - activity check detected active pool 4094 * - ENODEV - claim could not be written to a device the config expects 4095 * - EIO - claim writes were issued to present devices and failed 4096 * - EINTR - activity check interrupted 4097 * - 0 - activity check detected no activity 4098 * 4099 * ENODEV and EIO are reported with ZPOOL_CONFIG_MMP_STATE set to 4100 * MMP_STATE_ACTIVE even though no remote host was seen. Nothing is actually 4101 * active in either case, but an older zpool(8) knows only the two existing 4102 * states and reaches zfs_error_aux() with an uninitialized buffer for any 4103 * other value, so the state is kept as one it understands and the real cause 4104 * travels in the result. 4105 */ 4106 static void 4107 spa_activity_set_load_info(spa_t *spa, nvlist_t *label, mmp_state_t state, 4108 uint64_t txg, uint16_t seq, int error) 4109 { 4110 mmp_thread_t *mmp = &spa->spa_mmp; 4111 const char *hostname = NULL; 4112 uint64_t hostid = 0; 4113 4114 /* Always report a zero txg and seq id for active pools. */ 4115 if (state == MMP_STATE_ACTIVE) { 4116 ASSERT0(txg); 4117 ASSERT0(seq); 4118 } 4119 4120 if (label) { 4121 if (nvlist_exists(label, ZPOOL_CONFIG_HOSTNAME)) { 4122 hostname = fnvlist_lookup_string(label, 4123 ZPOOL_CONFIG_HOSTNAME); 4124 fnvlist_add_string(spa->spa_load_info, 4125 ZPOOL_CONFIG_MMP_HOSTNAME, hostname); 4126 } 4127 4128 if (nvlist_exists(label, ZPOOL_CONFIG_HOSTID)) { 4129 hostid = fnvlist_lookup_uint64(label, 4130 ZPOOL_CONFIG_HOSTID); 4131 fnvlist_add_uint64(spa->spa_load_info, 4132 ZPOOL_CONFIG_MMP_HOSTID, hostid); 4133 } 4134 } 4135 4136 fnvlist_add_uint64(spa->spa_load_info, ZPOOL_CONFIG_MMP_STATE, state); 4137 fnvlist_add_uint64(spa->spa_load_info, ZPOOL_CONFIG_MMP_TXG, txg); 4138 fnvlist_add_uint16(spa->spa_load_info, ZPOOL_CONFIG_MMP_SEQ, seq); 4139 fnvlist_add_uint32(spa->spa_load_info, ZPOOL_CONFIG_MMP_RESULT, error); 4140 4141 if (mmp->mmp_tryimport_ns > 0) { 4142 fnvlist_add_uint64(spa->spa_load_info, 4143 ZPOOL_CONFIG_MMP_TRYIMPORT_NS, mmp->mmp_tryimport_ns); 4144 } 4145 4146 if (mmp->mmp_import_ns > 0) { 4147 fnvlist_add_uint64(spa->spa_load_info, 4148 ZPOOL_CONFIG_MMP_IMPORT_NS, mmp->mmp_import_ns); 4149 } 4150 4151 if (mmp->mmp_claim_ns > 0) { 4152 fnvlist_add_uint64(spa->spa_load_info, 4153 ZPOOL_CONFIG_MMP_CLAIM_NS, mmp->mmp_claim_ns); 4154 } 4155 4156 zfs_dbgmsg("mmp: set spa_load_info, spa=%s hostname=%s hostid=%llx " 4157 "state=%d txg=%llu seq=%llu tryimport_ns=%lld import_ns=%lld " 4158 "claim_ns=%lld", spa_load_name(spa), 4159 hostname != NULL ? hostname : "none", (u_longlong_t)hostid, 4160 (int)state, (u_longlong_t)txg, (u_longlong_t)seq, 4161 (longlong_t)mmp->mmp_tryimport_ns, (longlong_t)mmp->mmp_import_ns, 4162 (longlong_t)mmp->mmp_claim_ns); 4163 } 4164 4165 static int 4166 spa_ld_activity_result(spa_t *spa, int error, const char *state) 4167 { 4168 switch (error) { 4169 case ENXIO: 4170 cmn_err(CE_WARN, "pool '%s' system hostid not set, " 4171 "aborted import during %s", spa_load_name(spa), state); 4172 /* Userspace expects EREMOTEIO for no system hostid */ 4173 error = EREMOTEIO; 4174 break; 4175 case ENODEV: 4176 cmn_err(CE_WARN, "pool '%s' could not claim every device the " 4177 "config expects present, aborted import during %s; if a " 4178 "device is permanently gone see 'zhack mmp reclaim'", 4179 spa_load_name(spa), state); 4180 /* Userspace expects EREMOTEIO for a failed claim */ 4181 error = EREMOTEIO; 4182 break; 4183 case EIO: 4184 cmn_err(CE_WARN, "pool '%s' had I/O errors writing the claim, " 4185 "aborted import during %s", spa_load_name(spa), state); 4186 /* Userspace expects EREMOTEIO for a failed claim */ 4187 error = EREMOTEIO; 4188 break; 4189 case EREMOTEIO: 4190 cmn_err(CE_WARN, "pool '%s' activity detected, aborted " 4191 "import during %s", spa_load_name(spa), state); 4192 break; 4193 case EINTR: 4194 cmn_err(CE_WARN, "pool '%s' activity check, interrupted " 4195 "import during %s", spa_load_name(spa), state); 4196 break; 4197 case 0: 4198 cmn_err(CE_NOTE, "pool '%s' activity check completed " 4199 "successfully", spa_load_name(spa)); 4200 break; 4201 } 4202 4203 return (error); 4204 } 4205 4206 4207 /* 4208 * Remote host activity check. Performed during tryimport when the pool 4209 * has passed on the basic sanity check and is open read-only. 4210 * 4211 * error results: 4212 * 0 - no activity detected 4213 * EREMOTEIO - remote activity detected 4214 * EINTR - user canceled the operation 4215 */ 4216 static int 4217 spa_activity_check_tryimport(spa_t *spa, uberblock_t *spa_ub, 4218 boolean_t importing) 4219 { 4220 kcondvar_t cv; 4221 kmutex_t mtx; 4222 int error = 0; 4223 4224 cv_init(&cv, NULL, CV_DEFAULT, NULL); 4225 mutex_init(&mtx, NULL, MUTEX_DEFAULT, NULL); 4226 mutex_enter(&mtx); 4227 4228 uint64_t import_delay = spa_activity_check_duration(spa, spa_ub); 4229 hrtime_t start_time = gethrtime(); 4230 4231 /* Add a small random factor in case of simultaneous imports (0-25%) */ 4232 import_delay += import_delay * random_in_range(250) / 1000; 4233 hrtime_t import_expire = gethrtime() + import_delay; 4234 4235 if (importing) { 4236 /* Console message includes tryimport and claim time */ 4237 hrtime_t extra_delay = MMP_IMPORT_VERIFY_ITERS * 4238 MSEC2NSEC(MMP_INTERVAL_VALID(spa_ub) ? 4239 MMP_INTERVAL(spa_ub) : MMP_MIN_INTERVAL); 4240 cmn_err(CE_NOTE, "pool '%s' activity check required, " 4241 "%llu seconds remaining", spa_load_name(spa), 4242 (u_longlong_t)MAX(NSEC2SEC(import_delay + extra_delay), 1)); 4243 spa_import_progress_set_notes(spa, "Checking MMP activity, " 4244 "waiting %llu ms", (u_longlong_t)NSEC2MSEC(import_delay)); 4245 } 4246 4247 hrtime_t now; 4248 nvlist_t *mmp_label = NULL; 4249 4250 while ((now = gethrtime()) < import_expire) { 4251 vdev_t *rvd = spa->spa_root_vdev; 4252 uberblock_t mmp_ub; 4253 4254 if (importing) { 4255 (void) spa_import_progress_set_mmp_check(spa_guid(spa), 4256 NSEC2SEC(import_expire - gethrtime())); 4257 } 4258 4259 vdev_uberblock_load(rvd, &mmp_ub, &mmp_label); 4260 4261 if (vdev_uberblock_compare(spa_ub, &mmp_ub)) { 4262 spa_load_failed(spa, "mmp: activity detected during " 4263 "tryimport, spa_ub_txg=%llu mmp_ub_txg=%llu " 4264 "spa_ub_seq=%llu mmp_ub_seq=%llu " 4265 "spa_ub_timestamp=%llu mmp_ub_timestamp=%llu " 4266 "spa_ub_config=%#llx mmp_ub_config=%#llx", 4267 (u_longlong_t)spa_ub->ub_txg, 4268 (u_longlong_t)mmp_ub.ub_txg, 4269 (u_longlong_t)(MMP_SEQ_VALID(spa_ub) ? 4270 MMP_SEQ(spa_ub) : 0), 4271 (u_longlong_t)(MMP_SEQ_VALID(&mmp_ub) ? 4272 MMP_SEQ(&mmp_ub) : 0), 4273 (u_longlong_t)spa_ub->ub_timestamp, 4274 (u_longlong_t)mmp_ub.ub_timestamp, 4275 (u_longlong_t)spa_ub->ub_mmp_config, 4276 (u_longlong_t)mmp_ub.ub_mmp_config); 4277 error = SET_ERROR(EREMOTEIO); 4278 break; 4279 } 4280 4281 if (mmp_label) { 4282 nvlist_free(mmp_label); 4283 mmp_label = NULL; 4284 } 4285 4286 error = cv_timedwait_sig(&cv, &mtx, ddi_get_lbolt() + hz); 4287 if (error != -1) { 4288 error = SET_ERROR(EINTR); 4289 break; 4290 } 4291 error = 0; 4292 } 4293 4294 mutex_exit(&mtx); 4295 mutex_destroy(&mtx); 4296 cv_destroy(&cv); 4297 4298 if (mmp_label) 4299 nvlist_free(mmp_label); 4300 4301 if (spa->spa_load_state == SPA_LOAD_IMPORT || 4302 spa->spa_load_state == SPA_LOAD_OPEN) { 4303 spa->spa_mmp.mmp_import_ns = gethrtime() - start_time; 4304 } else { 4305 spa->spa_mmp.mmp_tryimport_ns = gethrtime() - start_time; 4306 } 4307 4308 return (error); 4309 } 4310 4311 /* 4312 * Remote host activity check. Performed during import when the pool has 4313 * passed most sanity check and has been reopened read/write. 4314 * 4315 * error results: 4316 * 0 - no activity detected 4317 * EREMOTEIO - remote activity detected 4318 * ENODEV - the claim could not be written to a device the config 4319 * expects to be present 4320 * EIO - the claim writes were issued to present devices and failed 4321 * EINTR - user canceled the operation 4322 */ 4323 static int 4324 spa_activity_check_claim(spa_t *spa) 4325 { 4326 vdev_t *rvd = spa->spa_root_vdev; 4327 nvlist_t *mmp_label; 4328 uberblock_t spa_ub; 4329 kcondvar_t cv; 4330 kmutex_t mtx; 4331 int error = 0; 4332 4333 cv_init(&cv, NULL, CV_DEFAULT, NULL); 4334 mutex_init(&mtx, NULL, MUTEX_DEFAULT, NULL); 4335 mutex_enter(&mtx); 4336 4337 hrtime_t start_time = gethrtime(); 4338 4339 /* 4340 * Load the best uberblock and verify it matches the uberblock already 4341 * identified and stored as spa->spa_uberblock to verify the pool has 4342 * not changed. 4343 */ 4344 vdev_uberblock_load(rvd, &spa_ub, &mmp_label); 4345 4346 if (memcmp(&spa->spa_uberblock, &spa_ub, sizeof (uberblock_t))) { 4347 spa_load_failed(spa, "mmp: uberblock changed on disk"); 4348 error = SET_ERROR(EREMOTEIO); 4349 goto out; 4350 } 4351 4352 if (!MMP_VALID(&spa_ub) || !MMP_INTERVAL_VALID(&spa_ub) || 4353 !MMP_SEQ_VALID(&spa_ub) || !MMP_FAIL_INT_VALID(&spa_ub)) { 4354 spa_load_failed(spa, "mmp: is not enabled in spa uberblock"); 4355 error = SET_ERROR(EREMOTEIO); 4356 goto out; 4357 } 4358 4359 nvlist_free(mmp_label); 4360 mmp_label = NULL; 4361 4362 uint64_t spa_ub_interval = MMP_INTERVAL(&spa_ub); 4363 uint16_t spa_ub_seq = MMP_SEQ(&spa_ub); 4364 4365 /* 4366 * In the highly unlikely event the sequence numbers have been 4367 * exhaused reset the sequence to zero. As long as the MMP 4368 * uberblock is updated on all of the vdevs the activity will 4369 * still be detected. 4370 */ 4371 if (MMP_SEQ_MAX == spa_ub_seq) 4372 spa_ub_seq = 0; 4373 4374 spa_import_progress_set_notes(spa, 4375 "Establishing MMP claim, waiting %llu ms", 4376 (u_longlong_t)(MMP_IMPORT_VERIFY_ITERS * spa_ub_interval)); 4377 4378 /* 4379 * Repeatedly sync out an MMP uberblock with a randomly selected 4380 * sequence number, then read it back after the MMP interval. This 4381 * random value acts as a claim token and is visible on other hosts. 4382 * If the same random value is read back we can be certain no other 4383 * pool is attempting to import the pool. 4384 */ 4385 for (int i = MMP_IMPORT_VERIFY_ITERS; i > 0; i--) { 4386 uberblock_t set_ub, mmp_ub; 4387 uint16_t mmp_seq; 4388 4389 (void) spa_import_progress_set_mmp_check(spa_guid(spa), 4390 NSEC2SEC(i * MSEC2NSEC(spa_ub_interval))); 4391 4392 set_ub = spa_ub; 4393 mmp_seq = spa_ub_seq + 1 + 4394 random_in_range(MMP_SEQ_MAX - spa_ub_seq); 4395 MMP_SEQ_CLEAR(&set_ub); 4396 set_ub.ub_mmp_config |= MMP_SEQ_SET(mmp_seq); 4397 4398 error = mmp_claim_uberblock(spa, rvd, &set_ub); 4399 if (error) { 4400 spa_load_failed(spa, "mmp: uberblock claim " 4401 "failed, error=%d", error); 4402 /* 4403 * ENODEV and EIO are both kept distinct from the 4404 * EREMOTEIO returned when another host is seen below. 4405 * Failing to write the claim is not evidence of a 4406 * remote host, and only the ENODEV case has a 4407 * recovery. 4408 */ 4409 break; 4410 } 4411 4412 error = cv_timedwait_sig(&cv, &mtx, ddi_get_lbolt() + 4413 MSEC_TO_TICK(spa_ub_interval)); 4414 if (error != -1) { 4415 error = SET_ERROR(EINTR); 4416 break; 4417 } 4418 4419 vdev_uberblock_load(rvd, &mmp_ub, &mmp_label); 4420 4421 if (vdev_uberblock_compare(&set_ub, &mmp_ub)) { 4422 spa_load_failed(spa, "mmp: activity detected during " 4423 "claim, set_ub_txg=%llu mmp_ub_txg=%llu " 4424 "set_ub_seq=%llu mmp_ub_seq=%llu " 4425 "set_ub_timestamp=%llu mmp_ub_timestamp=%llu " 4426 "set_ub_config=%#llx mmp_ub_config=%#llx", 4427 (u_longlong_t)set_ub.ub_txg, 4428 (u_longlong_t)mmp_ub.ub_txg, 4429 (u_longlong_t)(MMP_SEQ_VALID(&set_ub) ? 4430 MMP_SEQ(&set_ub) : 0), 4431 (u_longlong_t)(MMP_SEQ_VALID(&mmp_ub) ? 4432 MMP_SEQ(&mmp_ub) : 0), 4433 (u_longlong_t)set_ub.ub_timestamp, 4434 (u_longlong_t)mmp_ub.ub_timestamp, 4435 (u_longlong_t)set_ub.ub_mmp_config, 4436 (u_longlong_t)mmp_ub.ub_mmp_config); 4437 error = SET_ERROR(EREMOTEIO); 4438 break; 4439 } 4440 4441 if (mmp_label) { 4442 nvlist_free(mmp_label); 4443 mmp_label = NULL; 4444 } 4445 4446 error = 0; 4447 } 4448 out: 4449 spa->spa_mmp.mmp_claim_ns = gethrtime() - start_time; 4450 (void) spa_import_progress_set_mmp_check(spa_guid(spa), 0); 4451 4452 /* 4453 * A claim shortfall reaches userspace as EREMOTEIO exactly as remote 4454 * activity does, so an older zpool(8) sees no change. The cause 4455 * travels in the result for a zpool(8) which knows to read it. 4456 */ 4457 if (error == EREMOTEIO || error == ENODEV || error == EIO) { 4458 spa_activity_set_load_info(spa, mmp_label, 4459 MMP_STATE_ACTIVE, 0, 0, error); 4460 } else { 4461 spa_activity_set_load_info(spa, mmp_label, 4462 MMP_STATE_INACTIVE, spa_ub.ub_txg, MMP_SEQ(&spa_ub), 0); 4463 } 4464 4465 /* 4466 * Restore the original sequence, this allows us to retry the 4467 * import procedure if a subsequent step fails during import. 4468 * Failure to restore it reduces the available sequence ids for 4469 * the next import but shouldn't be considered fatal. 4470 */ 4471 int restore_error = mmp_claim_uberblock(spa, rvd, &spa_ub); 4472 if (restore_error) { 4473 zfs_dbgmsg("mmp: uberblock restore failed, spa=%s error=%d", 4474 spa_load_name(spa), restore_error); 4475 } 4476 4477 if (mmp_label) 4478 nvlist_free(mmp_label); 4479 4480 mutex_exit(&mtx); 4481 mutex_destroy(&mtx); 4482 cv_destroy(&cv); 4483 4484 return (error); 4485 } 4486 4487 static int 4488 spa_ld_activity_check(spa_t *spa, uberblock_t *ub, nvlist_t *label) 4489 { 4490 vdev_t *rvd = spa->spa_root_vdev; 4491 int error; 4492 4493 if (ub->ub_mmp_magic == MMP_MAGIC && ub->ub_mmp_delay && 4494 spa_get_hostid(spa) == 0) { 4495 spa_activity_set_load_info(spa, label, MMP_STATE_NO_HOSTID, 4496 ub->ub_txg, MMP_SEQ_VALID(ub) ? MMP_SEQ(ub) : 0, ENXIO); 4497 zfs_dbgmsg("mmp: system hostid not set, ub_mmp_magic=%llx " 4498 "ub_mmp_delay=%llu hostid=%llx", 4499 (u_longlong_t)ub->ub_mmp_magic, 4500 (u_longlong_t)ub->ub_mmp_delay, 4501 (u_longlong_t)spa_get_hostid(spa)); 4502 return (spa_vdev_err(rvd, VDEV_AUX_ACTIVE, ENXIO)); 4503 } 4504 4505 switch (spa->spa_load_state) { 4506 case SPA_LOAD_TRYIMPORT: 4507 tryimport: 4508 error = spa_activity_check_tryimport(spa, ub, B_TRUE); 4509 if (error == EREMOTEIO) { 4510 spa_activity_set_load_info(spa, label, 4511 MMP_STATE_ACTIVE, 0, 0, EREMOTEIO); 4512 return (spa_vdev_err(rvd, VDEV_AUX_ACTIVE, EREMOTEIO)); 4513 } else if (error) { 4514 ASSERT3S(error, ==, EINTR); 4515 spa_activity_set_load_info(spa, label, 4516 MMP_STATE_ACTIVE, 0, 0, EINTR); 4517 return (error); 4518 } 4519 4520 spa_activity_set_load_info(spa, label, MMP_STATE_INACTIVE, 4521 ub->ub_txg, MMP_SEQ_VALID(ub) ? MMP_SEQ(ub) : 0, 0); 4522 4523 break; 4524 4525 case SPA_LOAD_IMPORT: 4526 case SPA_LOAD_OPEN: 4527 error = spa_activity_verify_config(spa, ub); 4528 if (error == EREMOTEIO) { 4529 spa_activity_set_load_info(spa, label, 4530 MMP_STATE_ACTIVE, 0, 0, EREMOTEIO); 4531 return (spa_vdev_err(rvd, VDEV_AUX_ACTIVE, EREMOTEIO)); 4532 } else if (error) { 4533 ASSERT3S(error, ==, ENOENT); 4534 goto tryimport; 4535 } 4536 4537 /* Load info set in spa_activity_check_claim() */ 4538 4539 break; 4540 4541 case SPA_LOAD_RECOVER: 4542 zfs_dbgmsg("mmp: skipping mmp check for rewind, spa=%s", 4543 spa_load_name(spa)); 4544 break; 4545 4546 default: 4547 spa_activity_set_load_info(spa, label, MMP_STATE_ACTIVE, 4548 0, 0, EREMOTEIO); 4549 zfs_dbgmsg("mmp: unreachable, spa=%s spa_load_state=%d", 4550 spa_load_name(spa), spa->spa_load_state); 4551 return (spa_vdev_err(rvd, VDEV_AUX_ACTIVE, EREMOTEIO)); 4552 } 4553 4554 return (0); 4555 } 4556 4557 /* 4558 * Called from zfs_ioc_clear for a pool that was suspended 4559 * after failing mmp write checks. 4560 */ 4561 boolean_t 4562 spa_mmp_remote_host_activity(spa_t *spa) 4563 { 4564 ASSERT(spa_multihost(spa) && spa_suspended(spa)); 4565 4566 nvlist_t *best_label; 4567 uberblock_t best_ub; 4568 4569 /* 4570 * Locate the best uberblock on disk 4571 */ 4572 vdev_uberblock_load(spa->spa_root_vdev, &best_ub, &best_label); 4573 if (best_label) { 4574 /* 4575 * confirm that the best hostid matches our hostid 4576 */ 4577 if (nvlist_exists(best_label, ZPOOL_CONFIG_HOSTID) && 4578 spa_get_hostid(spa) != 4579 fnvlist_lookup_uint64(best_label, ZPOOL_CONFIG_HOSTID)) { 4580 nvlist_free(best_label); 4581 return (B_TRUE); 4582 } 4583 nvlist_free(best_label); 4584 } else { 4585 return (B_TRUE); 4586 } 4587 4588 if (!MMP_VALID(&best_ub) || 4589 !MMP_FAIL_INT_VALID(&best_ub) || 4590 MMP_FAIL_INT(&best_ub) == 0) { 4591 return (B_TRUE); 4592 } 4593 4594 if (best_ub.ub_txg != spa->spa_uberblock.ub_txg || 4595 best_ub.ub_timestamp != spa->spa_uberblock.ub_timestamp) { 4596 zfs_dbgmsg("mmp: txg mismatch detected during pool clear, " 4597 "spa=%s txg=%llu ub_txg=%llu timestamp=%llu " 4598 "ub_timestamp=%llu", spa_name(spa), 4599 (u_longlong_t)spa->spa_uberblock.ub_txg, 4600 (u_longlong_t)best_ub.ub_txg, 4601 (u_longlong_t)spa->spa_uberblock.ub_timestamp, 4602 (u_longlong_t)best_ub.ub_timestamp); 4603 return (B_TRUE); 4604 } 4605 4606 /* 4607 * Perform an activity check looking for any remote writer 4608 */ 4609 return (spa_activity_check_tryimport(spa, &best_ub, B_FALSE) != 0); 4610 } 4611 4612 static int 4613 spa_verify_host(spa_t *spa, nvlist_t *mos_config) 4614 { 4615 uint64_t hostid; 4616 const char *hostname; 4617 uint64_t myhostid = 0; 4618 4619 if (!spa_is_root(spa) && nvlist_lookup_uint64(mos_config, 4620 ZPOOL_CONFIG_HOSTID, &hostid) == 0) { 4621 hostname = fnvlist_lookup_string(mos_config, 4622 ZPOOL_CONFIG_HOSTNAME); 4623 4624 myhostid = zone_get_hostid(NULL); 4625 4626 if (hostid != 0 && myhostid != 0 && hostid != myhostid) { 4627 cmn_err(CE_WARN, "pool '%s' could not be " 4628 "loaded as it was last accessed by " 4629 "another system (host: %s hostid: 0x%llx). " 4630 "See: https://openzfs.github.io/openzfs-docs/msg/" 4631 "ZFS-8000-EY", 4632 spa_name(spa), hostname, (u_longlong_t)hostid); 4633 spa_load_failed(spa, "hostid verification failed: pool " 4634 "last accessed by host: %s (hostid: 0x%llx)", 4635 hostname, (u_longlong_t)hostid); 4636 return (SET_ERROR(EBADF)); 4637 } 4638 } 4639 4640 return (0); 4641 } 4642 4643 static int 4644 spa_ld_parse_config(spa_t *spa, spa_import_type_t type) 4645 { 4646 int error = 0; 4647 nvlist_t *nvtree, *nvl, *config = spa->spa_config; 4648 int parse; 4649 vdev_t *rvd; 4650 uint64_t pool_guid; 4651 const char *comment; 4652 const char *compatibility; 4653 4654 /* 4655 * Versioning wasn't explicitly added to the label until later, so if 4656 * it's not present treat it as the initial version. 4657 */ 4658 if (nvlist_lookup_uint64(config, ZPOOL_CONFIG_VERSION, 4659 &spa->spa_ubsync.ub_version) != 0) 4660 spa->spa_ubsync.ub_version = SPA_VERSION_INITIAL; 4661 4662 if (nvlist_lookup_uint64(config, ZPOOL_CONFIG_POOL_GUID, &pool_guid)) { 4663 spa_load_failed(spa, "invalid config provided: '%s' missing", 4664 ZPOOL_CONFIG_POOL_GUID); 4665 return (SET_ERROR(EINVAL)); 4666 } 4667 4668 /* 4669 * If we are doing an import, ensure that the pool is not already 4670 * imported by checking if its pool guid already exists in the 4671 * spa namespace. 4672 * 4673 * The only case that we allow an already imported pool to be 4674 * imported again, is when the pool is checkpointed and we want to 4675 * look at its checkpointed state from userland tools like zdb. 4676 */ 4677 #ifdef _KERNEL 4678 if ((spa->spa_load_state == SPA_LOAD_IMPORT || 4679 spa->spa_load_state == SPA_LOAD_TRYIMPORT) && 4680 spa_guid_exists(pool_guid, 0)) { 4681 #else 4682 if ((spa->spa_load_state == SPA_LOAD_IMPORT || 4683 spa->spa_load_state == SPA_LOAD_TRYIMPORT) && 4684 spa_guid_exists(pool_guid, 0) && 4685 !spa_importing_readonly_checkpoint(spa)) { 4686 #endif 4687 spa_load_failed(spa, "a pool with guid %llu is already open", 4688 (u_longlong_t)pool_guid); 4689 return (SET_ERROR(EEXIST)); 4690 } 4691 4692 spa->spa_config_guid = pool_guid; 4693 4694 nvlist_free(spa->spa_load_info); 4695 spa->spa_load_info = fnvlist_alloc(); 4696 4697 ASSERT0P(spa->spa_comment); 4698 if (nvlist_lookup_string(config, ZPOOL_CONFIG_COMMENT, &comment) == 0) 4699 spa->spa_comment = spa_strdup(comment); 4700 4701 ASSERT0P(spa->spa_compatibility); 4702 if (nvlist_lookup_string(config, ZPOOL_CONFIG_COMPATIBILITY, 4703 &compatibility) == 0) 4704 spa->spa_compatibility = spa_strdup(compatibility); 4705 4706 (void) nvlist_lookup_uint64(config, ZPOOL_CONFIG_POOL_TXG, 4707 &spa->spa_config_txg); 4708 4709 if (nvlist_lookup_nvlist(config, ZPOOL_CONFIG_SPLIT, &nvl) == 0) 4710 spa->spa_config_splitting = fnvlist_dup(nvl); 4711 4712 if (nvlist_lookup_nvlist(config, ZPOOL_CONFIG_VDEV_TREE, &nvtree)) { 4713 spa_load_failed(spa, "invalid config provided: '%s' missing", 4714 ZPOOL_CONFIG_VDEV_TREE); 4715 return (SET_ERROR(EINVAL)); 4716 } 4717 4718 /* 4719 * Create "The Godfather" zio to hold all async IOs 4720 */ 4721 spa->spa_async_zio_root = kmem_alloc(max_ncpus * sizeof (void *), 4722 KM_SLEEP); 4723 for (int i = 0; i < max_ncpus; i++) { 4724 spa->spa_async_zio_root[i] = zio_root(spa, NULL, NULL, 4725 ZIO_FLAG_CANFAIL | ZIO_FLAG_SPECULATIVE | 4726 ZIO_FLAG_GODFATHER); 4727 } 4728 4729 /* 4730 * Parse the configuration into a vdev tree. We explicitly set the 4731 * value that will be returned by spa_version() since parsing the 4732 * configuration requires knowing the version number. 4733 */ 4734 spa_config_enter(spa, SCL_ALL, FTAG, RW_WRITER); 4735 parse = (type == SPA_IMPORT_EXISTING ? 4736 VDEV_ALLOC_LOAD : VDEV_ALLOC_SPLIT); 4737 error = spa_config_parse(spa, &rvd, nvtree, NULL, 0, parse); 4738 spa_config_exit(spa, SCL_ALL, FTAG); 4739 4740 if (error != 0) { 4741 spa_load_failed(spa, "unable to parse config [error=%d]", 4742 error); 4743 return (error); 4744 } 4745 4746 ASSERT(spa->spa_root_vdev == rvd); 4747 ASSERT3U(spa->spa_min_ashift, >=, SPA_MINBLOCKSHIFT); 4748 ASSERT3U(spa->spa_max_ashift, <=, SPA_MAXBLOCKSHIFT); 4749 4750 if (type != SPA_IMPORT_ASSEMBLE) { 4751 ASSERT(spa_guid(spa) == pool_guid); 4752 } 4753 4754 return (0); 4755 } 4756 4757 /* 4758 * Recursively open all vdevs in the vdev tree. This function is called twice: 4759 * first with the untrusted config, then with the trusted config. 4760 */ 4761 static int 4762 spa_ld_open_vdevs(spa_t *spa) 4763 { 4764 int error = 0; 4765 4766 /* 4767 * spa_missing_tvds_allowed defines how many top-level vdevs can be 4768 * missing/unopenable for the root vdev to be still considered openable. 4769 */ 4770 if (spa->spa_trust_config) { 4771 spa->spa_missing_tvds_allowed = zfs_max_missing_tvds; 4772 } else if (spa->spa_config_source == SPA_CONFIG_SRC_CACHEFILE) { 4773 spa->spa_missing_tvds_allowed = zfs_max_missing_tvds_cachefile; 4774 } else if (spa->spa_config_source == SPA_CONFIG_SRC_SCAN) { 4775 spa->spa_missing_tvds_allowed = zfs_max_missing_tvds_scan; 4776 } else { 4777 spa->spa_missing_tvds_allowed = 0; 4778 } 4779 4780 spa->spa_missing_tvds_allowed = 4781 MAX(zfs_max_missing_tvds, spa->spa_missing_tvds_allowed); 4782 4783 spa_config_enter(spa, SCL_ALL, FTAG, RW_WRITER); 4784 error = vdev_open(spa->spa_root_vdev, CRED()); 4785 spa_config_exit(spa, SCL_ALL, FTAG); 4786 4787 if (spa->spa_missing_tvds != 0) { 4788 spa_load_note(spa, "vdev tree has %lld missing top-level " 4789 "vdevs.", (u_longlong_t)spa->spa_missing_tvds); 4790 if (spa->spa_trust_config && (spa->spa_mode & SPA_MODE_WRITE)) { 4791 /* 4792 * Although theoretically we could allow users to open 4793 * incomplete pools in RW mode, we'd need to add a lot 4794 * of extra logic (e.g. adjust pool space to account 4795 * for missing vdevs). 4796 * This limitation also prevents users from accidentally 4797 * opening the pool in RW mode during data recovery and 4798 * damaging it further. 4799 */ 4800 spa_load_note(spa, "pools with missing top-level " 4801 "vdevs can only be opened in read-only mode."); 4802 error = SET_ERROR(ENXIO); 4803 } else { 4804 spa_load_note(spa, "current settings allow for maximum " 4805 "%lld missing top-level vdevs at this stage.", 4806 (u_longlong_t)spa->spa_missing_tvds_allowed); 4807 } 4808 } 4809 if (error != 0) { 4810 spa_load_failed(spa, "unable to open vdev tree [error=%d]", 4811 error); 4812 } 4813 if (spa->spa_missing_tvds != 0 || error != 0) 4814 vdev_dbgmsg_print_tree(spa->spa_root_vdev, 2); 4815 4816 return (error); 4817 } 4818 4819 /* 4820 * We need to validate the vdev labels against the configuration that 4821 * we have in hand. This function is called twice: first with an untrusted 4822 * config, then with a trusted config. The validation is more strict when the 4823 * config is trusted. 4824 */ 4825 static int 4826 spa_ld_validate_vdevs(spa_t *spa) 4827 { 4828 int error = 0; 4829 vdev_t *rvd = spa->spa_root_vdev; 4830 4831 spa_config_enter(spa, SCL_ALL, FTAG, RW_WRITER); 4832 error = vdev_validate(rvd); 4833 spa_config_exit(spa, SCL_ALL, FTAG); 4834 4835 if (error != 0) { 4836 spa_load_failed(spa, "vdev_validate failed [error=%d]", error); 4837 return (error); 4838 } 4839 4840 if (rvd->vdev_state <= VDEV_STATE_CANT_OPEN) { 4841 spa_load_failed(spa, "cannot open vdev tree after invalidating " 4842 "some vdevs"); 4843 vdev_dbgmsg_print_tree(rvd, 2); 4844 return (SET_ERROR(ENXIO)); 4845 } 4846 4847 return (0); 4848 } 4849 4850 static void 4851 spa_ld_select_uberblock_done(spa_t *spa, uberblock_t *ub) 4852 { 4853 spa->spa_state = POOL_STATE_ACTIVE; 4854 spa->spa_ubsync = spa->spa_uberblock; 4855 spa->spa_verify_min_txg = spa->spa_extreme_rewind ? 4856 TXG_INITIAL - 1 : spa_last_synced_txg(spa) - TXG_DEFER_SIZE - 1; 4857 spa->spa_first_txg = spa->spa_last_ubsync_txg ? 4858 spa->spa_last_ubsync_txg : spa_last_synced_txg(spa) + 1; 4859 spa->spa_claim_max_txg = spa->spa_first_txg; 4860 spa->spa_prev_software_version = ub->ub_software_version; 4861 } 4862 4863 static int 4864 spa_ld_select_uberblock(spa_t *spa, spa_import_type_t type) 4865 { 4866 vdev_t *rvd = spa->spa_root_vdev; 4867 nvlist_t *label; 4868 uberblock_t *ub = &spa->spa_uberblock; 4869 4870 /* 4871 * If we are opening the checkpointed state of the pool by 4872 * rewinding to it, at this point we will have written the 4873 * checkpointed uberblock to the vdev labels, so searching 4874 * the labels will find the right uberblock. However, if 4875 * we are opening the checkpointed state read-only, we have 4876 * not modified the labels. Therefore, we must ignore the 4877 * labels and continue using the spa_uberblock that was set 4878 * by spa_ld_checkpoint_rewind. 4879 * 4880 * Note that it would be fine to ignore the labels when 4881 * rewinding (opening writeable) as well. However, if we 4882 * crash just after writing the labels, we will end up 4883 * searching the labels. Doing so in the common case means 4884 * that this code path gets exercised normally, rather than 4885 * just in the edge case. 4886 */ 4887 if (ub->ub_checkpoint_txg != 0 && 4888 spa_importing_readonly_checkpoint(spa)) { 4889 spa_ld_select_uberblock_done(spa, ub); 4890 return (0); 4891 } 4892 4893 /* 4894 * Find the best uberblock. 4895 */ 4896 vdev_uberblock_load(rvd, ub, &label); 4897 4898 /* 4899 * If we weren't able to find a single valid uberblock, return failure. 4900 */ 4901 if (ub->ub_txg == 0) { 4902 nvlist_free(label); 4903 spa_load_failed(spa, "no valid uberblock found"); 4904 return (spa_vdev_err(rvd, VDEV_AUX_CORRUPT_DATA, ENXIO)); 4905 } 4906 4907 if (spa->spa_load_max_txg != UINT64_MAX) { 4908 (void) spa_import_progress_set_max_txg(spa_guid(spa), 4909 (u_longlong_t)spa->spa_load_max_txg); 4910 } 4911 spa_load_note(spa, "using uberblock with txg=%llu", 4912 (u_longlong_t)ub->ub_txg); 4913 if (ub->ub_raidz_reflow_info != 0) { 4914 spa_load_note(spa, "uberblock raidz_reflow_info: " 4915 "state=%u offset=%llu", 4916 (int)RRSS_GET_STATE(ub), 4917 (u_longlong_t)RRSS_GET_OFFSET(ub)); 4918 } 4919 4920 /* 4921 * For pools which have the multihost property on determine if the 4922 * pool is truly inactive and can be safely imported. Prevent 4923 * hosts which don't have a hostid set from importing the pool. 4924 */ 4925 spa->spa_activity_check = spa_activity_check_required(spa, ub, label); 4926 if (spa->spa_activity_check) { 4927 int error = spa_ld_activity_check(spa, ub, label); 4928 if (error) { 4929 spa_load_state_t state = spa->spa_load_state; 4930 error = spa_ld_activity_result(spa, error, 4931 state == SPA_LOAD_TRYIMPORT ? "tryimport" : 4932 state == SPA_LOAD_IMPORT ? "import" : "open"); 4933 nvlist_free(label); 4934 return (error); 4935 } 4936 } else { 4937 fnvlist_add_uint32(spa->spa_load_info, 4938 ZPOOL_CONFIG_MMP_RESULT, ESRCH); 4939 } 4940 4941 /* 4942 * If the pool has an unsupported version we can't open it. 4943 */ 4944 if (!SPA_VERSION_IS_SUPPORTED(ub->ub_version)) { 4945 nvlist_free(label); 4946 spa_load_failed(spa, "version %llu is not supported", 4947 (u_longlong_t)ub->ub_version); 4948 return (spa_vdev_err(rvd, VDEV_AUX_VERSION_NEWER, ENOTSUP)); 4949 } 4950 4951 if (ub->ub_version >= SPA_VERSION_FEATURES) { 4952 nvlist_t *features; 4953 4954 /* 4955 * If we weren't able to find what's necessary for reading the 4956 * MOS in the label, return failure. 4957 */ 4958 if (label == NULL) { 4959 spa_load_failed(spa, "label config unavailable"); 4960 return (spa_vdev_err(rvd, VDEV_AUX_CORRUPT_DATA, 4961 ENXIO)); 4962 } 4963 4964 if (nvlist_lookup_nvlist(label, ZPOOL_CONFIG_FEATURES_FOR_READ, 4965 &features) != 0) { 4966 nvlist_free(label); 4967 spa_load_failed(spa, "invalid label: '%s' missing", 4968 ZPOOL_CONFIG_FEATURES_FOR_READ); 4969 return (spa_vdev_err(rvd, VDEV_AUX_CORRUPT_DATA, 4970 ENXIO)); 4971 } 4972 4973 /* 4974 * Update our in-core representation with the definitive values 4975 * from the label. 4976 */ 4977 nvlist_free(spa->spa_label_features); 4978 spa->spa_label_features = fnvlist_dup(features); 4979 } 4980 4981 nvlist_free(label); 4982 4983 /* 4984 * Look through entries in the label nvlist's features_for_read. If 4985 * there is a feature listed there which we don't understand then we 4986 * cannot open a pool. 4987 */ 4988 if (ub->ub_version >= SPA_VERSION_FEATURES) { 4989 nvlist_t *unsup_feat; 4990 4991 unsup_feat = fnvlist_alloc(); 4992 4993 for (nvpair_t *nvp = nvlist_next_nvpair(spa->spa_label_features, 4994 NULL); nvp != NULL; 4995 nvp = nvlist_next_nvpair(spa->spa_label_features, nvp)) { 4996 if (!zfeature_is_supported(nvpair_name(nvp))) { 4997 fnvlist_add_string(unsup_feat, 4998 nvpair_name(nvp), ""); 4999 } 5000 } 5001 5002 if (!nvlist_empty(unsup_feat)) { 5003 fnvlist_add_nvlist(spa->spa_load_info, 5004 ZPOOL_CONFIG_UNSUP_FEAT, unsup_feat); 5005 nvlist_free(unsup_feat); 5006 spa_load_failed(spa, "some features are unsupported"); 5007 return (spa_vdev_err(rvd, VDEV_AUX_UNSUP_FEAT, 5008 ENOTSUP)); 5009 } 5010 5011 nvlist_free(unsup_feat); 5012 } 5013 5014 if (type != SPA_IMPORT_ASSEMBLE && spa->spa_config_splitting) { 5015 spa_config_enter(spa, SCL_ALL, FTAG, RW_WRITER); 5016 spa_try_repair(spa, spa->spa_config); 5017 spa_config_exit(spa, SCL_ALL, FTAG); 5018 nvlist_free(spa->spa_config_splitting); 5019 spa->spa_config_splitting = NULL; 5020 } 5021 5022 /* 5023 * Initialize internal SPA structures. 5024 */ 5025 spa_ld_select_uberblock_done(spa, ub); 5026 5027 return (0); 5028 } 5029 5030 static int 5031 spa_ld_open_rootbp(spa_t *spa) 5032 { 5033 int error = 0; 5034 vdev_t *rvd = spa->spa_root_vdev; 5035 5036 error = dsl_pool_init(spa, spa->spa_first_txg, &spa->spa_dsl_pool); 5037 if (error != 0) { 5038 spa_load_failed(spa, "unable to open rootbp in dsl_pool_init " 5039 "[error=%d]", error); 5040 return (spa_vdev_err(rvd, VDEV_AUX_CORRUPT_DATA, EIO)); 5041 } 5042 spa->spa_meta_objset = spa->spa_dsl_pool->dp_meta_objset; 5043 5044 return (0); 5045 } 5046 5047 static int 5048 spa_ld_trusted_config(spa_t *spa, spa_import_type_t type, 5049 boolean_t reloading) 5050 { 5051 vdev_t *mrvd, *rvd = spa->spa_root_vdev; 5052 nvlist_t *nv, *mos_config, *policy; 5053 int error = 0, copy_error; 5054 uint64_t healthy_tvds, healthy_tvds_mos; 5055 uint64_t mos_config_txg; 5056 5057 if (spa_dir_prop(spa, DMU_POOL_CONFIG, &spa->spa_config_object, B_TRUE) 5058 != 0) 5059 return (spa_vdev_err(rvd, VDEV_AUX_CORRUPT_DATA, EIO)); 5060 5061 /* 5062 * If we're assembling a pool from a split, the config provided is 5063 * already trusted so there is nothing to do. 5064 */ 5065 if (type == SPA_IMPORT_ASSEMBLE) 5066 return (0); 5067 5068 healthy_tvds = spa_healthy_core_tvds(spa); 5069 5070 if (load_nvlist(spa, spa->spa_config_object, &mos_config) 5071 != 0) { 5072 spa_load_failed(spa, "unable to retrieve MOS config"); 5073 return (spa_vdev_err(rvd, VDEV_AUX_CORRUPT_DATA, EIO)); 5074 } 5075 5076 /* 5077 * If we are doing an open, pool owner wasn't verified yet, thus do 5078 * the verification here. 5079 */ 5080 if (spa->spa_load_state == SPA_LOAD_OPEN) { 5081 error = spa_verify_host(spa, mos_config); 5082 if (error != 0) { 5083 nvlist_free(mos_config); 5084 return (error); 5085 } 5086 } 5087 5088 nv = fnvlist_lookup_nvlist(mos_config, ZPOOL_CONFIG_VDEV_TREE); 5089 5090 spa_config_enter(spa, SCL_ALL, FTAG, RW_WRITER); 5091 5092 /* 5093 * Build a new vdev tree from the trusted config 5094 */ 5095 error = spa_config_parse(spa, &mrvd, nv, NULL, 0, VDEV_ALLOC_LOAD); 5096 if (error != 0) { 5097 nvlist_free(mos_config); 5098 spa_config_exit(spa, SCL_ALL, FTAG); 5099 spa_load_failed(spa, "spa_config_parse failed [error=%d]", 5100 error); 5101 return (spa_vdev_err(rvd, VDEV_AUX_CORRUPT_DATA, error)); 5102 } 5103 5104 /* 5105 * Vdev paths in the MOS may be obsolete. If the untrusted config was 5106 * obtained by scanning /dev/dsk, then it will have the right vdev 5107 * paths. We update the trusted MOS config with this information. 5108 * We first try to copy the paths with vdev_copy_path_strict, which 5109 * succeeds only when both configs have exactly the same vdev tree. 5110 * If that fails, we fall back to a more flexible method that has a 5111 * best effort policy. 5112 */ 5113 copy_error = vdev_copy_path_strict(rvd, mrvd); 5114 if (copy_error != 0 || spa_load_print_vdev_tree) { 5115 spa_load_note(spa, "provided vdev tree:"); 5116 vdev_dbgmsg_print_tree(rvd, 2); 5117 spa_load_note(spa, "MOS vdev tree:"); 5118 vdev_dbgmsg_print_tree(mrvd, 2); 5119 } 5120 if (copy_error != 0) { 5121 spa_load_note(spa, "vdev_copy_path_strict failed, falling " 5122 "back to vdev_copy_path_relaxed"); 5123 vdev_copy_path_relaxed(rvd, mrvd); 5124 } 5125 5126 vdev_close(rvd); 5127 vdev_free(rvd); 5128 spa->spa_root_vdev = mrvd; 5129 rvd = mrvd; 5130 spa_config_exit(spa, SCL_ALL, FTAG); 5131 5132 /* 5133 * If 'zpool import' used a cached config, then the on-disk hostid and 5134 * hostname may be different to the cached config in ways that should 5135 * prevent import. Userspace can't discover this without a scan, but 5136 * we know, so we add these values to LOAD_INFO so the caller can know 5137 * the difference. 5138 * 5139 * Note that we have to do this before the config is regenerated, 5140 * because the new config will have the hostid and hostname for this 5141 * host, in readiness for import. 5142 */ 5143 if (nvlist_exists(mos_config, ZPOOL_CONFIG_HOSTID)) 5144 fnvlist_add_uint64(spa->spa_load_info, ZPOOL_CONFIG_HOSTID, 5145 fnvlist_lookup_uint64(mos_config, ZPOOL_CONFIG_HOSTID)); 5146 if (nvlist_exists(mos_config, ZPOOL_CONFIG_HOSTNAME)) 5147 fnvlist_add_string(spa->spa_load_info, ZPOOL_CONFIG_HOSTNAME, 5148 fnvlist_lookup_string(mos_config, ZPOOL_CONFIG_HOSTNAME)); 5149 5150 /* 5151 * We will use spa_config if we decide to reload the spa or if spa_load 5152 * fails and we rewind. We must thus regenerate the config using the 5153 * MOS information with the updated paths. ZPOOL_LOAD_POLICY is used to 5154 * pass settings on how to load the pool and is not stored in the MOS. 5155 * We copy it over to our new, trusted config. 5156 */ 5157 mos_config_txg = fnvlist_lookup_uint64(mos_config, 5158 ZPOOL_CONFIG_POOL_TXG); 5159 nvlist_free(mos_config); 5160 mos_config = spa_config_generate(spa, NULL, mos_config_txg, B_FALSE); 5161 if (nvlist_lookup_nvlist(spa->spa_config, ZPOOL_LOAD_POLICY, 5162 &policy) == 0) 5163 fnvlist_add_nvlist(mos_config, ZPOOL_LOAD_POLICY, policy); 5164 spa_config_set(spa, mos_config); 5165 spa->spa_config_source = SPA_CONFIG_SRC_MOS; 5166 5167 /* 5168 * Now that we got the config from the MOS, we should be more strict 5169 * in checking blkptrs and can make assumptions about the consistency 5170 * of the vdev tree. spa_trust_config must be set to true before opening 5171 * vdevs in order for them to be writeable. 5172 */ 5173 spa->spa_trust_config = B_TRUE; 5174 5175 /* 5176 * Open and validate the new vdev tree 5177 */ 5178 error = spa_ld_open_vdevs(spa); 5179 if (error != 0) 5180 return (error); 5181 5182 error = spa_ld_validate_vdevs(spa); 5183 if (error != 0) 5184 return (error); 5185 5186 if (copy_error != 0 || spa_load_print_vdev_tree) { 5187 spa_load_note(spa, "final vdev tree:"); 5188 vdev_dbgmsg_print_tree(rvd, 2); 5189 } 5190 5191 if (spa->spa_load_state != SPA_LOAD_TRYIMPORT && 5192 !spa->spa_extreme_rewind && zfs_max_missing_tvds == 0) { 5193 /* 5194 * Sanity check to make sure that we are indeed loading the 5195 * latest uberblock. If we missed SPA_SYNC_MIN_VDEVS tvds 5196 * in the config provided and they happened to be the only ones 5197 * to have the latest uberblock, we could involuntarily perform 5198 * an extreme rewind. 5199 */ 5200 healthy_tvds_mos = spa_healthy_core_tvds(spa); 5201 if (healthy_tvds_mos - healthy_tvds >= 5202 SPA_SYNC_MIN_VDEVS) { 5203 spa_load_note(spa, "config provided misses too many " 5204 "top-level vdevs compared to MOS (%lld vs %lld). ", 5205 (u_longlong_t)healthy_tvds, 5206 (u_longlong_t)healthy_tvds_mos); 5207 spa_load_note(spa, "vdev tree:"); 5208 vdev_dbgmsg_print_tree(rvd, 2); 5209 if (reloading) { 5210 spa_load_failed(spa, "config was already " 5211 "provided from MOS. Aborting."); 5212 return (spa_vdev_err(rvd, 5213 VDEV_AUX_CORRUPT_DATA, EIO)); 5214 } 5215 spa_load_note(spa, "spa must be reloaded using MOS " 5216 "config"); 5217 return (SET_ERROR(EAGAIN)); 5218 } 5219 } 5220 5221 /* 5222 * Final sanity check for multihost pools that no other host is 5223 * accessing the pool. All of the read-only check have passed at 5224 * this point, perform targetted updates to the mmp uberblocks to 5225 * safely force a visible change. 5226 */ 5227 if (spa->spa_load_state != SPA_LOAD_TRYIMPORT && 5228 !spa->spa_extreme_rewind && spa->spa_activity_check) { 5229 5230 error = spa_activity_check_claim(spa); 5231 error = spa_ld_activity_result(spa, error, "claim"); 5232 5233 if (error == EREMOTEIO) 5234 return (spa_vdev_err(rvd, VDEV_AUX_ACTIVE, EREMOTEIO)); 5235 else if (error) 5236 return (error); 5237 } 5238 5239 error = spa_check_for_missing_logs(spa); 5240 if (error != 0) 5241 return (spa_vdev_err(rvd, VDEV_AUX_BAD_GUID_SUM, ENXIO)); 5242 5243 if (rvd->vdev_guid_sum != spa->spa_uberblock.ub_guid_sum) { 5244 spa_load_failed(spa, "uberblock guid sum doesn't match MOS " 5245 "guid sum (%llu != %llu)", 5246 (u_longlong_t)spa->spa_uberblock.ub_guid_sum, 5247 (u_longlong_t)rvd->vdev_guid_sum); 5248 return (spa_vdev_err(rvd, VDEV_AUX_BAD_GUID_SUM, 5249 ENXIO)); 5250 } 5251 5252 return (0); 5253 } 5254 5255 static int 5256 spa_ld_open_indirect_vdev_metadata(spa_t *spa) 5257 { 5258 int error = 0; 5259 vdev_t *rvd = spa->spa_root_vdev; 5260 5261 /* 5262 * Everything that we read before spa_remove_init() must be stored 5263 * on concreted vdevs. Therefore we do this as early as possible. 5264 */ 5265 error = spa_remove_init(spa); 5266 if (error != 0) { 5267 spa_load_failed(spa, "spa_remove_init failed [error=%d]", 5268 error); 5269 return (spa_vdev_err(rvd, VDEV_AUX_CORRUPT_DATA, EIO)); 5270 } 5271 5272 /* 5273 * Retrieve information needed to condense indirect vdev mappings. 5274 */ 5275 error = spa_condense_init(spa); 5276 if (error != 0) { 5277 spa_load_failed(spa, "spa_condense_init failed [error=%d]", 5278 error); 5279 return (spa_vdev_err(rvd, VDEV_AUX_CORRUPT_DATA, error)); 5280 } 5281 5282 return (0); 5283 } 5284 5285 static int 5286 spa_ld_check_features(spa_t *spa, boolean_t *missing_feat_writep) 5287 { 5288 int error = 0; 5289 vdev_t *rvd = spa->spa_root_vdev; 5290 5291 if (spa_version(spa) >= SPA_VERSION_FEATURES) { 5292 boolean_t missing_feat_read = B_FALSE; 5293 nvlist_t *unsup_feat, *enabled_feat; 5294 5295 if (spa_dir_prop(spa, DMU_POOL_FEATURES_FOR_READ, 5296 &spa->spa_feat_for_read_obj, B_TRUE) != 0) { 5297 return (spa_vdev_err(rvd, VDEV_AUX_CORRUPT_DATA, EIO)); 5298 } 5299 5300 if (spa_dir_prop(spa, DMU_POOL_FEATURES_FOR_WRITE, 5301 &spa->spa_feat_for_write_obj, B_TRUE) != 0) { 5302 return (spa_vdev_err(rvd, VDEV_AUX_CORRUPT_DATA, EIO)); 5303 } 5304 5305 if (spa_dir_prop(spa, DMU_POOL_FEATURE_DESCRIPTIONS, 5306 &spa->spa_feat_desc_obj, B_TRUE) != 0) { 5307 return (spa_vdev_err(rvd, VDEV_AUX_CORRUPT_DATA, EIO)); 5308 } 5309 5310 enabled_feat = fnvlist_alloc(); 5311 unsup_feat = fnvlist_alloc(); 5312 5313 if (!spa_features_check(spa, B_FALSE, 5314 unsup_feat, enabled_feat)) 5315 missing_feat_read = B_TRUE; 5316 5317 if (spa_writeable(spa) || 5318 spa->spa_load_state == SPA_LOAD_TRYIMPORT) { 5319 if (!spa_features_check(spa, B_TRUE, 5320 unsup_feat, enabled_feat)) { 5321 *missing_feat_writep = B_TRUE; 5322 } 5323 } 5324 5325 fnvlist_add_nvlist(spa->spa_load_info, 5326 ZPOOL_CONFIG_ENABLED_FEAT, enabled_feat); 5327 5328 if (!nvlist_empty(unsup_feat)) { 5329 fnvlist_add_nvlist(spa->spa_load_info, 5330 ZPOOL_CONFIG_UNSUP_FEAT, unsup_feat); 5331 } 5332 5333 fnvlist_free(enabled_feat); 5334 fnvlist_free(unsup_feat); 5335 5336 if (!missing_feat_read) { 5337 fnvlist_add_boolean(spa->spa_load_info, 5338 ZPOOL_CONFIG_CAN_RDONLY); 5339 } 5340 5341 /* 5342 * If the state is SPA_LOAD_TRYIMPORT, our objective is 5343 * twofold: to determine whether the pool is available for 5344 * import in read-write mode and (if it is not) whether the 5345 * pool is available for import in read-only mode. If the pool 5346 * is available for import in read-write mode, it is displayed 5347 * as available in userland; if it is not available for import 5348 * in read-only mode, it is displayed as unavailable in 5349 * userland. If the pool is available for import in read-only 5350 * mode but not read-write mode, it is displayed as unavailable 5351 * in userland with a special note that the pool is actually 5352 * available for open in read-only mode. 5353 * 5354 * As a result, if the state is SPA_LOAD_TRYIMPORT and we are 5355 * missing a feature for write, we must first determine whether 5356 * the pool can be opened read-only before returning to 5357 * userland in order to know whether to display the 5358 * abovementioned note. 5359 */ 5360 if (missing_feat_read || (*missing_feat_writep && 5361 spa_writeable(spa))) { 5362 spa_load_failed(spa, "pool uses unsupported features"); 5363 return (spa_vdev_err(rvd, VDEV_AUX_UNSUP_FEAT, 5364 ENOTSUP)); 5365 } 5366 5367 /* 5368 * Load refcounts for ZFS features from disk into an in-memory 5369 * cache during SPA initialization. 5370 */ 5371 for (spa_feature_t i = 0; i < SPA_FEATURES; i++) { 5372 uint64_t refcount; 5373 5374 error = feature_get_refcount_from_disk(spa, 5375 &spa_feature_table[i], &refcount); 5376 if (error == 0) { 5377 spa->spa_feat_refcount_cache[i] = refcount; 5378 } else if (error == ENOTSUP) { 5379 spa->spa_feat_refcount_cache[i] = 5380 SPA_FEATURE_DISABLED; 5381 } else { 5382 spa_load_failed(spa, "error getting refcount " 5383 "for feature %s [error=%d]", 5384 spa_feature_table[i].fi_guid, error); 5385 return (spa_vdev_err(rvd, 5386 VDEV_AUX_CORRUPT_DATA, EIO)); 5387 } 5388 } 5389 } 5390 5391 if (spa_feature_is_active(spa, SPA_FEATURE_ENABLED_TXG)) { 5392 if (spa_dir_prop(spa, DMU_POOL_FEATURE_ENABLED_TXG, 5393 &spa->spa_feat_enabled_txg_obj, B_TRUE) != 0) 5394 return (spa_vdev_err(rvd, VDEV_AUX_CORRUPT_DATA, EIO)); 5395 } 5396 5397 /* 5398 * Encryption was added before bookmark_v2, even though bookmark_v2 5399 * is now a dependency. If this pool has encryption enabled without 5400 * bookmark_v2, trigger an errata message. 5401 */ 5402 if (spa_feature_is_enabled(spa, SPA_FEATURE_ENCRYPTION) && 5403 !spa_feature_is_enabled(spa, SPA_FEATURE_BOOKMARK_V2)) { 5404 spa->spa_errata = ZPOOL_ERRATA_ZOL_8308_ENCRYPTION; 5405 } 5406 5407 return (0); 5408 } 5409 5410 static int 5411 spa_ld_load_special_directories(spa_t *spa) 5412 { 5413 int error = 0; 5414 vdev_t *rvd = spa->spa_root_vdev; 5415 5416 spa->spa_is_initializing = B_TRUE; 5417 error = dsl_pool_open(spa->spa_dsl_pool); 5418 spa->spa_is_initializing = B_FALSE; 5419 if (error != 0) { 5420 spa_load_failed(spa, "dsl_pool_open failed [error=%d]", error); 5421 return (spa_vdev_err(rvd, VDEV_AUX_CORRUPT_DATA, EIO)); 5422 } 5423 5424 return (0); 5425 } 5426 5427 static int 5428 spa_ld_get_props(spa_t *spa) 5429 { 5430 int error = 0; 5431 uint64_t obj; 5432 vdev_t *rvd = spa->spa_root_vdev; 5433 5434 /* Grab the checksum salt from the MOS. */ 5435 error = zap_lookup(spa->spa_meta_objset, DMU_POOL_DIRECTORY_OBJECT, 5436 DMU_POOL_CHECKSUM_SALT, 1, 5437 sizeof (spa->spa_cksum_salt.zcs_bytes), 5438 spa->spa_cksum_salt.zcs_bytes); 5439 if (error == ENOENT) { 5440 /* Generate a new salt for subsequent use */ 5441 (void) random_get_pseudo_bytes(spa->spa_cksum_salt.zcs_bytes, 5442 sizeof (spa->spa_cksum_salt.zcs_bytes)); 5443 } else if (error != 0) { 5444 spa_load_failed(spa, "unable to retrieve checksum salt from " 5445 "MOS [error=%d]", error); 5446 return (spa_vdev_err(rvd, VDEV_AUX_CORRUPT_DATA, EIO)); 5447 } 5448 5449 if (spa_dir_prop(spa, DMU_POOL_SYNC_BPOBJ, &obj, B_TRUE) != 0) 5450 return (spa_vdev_err(rvd, VDEV_AUX_CORRUPT_DATA, EIO)); 5451 error = bpobj_open(&spa->spa_deferred_bpobj, spa->spa_meta_objset, obj); 5452 if (error != 0) { 5453 spa_load_failed(spa, "error opening deferred-frees bpobj " 5454 "[error=%d]", error); 5455 return (spa_vdev_err(rvd, VDEV_AUX_CORRUPT_DATA, EIO)); 5456 } 5457 5458 /* 5459 * Load the bit that tells us to use the new accounting function 5460 * (raid-z deflation). If we have an older pool, this will not 5461 * be present. 5462 */ 5463 error = spa_dir_prop(spa, DMU_POOL_DEFLATE, &spa->spa_deflate, B_FALSE); 5464 if (error != 0 && error != ENOENT) 5465 return (spa_vdev_err(rvd, VDEV_AUX_CORRUPT_DATA, EIO)); 5466 5467 error = spa_dir_prop(spa, DMU_POOL_CREATION_VERSION, 5468 &spa->spa_creation_version, B_FALSE); 5469 if (error != 0 && error != ENOENT) 5470 return (spa_vdev_err(rvd, VDEV_AUX_CORRUPT_DATA, EIO)); 5471 5472 /* Load time log */ 5473 spa_load_txg_log_time(spa); 5474 5475 /* 5476 * Load the persistent error log. If we have an older pool, this will 5477 * not be present. 5478 */ 5479 error = spa_dir_prop(spa, DMU_POOL_ERRLOG_LAST, &spa->spa_errlog_last, 5480 B_FALSE); 5481 if (error != 0 && error != ENOENT) 5482 return (spa_vdev_err(rvd, VDEV_AUX_CORRUPT_DATA, EIO)); 5483 5484 error = spa_dir_prop(spa, DMU_POOL_ERRLOG_SCRUB, 5485 &spa->spa_errlog_scrub, B_FALSE); 5486 if (error != 0 && error != ENOENT) 5487 return (spa_vdev_err(rvd, VDEV_AUX_CORRUPT_DATA, EIO)); 5488 5489 /* Load the last scrubbed txg. */ 5490 error = spa_dir_prop(spa, DMU_POOL_LAST_SCRUBBED_TXG, 5491 &spa->spa_scrubbed_last_txg, B_FALSE); 5492 if (error != 0 && error != ENOENT) 5493 return (spa_vdev_err(rvd, VDEV_AUX_CORRUPT_DATA, EIO)); 5494 5495 /* 5496 * Load the livelist deletion field. If a livelist is queued for 5497 * deletion, indicate that in the spa 5498 */ 5499 error = spa_dir_prop(spa, DMU_POOL_DELETED_CLONES, 5500 &spa->spa_livelists_to_delete, B_FALSE); 5501 if (error != 0 && error != ENOENT) 5502 return (spa_vdev_err(rvd, VDEV_AUX_CORRUPT_DATA, EIO)); 5503 5504 /* 5505 * Load the history object. If we have an older pool, this 5506 * will not be present. 5507 */ 5508 error = spa_dir_prop(spa, DMU_POOL_HISTORY, &spa->spa_history, B_FALSE); 5509 if (error != 0 && error != ENOENT) 5510 return (spa_vdev_err(rvd, VDEV_AUX_CORRUPT_DATA, EIO)); 5511 5512 /* 5513 * Load the per-vdev ZAP map. If we have an older pool, this will not 5514 * be present; in this case, defer its creation to a later time to 5515 * avoid dirtying the MOS this early / out of sync context. See 5516 * spa_sync_config_object. 5517 */ 5518 5519 /* The sentinel is only available in the MOS config. */ 5520 nvlist_t *mos_config; 5521 if (load_nvlist(spa, spa->spa_config_object, &mos_config) != 0) { 5522 spa_load_failed(spa, "unable to retrieve MOS config"); 5523 return (spa_vdev_err(rvd, VDEV_AUX_CORRUPT_DATA, EIO)); 5524 } 5525 5526 error = spa_dir_prop(spa, DMU_POOL_VDEV_ZAP_MAP, 5527 &spa->spa_all_vdev_zaps, B_FALSE); 5528 5529 if (error == ENOENT) { 5530 VERIFY(!nvlist_exists(mos_config, 5531 ZPOOL_CONFIG_HAS_PER_VDEV_ZAPS)); 5532 spa->spa_avz_action = AVZ_ACTION_INITIALIZE; 5533 ASSERT0(vdev_count_verify_zaps(spa->spa_root_vdev)); 5534 } else if (error != 0) { 5535 nvlist_free(mos_config); 5536 return (spa_vdev_err(rvd, VDEV_AUX_CORRUPT_DATA, EIO)); 5537 } else if (!nvlist_exists(mos_config, ZPOOL_CONFIG_HAS_PER_VDEV_ZAPS)) { 5538 /* 5539 * An older version of ZFS overwrote the sentinel value, so 5540 * we have orphaned per-vdev ZAPs in the MOS. Defer their 5541 * destruction to later; see spa_sync_config_object. 5542 */ 5543 spa->spa_avz_action = AVZ_ACTION_DESTROY; 5544 /* 5545 * We're assuming that no vdevs have had their ZAPs created 5546 * before this. Better be sure of it. 5547 */ 5548 ASSERT0(vdev_count_verify_zaps(spa->spa_root_vdev)); 5549 } 5550 nvlist_free(mos_config); 5551 5552 spa->spa_delegation = zpool_prop_default_numeric(ZPOOL_PROP_DELEGATION); 5553 5554 error = spa_dir_prop(spa, DMU_POOL_PROPS, &spa->spa_pool_props_object, 5555 B_FALSE); 5556 if (error && error != ENOENT) 5557 return (spa_vdev_err(rvd, VDEV_AUX_CORRUPT_DATA, EIO)); 5558 5559 if (error == 0) { 5560 uint64_t autoreplace = 0; 5561 5562 spa_prop_find(spa, ZPOOL_PROP_BOOTFS, &spa->spa_bootfs); 5563 spa_prop_find(spa, ZPOOL_PROP_AUTOREPLACE, &autoreplace); 5564 spa_prop_find(spa, ZPOOL_PROP_DELEGATION, &spa->spa_delegation); 5565 spa_prop_find(spa, ZPOOL_PROP_FAILUREMODE, &spa->spa_failmode); 5566 spa_prop_find(spa, ZPOOL_PROP_AUTOEXPAND, &spa->spa_autoexpand); 5567 spa_prop_find(spa, ZPOOL_PROP_DEDUP_TABLE_QUOTA, 5568 &spa->spa_dedup_table_quota); 5569 spa_prop_find(spa, ZPOOL_PROP_MULTIHOST, &spa->spa_multihost); 5570 spa_prop_find(spa, ZPOOL_PROP_AUTOTRIM, &spa->spa_autotrim); 5571 spa->spa_autoreplace = (autoreplace != 0); 5572 } 5573 5574 /* 5575 * If we are importing a pool with missing top-level vdevs, 5576 * we enforce that the pool doesn't panic or get suspended on 5577 * error since the likelihood of missing data is extremely high. 5578 */ 5579 if (spa->spa_missing_tvds > 0 && 5580 spa->spa_failmode != ZIO_FAILURE_MODE_CONTINUE && 5581 spa->spa_load_state != SPA_LOAD_TRYIMPORT) { 5582 spa_load_note(spa, "forcing failmode to 'continue' " 5583 "as some top level vdevs are missing"); 5584 spa->spa_failmode = ZIO_FAILURE_MODE_CONTINUE; 5585 } 5586 5587 return (0); 5588 } 5589 5590 static int 5591 spa_ld_open_aux_vdevs(spa_t *spa, spa_import_type_t type) 5592 { 5593 int error = 0; 5594 vdev_t *rvd = spa->spa_root_vdev; 5595 5596 /* 5597 * If we're assembling the pool from the split-off vdevs of 5598 * an existing pool, we don't want to attach the spares & cache 5599 * devices. 5600 */ 5601 5602 /* 5603 * Load any hot spares for this pool. 5604 */ 5605 error = spa_dir_prop(spa, DMU_POOL_SPARES, &spa->spa_spares.sav_object, 5606 B_FALSE); 5607 if (error != 0 && error != ENOENT) 5608 return (spa_vdev_err(rvd, VDEV_AUX_CORRUPT_DATA, EIO)); 5609 if (error == 0 && type != SPA_IMPORT_ASSEMBLE) { 5610 ASSERT(spa_version(spa) >= SPA_VERSION_SPARES); 5611 error = load_nvlist(spa, spa->spa_spares.sav_object, 5612 &spa->spa_spares.sav_config); 5613 if (error != 0) { 5614 if (!zfs_recover && spa_writeable(spa)) { 5615 spa_load_failed(spa, "error loading spares " 5616 "nvlist [error=%d]", error); 5617 return (spa_vdev_err(rvd, 5618 VDEV_AUX_CORRUPT_DATA, EIO)); 5619 } 5620 spa_load_note(spa, "ignoring spares nvlist " 5621 "[error=%d], no spares will be available", error); 5622 /* Leak the object, its dnode may be unreadable. */ 5623 spa->spa_spares.sav_object = 0; 5624 spa->spa_spares.sav_sync = B_TRUE; 5625 } else { 5626 spa_config_enter(spa, SCL_ALL, FTAG, RW_WRITER); 5627 spa_load_spares(spa); 5628 spa_config_exit(spa, SCL_ALL, FTAG); 5629 } 5630 } else if (error == 0) { 5631 spa->spa_spares.sav_sync = B_TRUE; 5632 } 5633 5634 /* 5635 * Load any level 2 ARC devices for this pool. 5636 */ 5637 error = spa_dir_prop(spa, DMU_POOL_L2CACHE, 5638 &spa->spa_l2cache.sav_object, B_FALSE); 5639 if (error != 0 && error != ENOENT) 5640 return (spa_vdev_err(rvd, VDEV_AUX_CORRUPT_DATA, EIO)); 5641 if (error == 0 && type != SPA_IMPORT_ASSEMBLE) { 5642 ASSERT(spa_version(spa) >= SPA_VERSION_L2CACHE); 5643 error = load_nvlist(spa, spa->spa_l2cache.sav_object, 5644 &spa->spa_l2cache.sav_config); 5645 if (error != 0) { 5646 if (!zfs_recover && spa_writeable(spa)) { 5647 spa_load_failed(spa, "error loading l2cache " 5648 "nvlist [error=%d]", error); 5649 return (spa_vdev_err(rvd, 5650 VDEV_AUX_CORRUPT_DATA, EIO)); 5651 } 5652 spa_load_note(spa, "ignoring l2cache nvlist " 5653 "[error=%d], no l2cache will be available", error); 5654 /* Leak the object, its dnode may be unreadable. */ 5655 spa->spa_l2cache.sav_object = 0; 5656 spa->spa_l2cache.sav_sync = B_TRUE; 5657 } else { 5658 spa_config_enter(spa, SCL_ALL, FTAG, RW_WRITER); 5659 spa_load_l2cache(spa); 5660 spa_config_exit(spa, SCL_ALL, FTAG); 5661 } 5662 } else if (error == 0) { 5663 spa->spa_l2cache.sav_sync = B_TRUE; 5664 } 5665 5666 return (0); 5667 } 5668 5669 static int 5670 spa_ld_load_vdev_metadata(spa_t *spa) 5671 { 5672 int error = 0; 5673 vdev_t *rvd = spa->spa_root_vdev; 5674 5675 /* 5676 * If the 'multihost' property is set, then never allow a pool to 5677 * be imported when the system hostid is zero. The exception to 5678 * this rule is zdb which is always allowed to access pools. 5679 */ 5680 if (spa_multihost(spa) && spa_get_hostid(spa) == 0 && 5681 (spa->spa_import_flags & ZFS_IMPORT_SKIP_MMP) == 0) { 5682 fnvlist_add_uint64(spa->spa_load_info, 5683 ZPOOL_CONFIG_MMP_STATE, MMP_STATE_NO_HOSTID); 5684 return (spa_vdev_err(rvd, VDEV_AUX_ACTIVE, EREMOTEIO)); 5685 } 5686 5687 /* 5688 * If the 'autoreplace' property is set, then post a resource notifying 5689 * the ZFS DE that it should not issue any faults for unopenable 5690 * devices. We also iterate over the vdevs, and post a sysevent for any 5691 * unopenable vdevs so that the normal autoreplace handler can take 5692 * over. 5693 */ 5694 if (spa->spa_autoreplace && spa->spa_load_state != SPA_LOAD_TRYIMPORT) { 5695 spa_check_removed(spa->spa_root_vdev); 5696 /* 5697 * For the import case, this is done in spa_import(), because 5698 * at this point we're using the spare definitions from 5699 * the MOS config, not necessarily from the userland config. 5700 */ 5701 if (spa->spa_load_state != SPA_LOAD_IMPORT) { 5702 spa_aux_check_removed(&spa->spa_spares); 5703 spa_aux_check_removed(&spa->spa_l2cache); 5704 } 5705 } 5706 5707 /* 5708 * Load the vdev metadata such as metaslabs, DTLs, spacemap object, etc. 5709 */ 5710 error = vdev_load(rvd); 5711 if (error != 0) { 5712 spa_load_failed(spa, "vdev_load failed [error=%d]", error); 5713 return (spa_vdev_err(rvd, VDEV_AUX_CORRUPT_DATA, error)); 5714 } 5715 5716 error = spa_ld_log_spacemaps(spa); 5717 if (error != 0) { 5718 spa_load_failed(spa, "spa_ld_log_spacemaps failed [error=%d]", 5719 error); 5720 return (spa_vdev_err(rvd, VDEV_AUX_CORRUPT_DATA, error)); 5721 } 5722 5723 /* 5724 * Propagate the leaf DTLs we just loaded all the way up the vdev tree. 5725 */ 5726 spa_config_enter(spa, SCL_ALL, FTAG, RW_WRITER); 5727 vdev_dtl_reassess(rvd, 0, 0, B_FALSE, B_FALSE); 5728 spa_config_exit(spa, SCL_ALL, FTAG); 5729 5730 return (0); 5731 } 5732 5733 static int 5734 spa_ld_load_dedup_tables(spa_t *spa) 5735 { 5736 int error = 0; 5737 vdev_t *rvd = spa->spa_root_vdev; 5738 5739 error = ddt_load(spa); 5740 if (error != 0) { 5741 spa_load_failed(spa, "ddt_load failed [error=%d]", error); 5742 return (spa_vdev_err(rvd, VDEV_AUX_CORRUPT_DATA, EIO)); 5743 } 5744 5745 return (0); 5746 } 5747 5748 static int 5749 spa_ld_load_brt(spa_t *spa) 5750 { 5751 int error = 0; 5752 vdev_t *rvd = spa->spa_root_vdev; 5753 5754 error = brt_load(spa); 5755 if (error != 0) { 5756 spa_load_failed(spa, "brt_load failed [error=%d]", error); 5757 return (spa_vdev_err(rvd, VDEV_AUX_CORRUPT_DATA, EIO)); 5758 } 5759 5760 return (0); 5761 } 5762 5763 static int 5764 spa_ld_verify_logs(spa_t *spa, spa_import_type_t type, const char **ereport) 5765 { 5766 vdev_t *rvd = spa->spa_root_vdev; 5767 5768 if (type != SPA_IMPORT_ASSEMBLE && spa_writeable(spa)) { 5769 boolean_t missing = spa_check_logs(spa); 5770 if (missing) { 5771 if (spa->spa_missing_tvds != 0) { 5772 spa_load_note(spa, "spa_check_logs failed " 5773 "so dropping the logs"); 5774 } else { 5775 *ereport = FM_EREPORT_ZFS_LOG_REPLAY; 5776 spa_load_failed(spa, "spa_check_logs failed"); 5777 return (spa_vdev_err(rvd, VDEV_AUX_BAD_LOG, 5778 ENXIO)); 5779 } 5780 } 5781 } 5782 5783 return (0); 5784 } 5785 5786 static int 5787 spa_ld_verify_pool_data(spa_t *spa) 5788 { 5789 int error = 0; 5790 vdev_t *rvd = spa->spa_root_vdev; 5791 5792 /* 5793 * We've successfully opened the pool, verify that we're ready 5794 * to start pushing transactions. 5795 */ 5796 if (spa->spa_load_state != SPA_LOAD_TRYIMPORT) { 5797 error = spa_load_verify(spa); 5798 if (error != 0) { 5799 spa_load_failed(spa, "spa_load_verify failed " 5800 "[error=%d]", error); 5801 return (spa_vdev_err(rvd, VDEV_AUX_CORRUPT_DATA, 5802 error)); 5803 } 5804 } 5805 5806 return (0); 5807 } 5808 5809 static void 5810 spa_ld_claim_log_blocks(spa_t *spa) 5811 { 5812 dmu_tx_t *tx; 5813 dsl_pool_t *dp = spa_get_dsl(spa); 5814 5815 /* 5816 * Claim log blocks that haven't been committed yet. 5817 * This must all happen in a single txg. 5818 * Note: spa_claim_max_txg is updated by spa_claim_notify(), 5819 * invoked from zil_claim_log_block()'s i/o done callback. 5820 * Price of rollback is that we abandon the log. 5821 */ 5822 spa->spa_claiming = B_TRUE; 5823 5824 tx = dmu_tx_create_assigned(dp, spa_first_txg(spa)); 5825 (void) dmu_objset_find_dp(dp, dp->dp_root_dir_obj, 5826 zil_claim, tx, DS_FIND_CHILDREN); 5827 dmu_tx_commit(tx); 5828 5829 spa->spa_claiming = B_FALSE; 5830 5831 spa_set_log_state(spa, SPA_LOG_GOOD); 5832 } 5833 5834 static void 5835 spa_ld_check_for_config_update(spa_t *spa, uint64_t config_cache_txg, 5836 boolean_t update_config_cache) 5837 { 5838 vdev_t *rvd = spa->spa_root_vdev; 5839 int need_update = B_FALSE; 5840 5841 /* 5842 * If the config cache is stale, or we have uninitialized 5843 * metaslabs (see spa_vdev_add()), then update the config. 5844 * 5845 * If this is a verbatim import, trust the current 5846 * in-core spa_config and update the disk labels. 5847 */ 5848 if (update_config_cache || config_cache_txg != spa->spa_config_txg || 5849 spa->spa_load_state == SPA_LOAD_IMPORT || 5850 spa->spa_load_state == SPA_LOAD_RECOVER || 5851 (spa->spa_import_flags & ZFS_IMPORT_VERBATIM)) 5852 need_update = B_TRUE; 5853 5854 for (int c = 0; c < rvd->vdev_children; c++) 5855 if (rvd->vdev_child[c]->vdev_ms_array == 0) 5856 need_update = B_TRUE; 5857 5858 /* 5859 * Update the config cache asynchronously in case we're the 5860 * root pool, in which case the config cache isn't writable yet. 5861 */ 5862 if (need_update) 5863 spa_async_request(spa, SPA_ASYNC_CONFIG_UPDATE); 5864 } 5865 5866 static void 5867 spa_ld_prepare_for_reload(spa_t *spa) 5868 { 5869 spa_mode_t mode = spa->spa_mode; 5870 int async_suspended = spa->spa_async_suspended; 5871 5872 spa_unload(spa); 5873 spa_deactivate(spa); 5874 spa_activate(spa, mode); 5875 5876 /* 5877 * We save the value of spa_async_suspended as it gets reset to 0 by 5878 * spa_unload(). We want to restore it back to the original value before 5879 * returning as we might be calling spa_async_resume() later. 5880 */ 5881 spa->spa_async_suspended = async_suspended; 5882 } 5883 5884 static int 5885 spa_ld_read_checkpoint_txg(spa_t *spa) 5886 { 5887 uberblock_t checkpoint; 5888 int error = 0; 5889 5890 ASSERT0(spa->spa_checkpoint_txg); 5891 ASSERT(spa_namespace_held() || 5892 spa->spa_load_thread == curthread); 5893 5894 error = zap_lookup(spa->spa_meta_objset, DMU_POOL_DIRECTORY_OBJECT, 5895 DMU_POOL_ZPOOL_CHECKPOINT, sizeof (uint64_t), 5896 sizeof (uberblock_t) / sizeof (uint64_t), &checkpoint); 5897 5898 if (error == ENOENT) 5899 return (0); 5900 5901 if (error != 0) 5902 return (error); 5903 5904 ASSERT3U(checkpoint.ub_txg, !=, 0); 5905 ASSERT3U(checkpoint.ub_checkpoint_txg, !=, 0); 5906 ASSERT3U(checkpoint.ub_timestamp, !=, 0); 5907 spa->spa_checkpoint_txg = checkpoint.ub_txg; 5908 spa->spa_checkpoint_info.sci_timestamp = checkpoint.ub_timestamp; 5909 5910 return (0); 5911 } 5912 5913 static int 5914 spa_ld_mos_init(spa_t *spa, spa_import_type_t type) 5915 { 5916 int error = 0; 5917 5918 ASSERT(spa_namespace_held()); 5919 ASSERT(spa->spa_config_source != SPA_CONFIG_SRC_NONE); 5920 5921 /* 5922 * Never trust the config that is provided unless we are assembling 5923 * a pool following a split. 5924 * This means don't trust blkptrs and the vdev tree in general. This 5925 * also effectively puts the spa in read-only mode since 5926 * spa_writeable() checks for spa_trust_config to be true. 5927 * We will later load a trusted config from the MOS. 5928 */ 5929 if (type != SPA_IMPORT_ASSEMBLE) 5930 spa->spa_trust_config = B_FALSE; 5931 5932 /* 5933 * Parse the config provided to create a vdev tree. 5934 */ 5935 error = spa_ld_parse_config(spa, type); 5936 if (error != 0) 5937 return (error); 5938 5939 spa_import_progress_add(spa); 5940 5941 /* 5942 * Now that we have the vdev tree, try to open each vdev. This involves 5943 * opening the underlying physical device, retrieving its geometry and 5944 * probing the vdev with a dummy I/O. The state of each vdev will be set 5945 * based on the success of those operations. After this we'll be ready 5946 * to read from the vdevs. 5947 */ 5948 error = spa_ld_open_vdevs(spa); 5949 if (error != 0) 5950 return (error); 5951 5952 /* 5953 * Read the label of each vdev and make sure that the GUIDs stored 5954 * there match the GUIDs in the config provided. 5955 * If we're assembling a new pool that's been split off from an 5956 * existing pool, the labels haven't yet been updated so we skip 5957 * validation for now. 5958 */ 5959 if (type != SPA_IMPORT_ASSEMBLE) { 5960 error = spa_ld_validate_vdevs(spa); 5961 if (error != 0) 5962 return (error); 5963 } 5964 5965 /* 5966 * Read all vdev labels to find the best uberblock (i.e. latest, 5967 * unless spa_load_max_txg is set) and store it in spa_uberblock. We 5968 * get the list of features required to read blkptrs in the MOS from 5969 * the vdev label with the best uberblock and verify that our version 5970 * of zfs supports them all. 5971 */ 5972 error = spa_ld_select_uberblock(spa, type); 5973 if (error != 0) 5974 return (error); 5975 5976 /* 5977 * Pass that uberblock to the dsl_pool layer which will open the root 5978 * blkptr. This blkptr points to the latest version of the MOS and will 5979 * allow us to read its contents. 5980 */ 5981 error = spa_ld_open_rootbp(spa); 5982 if (error != 0) 5983 return (error); 5984 5985 return (0); 5986 } 5987 5988 static int 5989 spa_ld_checkpoint_rewind(spa_t *spa) 5990 { 5991 uberblock_t checkpoint; 5992 int error = 0; 5993 5994 ASSERT(spa_namespace_held()); 5995 ASSERT(spa->spa_import_flags & ZFS_IMPORT_CHECKPOINT); 5996 5997 error = zap_lookup(spa->spa_meta_objset, DMU_POOL_DIRECTORY_OBJECT, 5998 DMU_POOL_ZPOOL_CHECKPOINT, sizeof (uint64_t), 5999 sizeof (uberblock_t) / sizeof (uint64_t), &checkpoint); 6000 6001 if (error != 0) { 6002 spa_load_failed(spa, "unable to retrieve checkpointed " 6003 "uberblock from the MOS config [error=%d]", error); 6004 6005 if (error == ENOENT) 6006 error = ZFS_ERR_NO_CHECKPOINT; 6007 6008 return (error); 6009 } 6010 6011 ASSERT3U(checkpoint.ub_txg, <, spa->spa_uberblock.ub_txg); 6012 ASSERT3U(checkpoint.ub_txg, ==, checkpoint.ub_checkpoint_txg); 6013 6014 /* 6015 * We need to update the txg and timestamp of the checkpointed 6016 * uberblock to be higher than the latest one. This ensures that 6017 * the checkpointed uberblock is selected if we were to close and 6018 * reopen the pool right after we've written it in the vdev labels. 6019 * (also see block comment in vdev_uberblock_compare) 6020 */ 6021 checkpoint.ub_txg = spa->spa_uberblock.ub_txg + 1; 6022 checkpoint.ub_timestamp = gethrestime_sec(); 6023 6024 /* 6025 * Set current uberblock to be the checkpointed uberblock. 6026 */ 6027 spa->spa_uberblock = checkpoint; 6028 6029 /* 6030 * If we are doing a normal rewind, then the pool is open for 6031 * writing and we sync the "updated" checkpointed uberblock to 6032 * disk. Once this is done, we've basically rewound the whole 6033 * pool and there is no way back. 6034 * 6035 * There are cases when we don't want to attempt and sync the 6036 * checkpointed uberblock to disk because we are opening a 6037 * pool as read-only. Specifically, verifying the checkpointed 6038 * state with zdb, and importing the checkpointed state to get 6039 * a "preview" of its content. 6040 */ 6041 if (spa_writeable(spa)) { 6042 vdev_t *rvd = spa->spa_root_vdev; 6043 6044 spa_config_enter(spa, SCL_ALL, FTAG, RW_WRITER); 6045 vdev_t *svd[SPA_SYNC_MIN_VDEVS] = { NULL }; 6046 int svdcount = 0; 6047 int children = rvd->vdev_children; 6048 int c0 = random_in_range(children); 6049 6050 for (int c = 0; c < children; c++) { 6051 vdev_t *vd = rvd->vdev_child[(c0 + c) % children]; 6052 6053 /* Stop when revisiting the first vdev */ 6054 if (c > 0 && svd[0] == vd) 6055 break; 6056 6057 if (vd->vdev_ms_array == 0 || vd->vdev_islog || 6058 !vdev_is_concrete(vd)) 6059 continue; 6060 6061 svd[svdcount++] = vd; 6062 if (svdcount == SPA_SYNC_MIN_VDEVS) 6063 break; 6064 } 6065 error = vdev_config_sync(spa, svd, svdcount, 6066 spa->spa_first_txg); 6067 if (error == 0) 6068 spa->spa_last_synced_guid = rvd->vdev_guid; 6069 spa_config_exit(spa, SCL_ALL, FTAG); 6070 6071 if (error != 0) { 6072 spa_load_failed(spa, "failed to write checkpointed " 6073 "uberblock to the vdev labels [error=%d]", error); 6074 return (error); 6075 } 6076 } 6077 6078 return (0); 6079 } 6080 6081 static int 6082 spa_ld_mos_with_trusted_config(spa_t *spa, spa_import_type_t type, 6083 boolean_t *update_config_cache) 6084 { 6085 int error; 6086 6087 /* 6088 * Parse the config for pool, open and validate vdevs, 6089 * select an uberblock, and use that uberblock to open 6090 * the MOS. 6091 */ 6092 error = spa_ld_mos_init(spa, type); 6093 if (error != 0) 6094 return (error); 6095 6096 /* 6097 * Retrieve the trusted config stored in the MOS and use it to create 6098 * a new, exact version of the vdev tree, then reopen all vdevs. 6099 */ 6100 error = spa_ld_trusted_config(spa, type, B_FALSE); 6101 if (error == EAGAIN) { 6102 if (update_config_cache != NULL) 6103 *update_config_cache = B_TRUE; 6104 6105 /* 6106 * Redo the loading process with the trusted config if it is 6107 * too different from the untrusted config. 6108 */ 6109 spa_ld_prepare_for_reload(spa); 6110 spa_load_note(spa, "RELOADING"); 6111 error = spa_ld_mos_init(spa, type); 6112 if (error != 0) 6113 return (error); 6114 6115 error = spa_ld_trusted_config(spa, type, B_TRUE); 6116 if (error != 0) 6117 return (error); 6118 6119 } else if (error != 0) { 6120 return (error); 6121 } 6122 6123 return (0); 6124 } 6125 6126 /* 6127 * Load an existing storage pool, using the config provided. This config 6128 * describes which vdevs are part of the pool and is later validated against 6129 * partial configs present in each vdev's label and an entire copy of the 6130 * config stored in the MOS. 6131 */ 6132 static int 6133 spa_load_impl(spa_t *spa, spa_import_type_t type, const char **ereport) 6134 { 6135 int error = 0; 6136 boolean_t missing_feat_write = B_FALSE; 6137 boolean_t checkpoint_rewind = 6138 (spa->spa_import_flags & ZFS_IMPORT_CHECKPOINT); 6139 boolean_t update_config_cache = B_FALSE; 6140 hrtime_t load_start = gethrtime(); 6141 6142 ASSERT(spa_namespace_held()); 6143 ASSERT(spa->spa_config_source != SPA_CONFIG_SRC_NONE); 6144 6145 spa_load_note(spa, "LOADING"); 6146 6147 error = spa_ld_mos_with_trusted_config(spa, type, &update_config_cache); 6148 if (error != 0) 6149 return (error); 6150 6151 /* 6152 * If we are rewinding to the checkpoint then we need to repeat 6153 * everything we've done so far in this function but this time 6154 * selecting the checkpointed uberblock and using that to open 6155 * the MOS. 6156 */ 6157 if (checkpoint_rewind) { 6158 /* 6159 * If we are rewinding to the checkpoint update config cache 6160 * anyway. 6161 */ 6162 update_config_cache = B_TRUE; 6163 6164 /* 6165 * Extract the checkpointed uberblock from the current MOS 6166 * and use this as the pool's uberblock from now on. If the 6167 * pool is imported as writeable we also write the checkpoint 6168 * uberblock to the labels, making the rewind permanent. 6169 */ 6170 error = spa_ld_checkpoint_rewind(spa); 6171 if (error != 0) 6172 return (error); 6173 6174 /* 6175 * Redo the loading process again with the 6176 * checkpointed uberblock. 6177 */ 6178 spa_ld_prepare_for_reload(spa); 6179 spa_load_note(spa, "LOADING checkpointed uberblock"); 6180 error = spa_ld_mos_with_trusted_config(spa, type, NULL); 6181 if (error != 0) 6182 return (error); 6183 } 6184 6185 /* 6186 * Drop the namespace lock for the rest of the function. 6187 */ 6188 spa->spa_load_thread = curthread; 6189 spa_namespace_exit(FTAG); 6190 6191 /* 6192 * Retrieve the checkpoint txg if the pool has a checkpoint. 6193 */ 6194 spa_import_progress_set_notes(spa, "Loading checkpoint txg"); 6195 error = spa_ld_read_checkpoint_txg(spa); 6196 if (error != 0) 6197 goto fail; 6198 6199 /* 6200 * Retrieve the mapping of indirect vdevs. Those vdevs were removed 6201 * from the pool and their contents were re-mapped to other vdevs. Note 6202 * that everything that we read before this step must have been 6203 * rewritten on concrete vdevs after the last device removal was 6204 * initiated. Otherwise we could be reading from indirect vdevs before 6205 * we have loaded their mappings. 6206 */ 6207 spa_import_progress_set_notes(spa, "Loading indirect vdev metadata"); 6208 error = spa_ld_open_indirect_vdev_metadata(spa); 6209 if (error != 0) 6210 goto fail; 6211 6212 /* 6213 * Retrieve the full list of active features from the MOS and check if 6214 * they are all supported. 6215 */ 6216 spa_import_progress_set_notes(spa, "Checking feature flags"); 6217 error = spa_ld_check_features(spa, &missing_feat_write); 6218 if (error != 0) 6219 goto fail; 6220 6221 /* 6222 * Load several special directories from the MOS needed by the dsl_pool 6223 * layer. 6224 */ 6225 spa_import_progress_set_notes(spa, "Loading special MOS directories"); 6226 error = spa_ld_load_special_directories(spa); 6227 if (error != 0) 6228 goto fail; 6229 6230 /* 6231 * Retrieve pool properties from the MOS. 6232 */ 6233 spa_import_progress_set_notes(spa, "Loading properties"); 6234 error = spa_ld_get_props(spa); 6235 if (error != 0) 6236 goto fail; 6237 6238 /* 6239 * Retrieve the list of auxiliary devices - cache devices and spares - 6240 * and open them. 6241 */ 6242 spa_import_progress_set_notes(spa, "Loading AUX vdevs"); 6243 error = spa_ld_open_aux_vdevs(spa, type); 6244 if (error != 0) 6245 goto fail; 6246 6247 /* 6248 * Load the metadata for all vdevs. Also check if unopenable devices 6249 * should be autoreplaced. 6250 */ 6251 spa_import_progress_set_notes(spa, "Loading vdev metadata"); 6252 error = spa_ld_load_vdev_metadata(spa); 6253 if (error != 0) 6254 goto fail; 6255 6256 spa_import_progress_set_notes(spa, "Loading dedup tables"); 6257 error = spa_ld_load_dedup_tables(spa); 6258 if (error != 0) 6259 goto fail; 6260 6261 spa_import_progress_set_notes(spa, "Loading BRT"); 6262 error = spa_ld_load_brt(spa); 6263 if (error != 0) 6264 goto fail; 6265 6266 /* 6267 * Verify the logs now to make sure we don't have any unexpected errors 6268 * when we claim log blocks later. 6269 */ 6270 spa_import_progress_set_notes(spa, "Verifying Log Devices"); 6271 error = spa_ld_verify_logs(spa, type, ereport); 6272 if (error != 0) 6273 goto fail; 6274 6275 if (missing_feat_write) { 6276 ASSERT(spa->spa_load_state == SPA_LOAD_TRYIMPORT); 6277 6278 /* 6279 * At this point, we know that we can open the pool in 6280 * read-only mode but not read-write mode. We now have enough 6281 * information and can return to userland. 6282 */ 6283 error = spa_vdev_err(spa->spa_root_vdev, VDEV_AUX_UNSUP_FEAT, 6284 ENOTSUP); 6285 goto fail; 6286 } 6287 6288 /* 6289 * Traverse the last txgs to make sure the pool was left off in a safe 6290 * state. When performing an extreme rewind, we verify the whole pool, 6291 * which can take a very long time. 6292 */ 6293 spa_import_progress_set_notes(spa, "Verifying pool data"); 6294 error = spa_ld_verify_pool_data(spa); 6295 if (error != 0) 6296 goto fail; 6297 6298 /* 6299 * Calculate the deflated space for the pool. This must be done before 6300 * we write anything to the pool because we'd need to update the space 6301 * accounting using the deflated sizes. 6302 */ 6303 spa_import_progress_set_notes(spa, "Calculating deflated space"); 6304 spa_update_dspace(spa); 6305 6306 /* 6307 * We have now retrieved all the information we needed to open the 6308 * pool. If we are importing the pool in read-write mode, a few 6309 * additional steps must be performed to finish the import. 6310 */ 6311 if (spa_writeable(spa) && (spa->spa_load_state == SPA_LOAD_RECOVER || 6312 spa->spa_load_max_txg == UINT64_MAX)) { 6313 uint64_t config_cache_txg = spa->spa_config_txg; 6314 6315 spa_import_progress_set_notes(spa, "Starting import"); 6316 6317 ASSERT(spa->spa_load_state != SPA_LOAD_TRYIMPORT); 6318 6319 /* 6320 * Before we do any zio_write's, complete the raidz expansion 6321 * scratch space copying, if necessary. 6322 */ 6323 if (RRSS_GET_STATE(&spa->spa_uberblock) == RRSS_SCRATCH_VALID) 6324 vdev_raidz_reflow_copy_scratch(spa); 6325 6326 /* 6327 * In case of a checkpoint rewind, log the original txg 6328 * of the checkpointed uberblock. 6329 */ 6330 if (checkpoint_rewind) { 6331 spa_history_log_internal(spa, "checkpoint rewind", 6332 NULL, "rewound state to txg=%llu", 6333 (u_longlong_t)spa->spa_uberblock.ub_checkpoint_txg); 6334 } 6335 6336 spa_import_progress_set_notes(spa, "Claiming ZIL blocks"); 6337 /* 6338 * Traverse the ZIL and claim all blocks. 6339 */ 6340 spa_ld_claim_log_blocks(spa); 6341 6342 /* 6343 * Kick-off the syncing thread. 6344 */ 6345 spa->spa_sync_on = B_TRUE; 6346 txg_sync_start(spa->spa_dsl_pool); 6347 mmp_thread_start(spa); 6348 6349 /* 6350 * Wait for all claims to sync. We sync up to the highest 6351 * claimed log block birth time so that claimed log blocks 6352 * don't appear to be from the future. spa_claim_max_txg 6353 * will have been set for us by ZIL traversal operations 6354 * performed above. 6355 */ 6356 spa_import_progress_set_notes(spa, "Syncing ZIL claims"); 6357 txg_wait_synced(spa->spa_dsl_pool, spa->spa_claim_max_txg); 6358 6359 /* 6360 * Check if we need to request an update of the config. On the 6361 * next sync, we would update the config stored in vdev labels 6362 * and the cachefile (by default /etc/zfs/zpool.cache). 6363 */ 6364 spa_import_progress_set_notes(spa, "Updating configs"); 6365 spa_ld_check_for_config_update(spa, config_cache_txg, 6366 update_config_cache); 6367 6368 /* 6369 * Check if a rebuild was in progress and if so resume it. 6370 * Then check all DTLs to see if anything needs resilvering. 6371 * The resilver will be deferred if a rebuild was started. 6372 */ 6373 spa_import_progress_set_notes(spa, "Starting resilvers"); 6374 if (vdev_rebuild_active(spa->spa_root_vdev)) { 6375 vdev_rebuild_restart(spa); 6376 } else if (!dsl_scan_resilvering(spa->spa_dsl_pool) && 6377 vdev_resilver_needed(spa->spa_root_vdev, NULL, NULL)) { 6378 spa_async_request(spa, SPA_ASYNC_RESILVER); 6379 } 6380 6381 /* 6382 * A scrub error log with no scan to own it is stale. 6383 * Promote it, so an error scrub can reach its entries. 6384 */ 6385 if (spa->spa_errlog_scrub != 0 && spa->spa_errlog_last == 0 && 6386 !dsl_scan_scrubbing(spa->spa_dsl_pool) && 6387 !dsl_scan_resilvering(spa->spa_dsl_pool) && 6388 !dsl_errorscrubbing(spa->spa_dsl_pool)) 6389 spa_errlog_rotate(spa); 6390 6391 /* 6392 * Log the fact that we booted up (so that we can detect if 6393 * we rebooted in the middle of an operation). 6394 */ 6395 spa_history_log_version(spa, "open", NULL); 6396 6397 spa_import_progress_set_notes(spa, 6398 "Restarting device removals"); 6399 spa_restart_removal(spa); 6400 spa_spawn_aux_threads(spa); 6401 6402 /* 6403 * Delete any inconsistent datasets. 6404 * 6405 * Note: 6406 * Since we may be issuing deletes for clones here, 6407 * we make sure to do so after we've spawned all the 6408 * auxiliary threads above (from which the livelist 6409 * deletion zthr is part of). 6410 */ 6411 spa_import_progress_set_notes(spa, 6412 "Cleaning up inconsistent objsets"); 6413 (void) dmu_objset_find(spa_name(spa), 6414 dsl_destroy_inconsistent, NULL, DS_FIND_CHILDREN); 6415 6416 /* 6417 * Clean up any stale temporary dataset userrefs. 6418 */ 6419 spa_import_progress_set_notes(spa, 6420 "Cleaning up temporary userrefs"); 6421 dsl_pool_clean_tmp_userrefs(spa->spa_dsl_pool); 6422 6423 /* 6424 * Anything still marked for deferred destruction because a 6425 * mount was holding it was left that way by a crash or an 6426 * export, and nothing is holding it now. The sweep walks 6427 * every snapshot, so leave it to the async thread rather than 6428 * spending import time on it. 6429 */ 6430 spa_async_request(spa, SPA_ASYNC_DEFER_DESTROY); 6431 6432 spa_config_enter(spa, SCL_CONFIG, FTAG, RW_READER); 6433 spa_import_progress_set_notes(spa, "Restarting initialize"); 6434 vdev_initialize_restart(spa->spa_root_vdev); 6435 spa_import_progress_set_notes(spa, "Restarting TRIM"); 6436 vdev_trim_restart(spa->spa_root_vdev); 6437 vdev_autotrim_restart(spa); 6438 spa_config_exit(spa, SCL_CONFIG, FTAG); 6439 spa_import_progress_set_notes(spa, "Finished importing"); 6440 } 6441 zio_handle_import_delay(spa, gethrtime() - load_start); 6442 6443 spa_import_progress_remove(spa_guid(spa)); 6444 spa_async_request(spa, SPA_ASYNC_L2CACHE_REBUILD); 6445 6446 spa_load_note(spa, "LOADED"); 6447 fail: 6448 spa_namespace_enter(FTAG); 6449 spa->spa_load_thread = NULL; 6450 spa_namespace_broadcast(); 6451 6452 return (error); 6453 6454 } 6455 6456 static int 6457 spa_load_retry(spa_t *spa, spa_load_state_t state) 6458 { 6459 spa_mode_t mode = spa->spa_mode; 6460 6461 spa_unload(spa); 6462 spa_deactivate(spa); 6463 6464 spa->spa_load_max_txg = spa->spa_uberblock.ub_txg - 1; 6465 6466 spa_activate(spa, mode); 6467 spa_async_suspend(spa); 6468 6469 spa_load_note(spa, "spa_load_retry: rewind, max txg: %llu", 6470 (u_longlong_t)spa->spa_load_max_txg); 6471 6472 return (spa_load(spa, state, SPA_IMPORT_EXISTING)); 6473 } 6474 6475 /* 6476 * If spa_load() fails this function will try loading prior txg's. If 6477 * 'state' is SPA_LOAD_RECOVER and one of these loads succeeds the pool 6478 * will be rewound to that txg. If 'state' is not SPA_LOAD_RECOVER this 6479 * function will not rewind the pool and will return the same error as 6480 * spa_load(), or ECANCELED if the load only probed the requested txg 6481 * instead of bringing the pool up. 6482 */ 6483 static int 6484 spa_load_best(spa_t *spa, spa_load_state_t state, uint64_t max_request, 6485 int rewind_flags) 6486 { 6487 nvlist_t *loadinfo = NULL; 6488 nvlist_t *config = NULL; 6489 int load_error, rewind_error; 6490 uint64_t safe_rewind_txg; 6491 uint64_t min_txg; 6492 6493 if (spa->spa_load_txg && state == SPA_LOAD_RECOVER) { 6494 spa->spa_load_max_txg = spa->spa_load_txg; 6495 spa_set_log_state(spa, SPA_LOG_CLEAR); 6496 } else { 6497 spa->spa_load_max_txg = max_request; 6498 if (max_request != UINT64_MAX) 6499 spa->spa_extreme_rewind = B_TRUE; 6500 } 6501 6502 load_error = rewind_error = spa_load(spa, state, SPA_IMPORT_EXISTING); 6503 if (load_error == 0) { 6504 /* 6505 * A load of an explicitly requested txg may only probe 6506 * whether that txg is usable, finishing without a syncing 6507 * thread. Such a pool is not functional, so it can not be 6508 * handed to the caller no matter how well it loaded. Report 6509 * the probed txg the same way an actual rewind would. 6510 */ 6511 if (spa_writeable(spa) && !spa->spa_sync_on) { 6512 loadinfo = fnvlist_alloc(); 6513 fnvlist_add_nvlist(loadinfo, ZPOOL_CONFIG_REWIND_INFO, 6514 spa->spa_load_info); 6515 fnvlist_free(spa->spa_load_info); 6516 spa->spa_load_info = loadinfo; 6517 spa_import_progress_remove(spa_guid(spa)); 6518 return (SET_ERROR(ECANCELED)); 6519 } 6520 return (0); 6521 } 6522 6523 /* Do not attempt to load uberblocks from previous txgs when: */ 6524 switch (load_error) { 6525 case ZFS_ERR_NO_CHECKPOINT: 6526 /* Attempting checkpoint-rewind on a pool with no checkpoint */ 6527 ASSERT(spa->spa_import_flags & ZFS_IMPORT_CHECKPOINT); 6528 zfs_fallthrough; 6529 case EREMOTEIO: 6530 /* MMP determines the pool is active on another host */ 6531 zfs_fallthrough; 6532 case EBADF: 6533 /* The config cache is out of sync (vdevs or hostid) */ 6534 zfs_fallthrough; 6535 case EINTR: 6536 /* The user interactively interrupted the import */ 6537 spa_import_progress_remove(spa_guid(spa)); 6538 return (load_error); 6539 } 6540 6541 if (spa->spa_root_vdev != NULL) 6542 config = spa_config_generate(spa, NULL, -1ULL, B_TRUE); 6543 6544 spa->spa_last_ubsync_txg = spa->spa_uberblock.ub_txg; 6545 spa->spa_last_ubsync_txg_ts = spa->spa_uberblock.ub_timestamp; 6546 6547 if (rewind_flags & ZPOOL_NEVER_REWIND) { 6548 nvlist_free(config); 6549 spa_import_progress_remove(spa_guid(spa)); 6550 return (load_error); 6551 } 6552 6553 if (state == SPA_LOAD_RECOVER) { 6554 /* Price of rolling back is discarding txgs, including log */ 6555 spa_set_log_state(spa, SPA_LOG_CLEAR); 6556 } else { 6557 /* 6558 * If we aren't rolling back save the load info from our first 6559 * import attempt so that we can restore it after attempting 6560 * to rewind. 6561 */ 6562 loadinfo = spa->spa_load_info; 6563 spa->spa_load_info = fnvlist_alloc(); 6564 } 6565 6566 spa->spa_load_max_txg = spa->spa_last_ubsync_txg; 6567 safe_rewind_txg = spa->spa_last_ubsync_txg - TXG_DEFER_SIZE; 6568 min_txg = (rewind_flags & ZPOOL_EXTREME_REWIND) ? 6569 TXG_INITIAL : safe_rewind_txg; 6570 6571 /* 6572 * Continue as long as we're finding errors, we're still within 6573 * the acceptable rewind range, and we're still finding uberblocks 6574 */ 6575 while (rewind_error && spa->spa_uberblock.ub_txg >= min_txg && 6576 spa->spa_uberblock.ub_txg <= spa->spa_load_max_txg) { 6577 if (spa->spa_load_max_txg < safe_rewind_txg) 6578 spa->spa_extreme_rewind = B_TRUE; 6579 rewind_error = spa_load_retry(spa, state); 6580 } 6581 6582 spa->spa_extreme_rewind = B_FALSE; 6583 spa->spa_load_max_txg = UINT64_MAX; 6584 6585 if (config && (rewind_error || state != SPA_LOAD_RECOVER)) 6586 spa_config_set(spa, config); 6587 else 6588 nvlist_free(config); 6589 6590 if (state == SPA_LOAD_RECOVER) { 6591 ASSERT0P(loadinfo); 6592 spa_import_progress_remove(spa_guid(spa)); 6593 return (rewind_error); 6594 } else { 6595 /* Store the rewind info as part of the initial load info */ 6596 fnvlist_add_nvlist(loadinfo, ZPOOL_CONFIG_REWIND_INFO, 6597 spa->spa_load_info); 6598 6599 /* Restore the initial load info */ 6600 fnvlist_free(spa->spa_load_info); 6601 spa->spa_load_info = loadinfo; 6602 6603 spa_import_progress_remove(spa_guid(spa)); 6604 return (load_error); 6605 } 6606 } 6607 6608 /* 6609 * Pool Open/Import 6610 * 6611 * The import case is identical to an open except that the configuration is sent 6612 * down from userland, instead of grabbed from the configuration cache. For the 6613 * case of an open, the pool configuration will exist in the 6614 * POOL_STATE_UNINITIALIZED state. 6615 * 6616 * The stats information (gen/count/ustats) is used to gather vdev statistics at 6617 * the same time open the pool, without having to keep around the spa_t in some 6618 * ambiguous state. 6619 */ 6620 static int 6621 spa_open_common(const char *pool, spa_t **spapp, const void *tag, 6622 nvlist_t *nvpolicy, nvlist_t **config) 6623 { 6624 spa_t *spa; 6625 spa_load_state_t state = SPA_LOAD_OPEN; 6626 int error; 6627 int locked = B_FALSE; 6628 int firstopen = B_FALSE; 6629 6630 *spapp = NULL; 6631 6632 /* 6633 * As disgusting as this is, we need to support recursive calls to this 6634 * function because dsl_dir_open() is called during spa_load(), and ends 6635 * up calling spa_open() again. The real fix is to figure out how to 6636 * avoid dsl_dir_open() calling this in the first place. 6637 */ 6638 if (!spa_namespace_held()) { 6639 spa_namespace_enter(FTAG); 6640 locked = B_TRUE; 6641 } 6642 6643 if ((spa = spa_lookup(pool)) == NULL) { 6644 if (locked) 6645 spa_namespace_exit(FTAG); 6646 return (SET_ERROR(ENOENT)); 6647 } 6648 6649 if (spa->spa_state == POOL_STATE_UNINITIALIZED) { 6650 zpool_load_policy_t policy; 6651 6652 firstopen = B_TRUE; 6653 6654 zpool_get_load_policy(nvpolicy ? nvpolicy : spa->spa_config, 6655 &policy); 6656 if (policy.zlp_rewind & ZPOOL_DO_REWIND) 6657 state = SPA_LOAD_RECOVER; 6658 6659 spa_activate(spa, spa_mode_global); 6660 6661 if (state != SPA_LOAD_RECOVER) 6662 spa->spa_last_ubsync_txg = spa->spa_load_txg = 0; 6663 spa->spa_config_source = SPA_CONFIG_SRC_CACHEFILE; 6664 6665 zfs_dbgmsg("spa_open_common: opening %s", pool); 6666 error = spa_load_best(spa, state, policy.zlp_txg, 6667 policy.zlp_rewind); 6668 6669 if (error == EBADF) { 6670 /* 6671 * If vdev_validate() returns failure (indicated by 6672 * EBADF), it indicates that one of the vdevs indicates 6673 * that the pool has been exported or destroyed. If 6674 * this is the case, the config cache is out of sync and 6675 * we should remove the pool from the namespace. 6676 */ 6677 spa_unload(spa); 6678 spa_deactivate(spa); 6679 spa_write_cachefile(spa, B_TRUE, B_TRUE, B_FALSE); 6680 spa_remove(spa); 6681 if (locked) 6682 spa_namespace_exit(FTAG); 6683 return (SET_ERROR(ENOENT)); 6684 } 6685 6686 if (error) { 6687 /* 6688 * We can't open the pool, but we still have useful 6689 * information: the state of each vdev after the 6690 * attempted vdev_open(). Return this to the user. 6691 */ 6692 if (config != NULL && spa->spa_config) { 6693 *config = fnvlist_dup(spa->spa_config); 6694 fnvlist_add_nvlist(*config, 6695 ZPOOL_CONFIG_LOAD_INFO, 6696 spa->spa_load_info); 6697 } 6698 spa_unload(spa); 6699 spa_deactivate(spa); 6700 spa->spa_last_open_failed = error; 6701 if (locked) 6702 spa_namespace_exit(FTAG); 6703 *spapp = NULL; 6704 return (error); 6705 } 6706 } 6707 6708 spa_open_ref(spa, tag); 6709 6710 if (config != NULL) 6711 *config = spa_config_generate(spa, NULL, -1ULL, B_TRUE); 6712 6713 /* 6714 * If we've recovered the pool, pass back any information we 6715 * gathered while doing the load. 6716 */ 6717 if (state == SPA_LOAD_RECOVER && config != NULL) { 6718 fnvlist_add_nvlist(*config, ZPOOL_CONFIG_LOAD_INFO, 6719 spa->spa_load_info); 6720 } 6721 6722 if (locked) { 6723 spa->spa_last_open_failed = 0; 6724 spa->spa_last_ubsync_txg = 0; 6725 spa->spa_load_txg = 0; 6726 spa_namespace_exit(FTAG); 6727 } 6728 6729 if (firstopen) 6730 zvol_create_minors(spa_name(spa)); 6731 6732 *spapp = spa; 6733 6734 return (0); 6735 } 6736 6737 int 6738 spa_open_rewind(const char *name, spa_t **spapp, const void *tag, 6739 nvlist_t *policy, nvlist_t **config) 6740 { 6741 return (spa_open_common(name, spapp, tag, policy, config)); 6742 } 6743 6744 int 6745 spa_open(const char *name, spa_t **spapp, const void *tag) 6746 { 6747 return (spa_open_common(name, spapp, tag, NULL, NULL)); 6748 } 6749 6750 /* 6751 * Lookup the given spa_t, incrementing the inject count in the process, 6752 * preventing it from being exported or destroyed. 6753 */ 6754 spa_t * 6755 spa_inject_addref(char *name) 6756 { 6757 spa_t *spa; 6758 6759 spa_namespace_enter(FTAG); 6760 if ((spa = spa_lookup(name)) == NULL) { 6761 spa_namespace_exit(FTAG); 6762 return (NULL); 6763 } 6764 spa->spa_inject_ref++; 6765 spa_namespace_exit(FTAG); 6766 6767 return (spa); 6768 } 6769 6770 void 6771 spa_inject_delref(spa_t *spa) 6772 { 6773 spa_namespace_enter(FTAG); 6774 spa->spa_inject_ref--; 6775 spa_namespace_exit(FTAG); 6776 } 6777 6778 /* 6779 * Add spares device information to the nvlist. 6780 */ 6781 static void 6782 spa_add_spares(spa_t *spa, nvlist_t *config) 6783 { 6784 nvlist_t **spares; 6785 uint_t i, nspares; 6786 nvlist_t *nvroot; 6787 uint64_t guid; 6788 vdev_stat_t *vs; 6789 uint_t vsc; 6790 uint64_t pool; 6791 6792 ASSERT(spa_config_held(spa, SCL_CONFIG, RW_READER)); 6793 6794 if (spa->spa_spares.sav_count == 0) 6795 return; 6796 6797 nvroot = fnvlist_lookup_nvlist(config, ZPOOL_CONFIG_VDEV_TREE); 6798 VERIFY0(nvlist_lookup_nvlist_array(spa->spa_spares.sav_config, 6799 ZPOOL_CONFIG_SPARES, &spares, &nspares)); 6800 if (nspares != 0) { 6801 fnvlist_add_nvlist_array(nvroot, ZPOOL_CONFIG_SPARES, 6802 (const nvlist_t * const *)spares, nspares); 6803 VERIFY0(nvlist_lookup_nvlist_array(nvroot, ZPOOL_CONFIG_SPARES, 6804 &spares, &nspares)); 6805 6806 /* 6807 * Go through and find any spares which have since been 6808 * repurposed as an active spare. If this is the case, update 6809 * their status appropriately. 6810 */ 6811 for (i = 0; i < nspares; i++) { 6812 guid = fnvlist_lookup_uint64(spares[i], 6813 ZPOOL_CONFIG_GUID); 6814 VERIFY0(nvlist_lookup_uint64_array(spares[i], 6815 ZPOOL_CONFIG_VDEV_STATS, (uint64_t **)&vs, &vsc)); 6816 if (spa_spare_exists(guid, &pool, NULL) && 6817 pool != 0ULL) { 6818 vs->vs_state = VDEV_STATE_CANT_OPEN; 6819 vs->vs_aux = VDEV_AUX_SPARED; 6820 } else { 6821 vs->vs_state = 6822 spa->spa_spares.sav_vdevs[i]->vdev_state; 6823 } 6824 } 6825 } 6826 } 6827 6828 /* 6829 * Add l2cache device information to the nvlist, including vdev stats. 6830 */ 6831 static void 6832 spa_add_l2cache(spa_t *spa, nvlist_t *config) 6833 { 6834 nvlist_t **l2cache; 6835 uint_t i, j, nl2cache; 6836 nvlist_t *nvroot; 6837 uint64_t guid; 6838 vdev_t *vd; 6839 vdev_stat_t *vs; 6840 uint_t vsc; 6841 6842 ASSERT(spa_config_held(spa, SCL_CONFIG, RW_READER)); 6843 6844 if (spa->spa_l2cache.sav_count == 0) 6845 return; 6846 6847 nvroot = fnvlist_lookup_nvlist(config, ZPOOL_CONFIG_VDEV_TREE); 6848 VERIFY0(nvlist_lookup_nvlist_array(spa->spa_l2cache.sav_config, 6849 ZPOOL_CONFIG_L2CACHE, &l2cache, &nl2cache)); 6850 if (nl2cache != 0) { 6851 fnvlist_add_nvlist_array(nvroot, ZPOOL_CONFIG_L2CACHE, 6852 (const nvlist_t * const *)l2cache, nl2cache); 6853 VERIFY0(nvlist_lookup_nvlist_array(nvroot, ZPOOL_CONFIG_L2CACHE, 6854 &l2cache, &nl2cache)); 6855 6856 /* 6857 * Update level 2 cache device stats. 6858 */ 6859 6860 for (i = 0; i < nl2cache; i++) { 6861 guid = fnvlist_lookup_uint64(l2cache[i], 6862 ZPOOL_CONFIG_GUID); 6863 6864 vd = NULL; 6865 for (j = 0; j < spa->spa_l2cache.sav_count; j++) { 6866 if (guid == 6867 spa->spa_l2cache.sav_vdevs[j]->vdev_guid) { 6868 vd = spa->spa_l2cache.sav_vdevs[j]; 6869 break; 6870 } 6871 } 6872 ASSERT(vd != NULL); 6873 6874 VERIFY0(nvlist_lookup_uint64_array(l2cache[i], 6875 ZPOOL_CONFIG_VDEV_STATS, (uint64_t **)&vs, &vsc)); 6876 vdev_get_stats(vd, vs); 6877 vdev_config_generate_stats(vd, l2cache[i]); 6878 6879 } 6880 } 6881 } 6882 6883 static void 6884 spa_feature_stats_from_disk(spa_t *spa, nvlist_t *features) 6885 { 6886 zap_cursor_t zc; 6887 zap_attribute_t *za = zap_attribute_alloc(); 6888 6889 if (spa->spa_feat_for_read_obj != 0) { 6890 for (zap_cursor_init(&zc, spa->spa_meta_objset, 6891 spa->spa_feat_for_read_obj); 6892 zap_cursor_retrieve(&zc, za) == 0; 6893 zap_cursor_advance(&zc)) { 6894 ASSERT(za->za_integer_length == sizeof (uint64_t) && 6895 za->za_num_integers == 1); 6896 VERIFY0(nvlist_add_uint64(features, za->za_name, 6897 za->za_first_integer)); 6898 } 6899 zap_cursor_fini(&zc); 6900 } 6901 6902 if (spa->spa_feat_for_write_obj != 0) { 6903 for (zap_cursor_init(&zc, spa->spa_meta_objset, 6904 spa->spa_feat_for_write_obj); 6905 zap_cursor_retrieve(&zc, za) == 0; 6906 zap_cursor_advance(&zc)) { 6907 ASSERT(za->za_integer_length == sizeof (uint64_t) && 6908 za->za_num_integers == 1); 6909 VERIFY0(nvlist_add_uint64(features, za->za_name, 6910 za->za_first_integer)); 6911 } 6912 zap_cursor_fini(&zc); 6913 } 6914 zap_attribute_free(za); 6915 } 6916 6917 static void 6918 spa_feature_stats_from_cache(spa_t *spa, nvlist_t *features) 6919 { 6920 int i; 6921 6922 for (i = 0; i < SPA_FEATURES; i++) { 6923 zfeature_info_t feature = spa_feature_table[i]; 6924 uint64_t refcount; 6925 6926 if (feature_get_refcount(spa, &feature, &refcount) != 0) 6927 continue; 6928 6929 VERIFY0(nvlist_add_uint64(features, feature.fi_guid, refcount)); 6930 } 6931 } 6932 6933 /* 6934 * Store a list of pool features and their reference counts in the 6935 * config. 6936 * 6937 * The first time this is called on a spa, allocate a new nvlist, fetch 6938 * the pool features and reference counts from disk, then save the list 6939 * in the spa. In subsequent calls on the same spa use the saved nvlist 6940 * and refresh its values from the cached reference counts. This 6941 * ensures we don't block here on I/O on a suspended pool so 'zpool 6942 * clear' can resume the pool. 6943 */ 6944 static void 6945 spa_add_feature_stats(spa_t *spa, nvlist_t *config) 6946 { 6947 nvlist_t *features; 6948 6949 ASSERT(spa_config_held(spa, SCL_CONFIG, RW_READER)); 6950 6951 mutex_enter(&spa->spa_feat_stats_lock); 6952 features = spa->spa_feat_stats; 6953 6954 if (features != NULL) { 6955 spa_feature_stats_from_cache(spa, features); 6956 } else { 6957 VERIFY0(nvlist_alloc(&features, NV_UNIQUE_NAME, KM_SLEEP)); 6958 spa->spa_feat_stats = features; 6959 spa_feature_stats_from_disk(spa, features); 6960 } 6961 6962 VERIFY0(nvlist_add_nvlist(config, ZPOOL_CONFIG_FEATURE_STATS, 6963 features)); 6964 6965 mutex_exit(&spa->spa_feat_stats_lock); 6966 } 6967 6968 int 6969 spa_get_stats(const char *name, nvlist_t **config, 6970 char *altroot, size_t buflen) 6971 { 6972 int error; 6973 spa_t *spa; 6974 6975 *config = NULL; 6976 error = spa_open_common(name, &spa, FTAG, NULL, config); 6977 6978 if (spa != NULL) { 6979 /* 6980 * This still leaves a window of inconsistency where the spares 6981 * or l2cache devices could change and the config would be 6982 * self-inconsistent. 6983 */ 6984 spa_config_enter(spa, SCL_CONFIG, FTAG, RW_READER); 6985 6986 if (*config != NULL) { 6987 uint64_t loadtimes[2]; 6988 6989 loadtimes[0] = spa->spa_loaded_ts.tv_sec; 6990 loadtimes[1] = spa->spa_loaded_ts.tv_nsec; 6991 fnvlist_add_uint64_array(*config, 6992 ZPOOL_CONFIG_LOADED_TIME, loadtimes, 2); 6993 6994 fnvlist_add_uint64(*config, 6995 ZPOOL_CONFIG_ERRCOUNT, 6996 spa_approx_errlog_size(spa)); 6997 6998 if (spa_suspended(spa)) { 6999 fnvlist_add_uint64(*config, 7000 ZPOOL_CONFIG_SUSPENDED, 7001 spa->spa_failmode); 7002 fnvlist_add_uint64(*config, 7003 ZPOOL_CONFIG_SUSPENDED_REASON, 7004 spa->spa_suspended); 7005 } 7006 7007 spa_add_spares(spa, *config); 7008 spa_add_l2cache(spa, *config); 7009 spa_add_feature_stats(spa, *config); 7010 } 7011 } 7012 7013 /* 7014 * We want to get the alternate root even for faulted pools, so we cheat 7015 * and call spa_lookup() directly. 7016 */ 7017 if (altroot) { 7018 if (spa == NULL) { 7019 spa_namespace_enter(FTAG); 7020 spa = spa_lookup(name); 7021 if (spa) 7022 spa_altroot(spa, altroot, buflen); 7023 else 7024 altroot[0] = '\0'; 7025 spa = NULL; 7026 spa_namespace_exit(FTAG); 7027 } else { 7028 spa_altroot(spa, altroot, buflen); 7029 } 7030 } 7031 7032 if (spa != NULL) { 7033 spa_config_exit(spa, SCL_CONFIG, FTAG); 7034 spa_close(spa, FTAG); 7035 } 7036 7037 return (error); 7038 } 7039 7040 /* 7041 * Validate that the auxiliary device array is well formed. We must have an 7042 * array of nvlists, each which describes a valid leaf vdev. If this is an 7043 * import (mode is VDEV_ALLOC_SPARE), then we allow corrupted spares to be 7044 * specified, as long as they are well-formed. 7045 */ 7046 static int 7047 spa_validate_aux_devs(spa_t *spa, nvlist_t *nvroot, uint64_t crtxg, int mode, 7048 spa_aux_vdev_t *sav, const char *config, uint64_t version, 7049 vdev_labeltype_t label) 7050 { 7051 nvlist_t **dev; 7052 uint_t i, ndev; 7053 vdev_t *vd; 7054 int error; 7055 7056 ASSERT(spa_config_held(spa, SCL_ALL, RW_WRITER) == SCL_ALL); 7057 7058 /* 7059 * It's acceptable to have no devs specified. 7060 */ 7061 if (nvlist_lookup_nvlist_array(nvroot, config, &dev, &ndev) != 0) 7062 return (0); 7063 7064 if (ndev == 0) 7065 return (SET_ERROR(EINVAL)); 7066 7067 /* 7068 * Make sure the pool is formatted with a version that supports this 7069 * device type. 7070 */ 7071 if (spa_version(spa) < version) 7072 return (SET_ERROR(ENOTSUP)); 7073 7074 /* 7075 * Set the pending device list so we correctly handle device in-use 7076 * checking. 7077 */ 7078 sav->sav_pending = dev; 7079 sav->sav_npending = ndev; 7080 7081 for (i = 0; i < ndev; i++) { 7082 if ((error = spa_config_parse(spa, &vd, dev[i], NULL, 0, 7083 mode)) != 0) 7084 goto out; 7085 7086 if (!vd->vdev_ops->vdev_op_leaf) { 7087 vdev_free(vd); 7088 error = SET_ERROR(EINVAL); 7089 goto out; 7090 } 7091 7092 vd->vdev_top = vd; 7093 7094 if ((error = vdev_open(vd, CRED())) == 0 && 7095 (error = vdev_label_init(vd, crtxg, label)) == 0) { 7096 fnvlist_add_uint64(dev[i], ZPOOL_CONFIG_GUID, 7097 vd->vdev_guid); 7098 } 7099 7100 vdev_free(vd); 7101 7102 if (error && 7103 (mode != VDEV_ALLOC_SPARE && mode != VDEV_ALLOC_L2CACHE)) 7104 goto out; 7105 else 7106 error = 0; 7107 } 7108 7109 out: 7110 sav->sav_pending = NULL; 7111 sav->sav_npending = 0; 7112 return (error); 7113 } 7114 7115 static int 7116 spa_validate_aux(spa_t *spa, nvlist_t *nvroot, uint64_t crtxg, int mode) 7117 { 7118 int error; 7119 7120 ASSERT(spa_config_held(spa, SCL_ALL, RW_WRITER) == SCL_ALL); 7121 7122 if ((error = spa_validate_aux_devs(spa, nvroot, crtxg, mode, 7123 &spa->spa_spares, ZPOOL_CONFIG_SPARES, SPA_VERSION_SPARES, 7124 VDEV_LABEL_SPARE)) != 0) { 7125 return (error); 7126 } 7127 7128 return (spa_validate_aux_devs(spa, nvroot, crtxg, mode, 7129 &spa->spa_l2cache, ZPOOL_CONFIG_L2CACHE, SPA_VERSION_L2CACHE, 7130 VDEV_LABEL_L2CACHE)); 7131 } 7132 7133 static void 7134 spa_set_aux_vdevs(spa_aux_vdev_t *sav, nvlist_t **devs, int ndevs, 7135 const char *config) 7136 { 7137 int i; 7138 7139 if (sav->sav_config != NULL) { 7140 nvlist_t **olddevs; 7141 uint_t oldndevs; 7142 nvlist_t **newdevs; 7143 7144 /* 7145 * Generate new dev list by concatenating with the 7146 * current dev list. 7147 */ 7148 VERIFY0(nvlist_lookup_nvlist_array(sav->sav_config, config, 7149 &olddevs, &oldndevs)); 7150 7151 newdevs = kmem_alloc(sizeof (void *) * 7152 (ndevs + oldndevs), KM_SLEEP); 7153 for (i = 0; i < oldndevs; i++) 7154 newdevs[i] = fnvlist_dup(olddevs[i]); 7155 for (i = 0; i < ndevs; i++) 7156 newdevs[i + oldndevs] = fnvlist_dup(devs[i]); 7157 7158 fnvlist_remove(sav->sav_config, config); 7159 7160 fnvlist_add_nvlist_array(sav->sav_config, config, 7161 (const nvlist_t * const *)newdevs, ndevs + oldndevs); 7162 for (i = 0; i < oldndevs + ndevs; i++) 7163 nvlist_free(newdevs[i]); 7164 kmem_free(newdevs, (oldndevs + ndevs) * sizeof (void *)); 7165 } else { 7166 /* 7167 * Generate a new dev list. 7168 */ 7169 sav->sav_config = fnvlist_alloc(); 7170 fnvlist_add_nvlist_array(sav->sav_config, config, 7171 (const nvlist_t * const *)devs, ndevs); 7172 } 7173 } 7174 7175 /* 7176 * Stop and drop level 2 ARC devices 7177 */ 7178 void 7179 spa_l2cache_drop(spa_t *spa) 7180 { 7181 vdev_t *vd; 7182 int i; 7183 spa_aux_vdev_t *sav = &spa->spa_l2cache; 7184 7185 for (i = 0; i < sav->sav_count; i++) { 7186 uint64_t pool; 7187 7188 vd = sav->sav_vdevs[i]; 7189 ASSERT(vd != NULL); 7190 7191 if (spa_l2cache_exists(vd->vdev_guid, &pool) && 7192 pool != 0ULL && l2arc_vdev_present(vd)) 7193 l2arc_remove_vdev(vd); 7194 } 7195 } 7196 7197 /* 7198 * Verify encryption parameters for spa creation. If we are encrypting, we must 7199 * have the encryption feature flag enabled. 7200 */ 7201 static int 7202 spa_create_check_encryption_params(dsl_crypto_params_t *dcp, 7203 boolean_t has_encryption) 7204 { 7205 if (dcp->cp_crypt != ZIO_CRYPT_OFF && 7206 dcp->cp_crypt != ZIO_CRYPT_INHERIT && 7207 !has_encryption) 7208 return (SET_ERROR(ENOTSUP)); 7209 7210 return (dmu_objset_create_crypt_check(NULL, dcp, NULL)); 7211 } 7212 7213 /* 7214 * Pool Creation 7215 */ 7216 int 7217 spa_create(const char *pool, nvlist_t *nvroot, nvlist_t *props, 7218 nvlist_t *zplprops, dsl_crypto_params_t *dcp, nvlist_t **errinfo) 7219 { 7220 spa_t *spa; 7221 const char *altroot = NULL; 7222 vdev_t *rvd; 7223 dsl_pool_t *dp; 7224 dmu_tx_t *tx; 7225 int error = 0; 7226 uint64_t txg = TXG_INITIAL; 7227 nvlist_t **spares, **l2cache; 7228 uint_t nspares, nl2cache; 7229 uint64_t version, obj, ndraid = 0, draid_nfgroup = 0; 7230 boolean_t has_features; 7231 boolean_t has_encryption; 7232 boolean_t has_allocclass; 7233 boolean_t has_draid; 7234 boolean_t has_draid_fdomains; 7235 spa_feature_t feat; 7236 const char *feat_name; 7237 const char *poolname; 7238 nvlist_t *nvl; 7239 7240 if (props == NULL || 7241 nvlist_lookup_string(props, 7242 zpool_prop_to_name(ZPOOL_PROP_TNAME), &poolname) != 0) 7243 poolname = (char *)pool; 7244 7245 /* 7246 * If this pool already exists, return failure. 7247 */ 7248 spa_namespace_enter(FTAG); 7249 if (spa_lookup(poolname) != NULL) { 7250 spa_namespace_exit(FTAG); 7251 return (SET_ERROR(EEXIST)); 7252 } 7253 7254 /* 7255 * Allocate a new spa_t structure. 7256 */ 7257 nvl = fnvlist_alloc(); 7258 fnvlist_add_string(nvl, ZPOOL_CONFIG_POOL_NAME, pool); 7259 (void) nvlist_lookup_string(props, 7260 zpool_prop_to_name(ZPOOL_PROP_ALTROOT), &altroot); 7261 spa = spa_add(poolname, nvl, altroot); 7262 fnvlist_free(nvl); 7263 spa_activate(spa, spa_mode_global); 7264 7265 if (props && (error = spa_prop_validate(spa, props))) { 7266 spa_deactivate(spa); 7267 spa_remove(spa); 7268 spa_namespace_exit(FTAG); 7269 return (error); 7270 } 7271 7272 /* 7273 * Temporary pool names should never be written to disk. 7274 */ 7275 if (poolname != pool) 7276 spa->spa_import_flags |= ZFS_IMPORT_TEMP_NAME; 7277 7278 has_features = B_FALSE; 7279 has_encryption = B_FALSE; 7280 has_allocclass = B_FALSE; 7281 has_draid = B_FALSE; 7282 has_draid_fdomains = B_FALSE; 7283 for (nvpair_t *elem = nvlist_next_nvpair(props, NULL); 7284 elem != NULL; elem = nvlist_next_nvpair(props, elem)) { 7285 if (zpool_prop_feature(nvpair_name(elem))) { 7286 has_features = B_TRUE; 7287 7288 feat_name = strchr(nvpair_name(elem), '@') + 1; 7289 VERIFY0(zfeature_lookup_name(feat_name, &feat)); 7290 if (feat == SPA_FEATURE_ENCRYPTION) 7291 has_encryption = B_TRUE; 7292 if (feat == SPA_FEATURE_ALLOCATION_CLASSES) 7293 has_allocclass = B_TRUE; 7294 if (feat == SPA_FEATURE_DRAID) 7295 has_draid = B_TRUE; 7296 if (feat == SPA_FEATURE_DRAID_FAIL_DOMAINS) 7297 has_draid_fdomains = B_TRUE; 7298 } 7299 } 7300 7301 /* verify encryption params, if they were provided */ 7302 if (dcp != NULL) { 7303 error = spa_create_check_encryption_params(dcp, has_encryption); 7304 if (error != 0) { 7305 spa_deactivate(spa); 7306 spa_remove(spa); 7307 spa_namespace_exit(FTAG); 7308 return (error); 7309 } 7310 } 7311 if (!has_allocclass && zfs_special_devs(nvroot, NULL)) { 7312 spa_deactivate(spa); 7313 spa_remove(spa); 7314 spa_namespace_exit(FTAG); 7315 return (ENOTSUP); 7316 } 7317 7318 if (has_features || nvlist_lookup_uint64(props, 7319 zpool_prop_to_name(ZPOOL_PROP_VERSION), &version) != 0) { 7320 version = SPA_VERSION; 7321 } 7322 ASSERT(SPA_VERSION_IS_SUPPORTED(version)); 7323 7324 spa->spa_first_txg = txg; 7325 spa->spa_uberblock.ub_txg = txg - 1; 7326 spa->spa_uberblock.ub_version = version; 7327 spa->spa_ubsync = spa->spa_uberblock; 7328 spa->spa_load_state = SPA_LOAD_CREATE; 7329 spa->spa_removing_phys.sr_state = DSS_NONE; 7330 spa->spa_removing_phys.sr_removing_vdev = -1; 7331 spa->spa_removing_phys.sr_prev_indirect_vdev = -1; 7332 spa->spa_indirect_vdevs_loaded = B_TRUE; 7333 spa->spa_deflate = (version >= SPA_VERSION_RAIDZ_DEFLATE); 7334 7335 /* 7336 * Create "The Godfather" zio to hold all async IOs 7337 */ 7338 spa->spa_async_zio_root = kmem_alloc(max_ncpus * sizeof (void *), 7339 KM_SLEEP); 7340 for (int i = 0; i < max_ncpus; i++) { 7341 spa->spa_async_zio_root[i] = zio_root(spa, NULL, NULL, 7342 ZIO_FLAG_CANFAIL | ZIO_FLAG_SPECULATIVE | 7343 ZIO_FLAG_GODFATHER); 7344 } 7345 7346 /* 7347 * Create the root vdev. 7348 */ 7349 spa_config_enter(spa, SCL_ALL, FTAG, RW_WRITER); 7350 7351 error = spa_config_parse(spa, &rvd, nvroot, NULL, 0, VDEV_ALLOC_ADD); 7352 7353 ASSERT(error != 0 || rvd != NULL); 7354 ASSERT(error != 0 || spa->spa_root_vdev == rvd); 7355 7356 if (error == 0 && !zfs_allocatable_devs(nvroot)) 7357 error = SET_ERROR(EINVAL); 7358 7359 if (error == 0 && 7360 (error = vdev_create(rvd, txg, B_FALSE)) == 0 && 7361 (error = vdev_draid_spare_create(nvroot, rvd, &ndraid, 7362 &draid_nfgroup, 0)) == 0 && 7363 (ndraid == 0 || has_draid || (error = SET_ERROR(ENOTSUP))) && 7364 (draid_nfgroup == 0 || has_draid_fdomains || 7365 (error = SET_ERROR(ENOTSUP))) && error == 0 && 7366 (error = spa_validate_aux(spa, nvroot, txg, VDEV_ALLOC_ADD)) == 0) { 7367 /* 7368 * instantiate the metaslab groups (this will dirty the vdevs) 7369 * we can no longer error exit past this point 7370 */ 7371 for (int c = 0; error == 0 && c < rvd->vdev_children; c++) { 7372 vdev_t *vd = rvd->vdev_child[c]; 7373 7374 vdev_metaslab_set_size(vd); 7375 vdev_expand(vd, txg); 7376 } 7377 } 7378 7379 spa_config_exit(spa, SCL_ALL, FTAG); 7380 7381 if (error != 0) { 7382 if (errinfo != NULL) { 7383 *errinfo = spa->spa_create_info; 7384 spa->spa_create_info = NULL; 7385 } 7386 spa_unload(spa); 7387 spa_deactivate(spa); 7388 spa_remove(spa); 7389 spa_namespace_exit(FTAG); 7390 return (error); 7391 } 7392 7393 /* 7394 * Get the list of spares, if specified. 7395 */ 7396 if (nvlist_lookup_nvlist_array(nvroot, ZPOOL_CONFIG_SPARES, 7397 &spares, &nspares) == 0) { 7398 spa->spa_spares.sav_config = fnvlist_alloc(); 7399 fnvlist_add_nvlist_array(spa->spa_spares.sav_config, 7400 ZPOOL_CONFIG_SPARES, (const nvlist_t * const *)spares, 7401 nspares); 7402 spa_config_enter(spa, SCL_ALL, FTAG, RW_WRITER); 7403 spa_load_spares(spa); 7404 spa_config_exit(spa, SCL_ALL, FTAG); 7405 spa->spa_spares.sav_sync = B_TRUE; 7406 } 7407 7408 /* 7409 * Get the list of level 2 cache devices, if specified. 7410 */ 7411 if (nvlist_lookup_nvlist_array(nvroot, ZPOOL_CONFIG_L2CACHE, 7412 &l2cache, &nl2cache) == 0) { 7413 VERIFY0(nvlist_alloc(&spa->spa_l2cache.sav_config, 7414 NV_UNIQUE_NAME, KM_SLEEP)); 7415 fnvlist_add_nvlist_array(spa->spa_l2cache.sav_config, 7416 ZPOOL_CONFIG_L2CACHE, (const nvlist_t * const *)l2cache, 7417 nl2cache); 7418 spa_config_enter(spa, SCL_ALL, FTAG, RW_WRITER); 7419 spa_load_l2cache(spa); 7420 spa_config_exit(spa, SCL_ALL, FTAG); 7421 spa->spa_l2cache.sav_sync = B_TRUE; 7422 } 7423 7424 spa->spa_is_initializing = B_TRUE; 7425 spa->spa_dsl_pool = dp = dsl_pool_create(spa, zplprops, dcp, txg); 7426 spa->spa_is_initializing = B_FALSE; 7427 7428 /* 7429 * Create DDTs (dedup tables). 7430 */ 7431 ddt_create(spa); 7432 /* 7433 * Create BRT table and BRT table object. 7434 */ 7435 brt_create(spa); 7436 7437 spa_update_dspace(spa); 7438 7439 tx = dmu_tx_create_assigned(dp, txg); 7440 7441 /* 7442 * Create the pool's history object. 7443 */ 7444 if (version >= SPA_VERSION_ZPOOL_HISTORY && !spa->spa_history) 7445 spa_history_create_obj(spa, tx); 7446 7447 spa_event_notify(spa, NULL, NULL, ESC_ZFS_POOL_CREATE); 7448 spa_history_log_version(spa, "create", tx); 7449 7450 /* 7451 * Create the pool config object. 7452 */ 7453 spa->spa_config_object = dmu_object_alloc(spa->spa_meta_objset, 7454 DMU_OT_PACKED_NVLIST, SPA_CONFIG_BLOCKSIZE, 7455 DMU_OT_PACKED_NVLIST_SIZE, sizeof (uint64_t), tx); 7456 7457 if (zap_add(spa->spa_meta_objset, 7458 DMU_POOL_DIRECTORY_OBJECT, DMU_POOL_CONFIG, 7459 sizeof (uint64_t), 1, &spa->spa_config_object, tx) != 0) { 7460 cmn_err(CE_PANIC, "failed to add pool config"); 7461 } 7462 7463 if (zap_add(spa->spa_meta_objset, 7464 DMU_POOL_DIRECTORY_OBJECT, DMU_POOL_CREATION_VERSION, 7465 sizeof (uint64_t), 1, &version, tx) != 0) { 7466 cmn_err(CE_PANIC, "failed to add pool version"); 7467 } 7468 7469 /* Newly created pools with the right version are always deflated. */ 7470 if (version >= SPA_VERSION_RAIDZ_DEFLATE) { 7471 if (zap_add(spa->spa_meta_objset, 7472 DMU_POOL_DIRECTORY_OBJECT, DMU_POOL_DEFLATE, 7473 sizeof (uint64_t), 1, &spa->spa_deflate, tx) != 0) { 7474 cmn_err(CE_PANIC, "failed to add deflate"); 7475 } 7476 } 7477 7478 /* 7479 * Create the deferred-free bpobj. Turn off compression 7480 * because sync-to-convergence takes longer if the blocksize 7481 * keeps changing. 7482 */ 7483 obj = bpobj_alloc(spa->spa_meta_objset, 1 << 14, tx); 7484 dmu_object_set_compress(spa->spa_meta_objset, obj, 7485 ZIO_COMPRESS_OFF, tx); 7486 if (zap_add(spa->spa_meta_objset, 7487 DMU_POOL_DIRECTORY_OBJECT, DMU_POOL_SYNC_BPOBJ, 7488 sizeof (uint64_t), 1, &obj, tx) != 0) { 7489 cmn_err(CE_PANIC, "failed to add bpobj"); 7490 } 7491 VERIFY3U(0, ==, bpobj_open(&spa->spa_deferred_bpobj, 7492 spa->spa_meta_objset, obj)); 7493 7494 /* 7495 * Generate some random noise for salted checksums to operate on. 7496 */ 7497 (void) random_get_pseudo_bytes(spa->spa_cksum_salt.zcs_bytes, 7498 sizeof (spa->spa_cksum_salt.zcs_bytes)); 7499 7500 /* 7501 * Set pool properties. 7502 */ 7503 spa->spa_bootfs = zpool_prop_default_numeric(ZPOOL_PROP_BOOTFS); 7504 spa->spa_delegation = zpool_prop_default_numeric(ZPOOL_PROP_DELEGATION); 7505 spa->spa_failmode = zpool_prop_default_numeric(ZPOOL_PROP_FAILUREMODE); 7506 spa->spa_autoexpand = zpool_prop_default_numeric(ZPOOL_PROP_AUTOEXPAND); 7507 spa->spa_multihost = zpool_prop_default_numeric(ZPOOL_PROP_MULTIHOST); 7508 spa->spa_autotrim = zpool_prop_default_numeric(ZPOOL_PROP_AUTOTRIM); 7509 spa->spa_dedup_table_quota = 7510 zpool_prop_default_numeric(ZPOOL_PROP_DEDUP_TABLE_QUOTA); 7511 7512 if (props != NULL) { 7513 spa_configfile_set(spa, props, B_FALSE); 7514 spa_sync_props(props, tx); 7515 } 7516 7517 for (int i = 0; i < ndraid; i++) 7518 spa_feature_incr(spa, SPA_FEATURE_DRAID, tx); 7519 7520 for (int i = 0; i < draid_nfgroup; i++) 7521 spa_feature_incr(spa, SPA_FEATURE_DRAID_FAIL_DOMAINS, tx); 7522 7523 dmu_tx_commit(tx); 7524 7525 spa->spa_sync_on = B_TRUE; 7526 txg_sync_start(dp); 7527 mmp_thread_start(spa); 7528 txg_wait_synced(dp, txg); 7529 7530 spa_spawn_aux_threads(spa); 7531 7532 spa_write_cachefile(spa, B_FALSE, B_TRUE, B_TRUE); 7533 7534 /* 7535 * Don't count references from objsets that are already closed 7536 * and are making their way through the eviction process. 7537 */ 7538 spa_evicting_os_wait(spa); 7539 spa->spa_minref = zfs_refcount_count(&spa->spa_refcount); 7540 spa->spa_load_state = SPA_LOAD_NONE; 7541 7542 spa_import_os(spa); 7543 7544 spa_namespace_exit(FTAG); 7545 7546 return (0); 7547 } 7548 7549 /* 7550 * Import a non-root pool into the system. 7551 */ 7552 int 7553 spa_import(char *pool, nvlist_t *config, nvlist_t *props, uint64_t flags) 7554 { 7555 spa_t *spa; 7556 const char *altroot = NULL; 7557 spa_load_state_t state = SPA_LOAD_IMPORT; 7558 zpool_load_policy_t policy; 7559 spa_mode_t mode = spa_mode_global; 7560 uint64_t readonly = B_FALSE; 7561 int error; 7562 nvlist_t *nvroot; 7563 nvlist_t **spares, **l2cache; 7564 uint_t nspares, nl2cache; 7565 7566 /* 7567 * If a pool with this name exists, return failure. 7568 */ 7569 spa_namespace_enter(FTAG); 7570 if (spa_lookup(pool) != NULL) { 7571 spa_namespace_exit(FTAG); 7572 return (SET_ERROR(EEXIST)); 7573 } 7574 7575 /* 7576 * Create and initialize the spa structure. 7577 */ 7578 (void) nvlist_lookup_string(props, 7579 zpool_prop_to_name(ZPOOL_PROP_ALTROOT), &altroot); 7580 (void) nvlist_lookup_uint64(props, 7581 zpool_prop_to_name(ZPOOL_PROP_READONLY), &readonly); 7582 if (readonly) 7583 mode = SPA_MODE_READ; 7584 spa = spa_add(pool, config, altroot); 7585 spa->spa_import_flags = flags; 7586 7587 /* 7588 * Verbatim import - Take a pool and insert it into the namespace 7589 * as if it had been loaded at boot. 7590 */ 7591 if (spa->spa_import_flags & ZFS_IMPORT_VERBATIM) { 7592 if (props != NULL) 7593 spa_configfile_set(spa, props, B_FALSE); 7594 7595 spa_write_cachefile(spa, B_FALSE, B_TRUE, B_FALSE); 7596 spa_event_notify(spa, NULL, NULL, ESC_ZFS_POOL_IMPORT); 7597 zfs_dbgmsg("spa_import: verbatim import of %s", pool); 7598 spa_namespace_exit(FTAG); 7599 return (0); 7600 } 7601 7602 spa_activate(spa, mode); 7603 7604 /* 7605 * Don't start async tasks until we know everything is healthy. 7606 */ 7607 spa_async_suspend(spa); 7608 7609 zpool_get_load_policy(config, &policy); 7610 if (policy.zlp_rewind & ZPOOL_DO_REWIND) 7611 state = SPA_LOAD_RECOVER; 7612 7613 spa->spa_config_source = SPA_CONFIG_SRC_TRYIMPORT; 7614 7615 if (state != SPA_LOAD_RECOVER) { 7616 spa->spa_last_ubsync_txg = spa->spa_load_txg = 0; 7617 zfs_dbgmsg("spa_import: importing %s", pool); 7618 } else { 7619 zfs_dbgmsg("spa_import: importing %s, max_txg=%lld " 7620 "(RECOVERY MODE)", pool, (longlong_t)policy.zlp_txg); 7621 } 7622 error = spa_load_best(spa, state, policy.zlp_txg, policy.zlp_rewind); 7623 7624 /* 7625 * Propagate anything learned while loading the pool and pass it 7626 * back to caller (i.e. rewind info, missing devices, etc). 7627 */ 7628 fnvlist_add_nvlist(config, ZPOOL_CONFIG_LOAD_INFO, spa->spa_load_info); 7629 7630 spa_config_enter(spa, SCL_ALL, FTAG, RW_WRITER); 7631 /* 7632 * Toss any existing sparelist, as it doesn't have any validity 7633 * anymore, and conflicts with spa_has_spare(). 7634 */ 7635 if (spa->spa_spares.sav_config) { 7636 nvlist_free(spa->spa_spares.sav_config); 7637 spa->spa_spares.sav_config = NULL; 7638 spa_load_spares(spa); 7639 } 7640 if (spa->spa_l2cache.sav_config) { 7641 nvlist_free(spa->spa_l2cache.sav_config); 7642 spa->spa_l2cache.sav_config = NULL; 7643 spa_load_l2cache(spa); 7644 } 7645 7646 nvroot = fnvlist_lookup_nvlist(config, ZPOOL_CONFIG_VDEV_TREE); 7647 spa_config_exit(spa, SCL_ALL, FTAG); 7648 7649 if (props != NULL) 7650 spa_configfile_set(spa, props, B_FALSE); 7651 7652 if (error != 0 || (props && spa_writeable(spa) && 7653 (error = spa_prop_set(spa, props)))) { 7654 spa_unload(spa); 7655 spa_deactivate(spa); 7656 spa_remove(spa); 7657 spa_namespace_exit(FTAG); 7658 return (error); 7659 } 7660 7661 spa_async_resume(spa); 7662 7663 /* 7664 * Override any spares and level 2 cache devices as specified by 7665 * the user, as these may have correct device names/devids, etc. 7666 */ 7667 if (nvlist_lookup_nvlist_array(nvroot, ZPOOL_CONFIG_SPARES, 7668 &spares, &nspares) == 0) { 7669 if (spa->spa_spares.sav_config) 7670 fnvlist_remove(spa->spa_spares.sav_config, 7671 ZPOOL_CONFIG_SPARES); 7672 else 7673 spa->spa_spares.sav_config = fnvlist_alloc(); 7674 fnvlist_add_nvlist_array(spa->spa_spares.sav_config, 7675 ZPOOL_CONFIG_SPARES, (const nvlist_t * const *)spares, 7676 nspares); 7677 spa_config_enter(spa, SCL_ALL, FTAG, RW_WRITER); 7678 spa_load_spares(spa); 7679 spa_config_exit(spa, SCL_ALL, FTAG); 7680 spa->spa_spares.sav_sync = B_TRUE; 7681 spa->spa_spares.sav_label_sync = B_TRUE; 7682 } 7683 if (nvlist_lookup_nvlist_array(nvroot, ZPOOL_CONFIG_L2CACHE, 7684 &l2cache, &nl2cache) == 0) { 7685 if (spa->spa_l2cache.sav_config) 7686 fnvlist_remove(spa->spa_l2cache.sav_config, 7687 ZPOOL_CONFIG_L2CACHE); 7688 else 7689 spa->spa_l2cache.sav_config = fnvlist_alloc(); 7690 fnvlist_add_nvlist_array(spa->spa_l2cache.sav_config, 7691 ZPOOL_CONFIG_L2CACHE, (const nvlist_t * const *)l2cache, 7692 nl2cache); 7693 spa_config_enter(spa, SCL_ALL, FTAG, RW_WRITER); 7694 spa_load_l2cache(spa); 7695 spa_config_exit(spa, SCL_ALL, FTAG); 7696 spa->spa_l2cache.sav_sync = B_TRUE; 7697 spa->spa_l2cache.sav_label_sync = B_TRUE; 7698 } 7699 7700 /* 7701 * Check for any removed devices. 7702 */ 7703 if (spa->spa_autoreplace) { 7704 spa_aux_check_removed(&spa->spa_spares); 7705 spa_aux_check_removed(&spa->spa_l2cache); 7706 } 7707 7708 if (spa_writeable(spa)) { 7709 /* 7710 * Update the config cache to include the newly-imported pool. 7711 */ 7712 spa_config_update(spa, SPA_CONFIG_UPDATE_POOL); 7713 } 7714 7715 /* 7716 * It's possible that the pool was expanded while it was exported. 7717 * We kick off an async task to handle this for us. 7718 */ 7719 spa_async_request(spa, SPA_ASYNC_AUTOEXPAND); 7720 7721 spa_history_log_version(spa, "import", NULL); 7722 7723 spa_event_notify(spa, NULL, NULL, ESC_ZFS_POOL_IMPORT); 7724 7725 spa_namespace_exit(FTAG); 7726 7727 zvol_create_minors(pool); 7728 7729 spa_import_os(spa); 7730 7731 return (0); 7732 } 7733 7734 nvlist_t * 7735 spa_tryimport(nvlist_t *tryconfig) 7736 { 7737 nvlist_t *config = NULL; 7738 const char *poolname, *cachefile; 7739 spa_t *spa; 7740 uint64_t state; 7741 int error; 7742 zpool_load_policy_t policy; 7743 7744 if (nvlist_lookup_string(tryconfig, ZPOOL_CONFIG_POOL_NAME, &poolname)) 7745 return (NULL); 7746 7747 if (nvlist_lookup_uint64(tryconfig, ZPOOL_CONFIG_POOL_STATE, &state)) 7748 return (NULL); 7749 7750 /* 7751 * Create and initialize the spa structure. 7752 */ 7753 char *name = kmem_alloc(MAXPATHLEN, KM_SLEEP); 7754 (void) snprintf(name, MAXPATHLEN, "%s-%llx-%s", 7755 TRYIMPORT_NAME, (u_longlong_t)(uintptr_t)curthread, poolname); 7756 7757 spa_namespace_enter(FTAG); 7758 spa = spa_add(name, tryconfig, NULL); 7759 spa_activate(spa, SPA_MODE_READ); 7760 kmem_free(name, MAXPATHLEN); 7761 7762 spa->spa_load_name = spa_strdup(poolname); 7763 7764 /* 7765 * Rewind pool if a max txg was provided. 7766 */ 7767 zpool_get_load_policy(spa->spa_config, &policy); 7768 if (policy.zlp_txg != UINT64_MAX) { 7769 spa->spa_load_max_txg = policy.zlp_txg; 7770 spa->spa_extreme_rewind = B_TRUE; 7771 zfs_dbgmsg("spa_tryimport: importing %s, max_txg=%lld", 7772 spa_load_name(spa), (longlong_t)policy.zlp_txg); 7773 } else { 7774 zfs_dbgmsg("spa_tryimport: importing %s", spa_load_name(spa)); 7775 } 7776 7777 if (nvlist_lookup_string(tryconfig, ZPOOL_CONFIG_CACHEFILE, &cachefile) 7778 == 0) { 7779 zfs_dbgmsg("spa_tryimport: using cachefile '%s'", cachefile); 7780 spa->spa_config_source = SPA_CONFIG_SRC_CACHEFILE; 7781 } else { 7782 spa->spa_config_source = SPA_CONFIG_SRC_SCAN; 7783 } 7784 7785 /* 7786 * spa_import() relies on a pool config fetched by spa_try_import() 7787 * for spare/cache devices. Import flags are not passed to 7788 * spa_tryimport(), which makes it return early due to a missing log 7789 * device and missing retrieving the cache device and spare eventually. 7790 * Passing ZFS_IMPORT_MISSING_LOG to spa_tryimport() makes it fetch 7791 * the correct configuration regardless of the missing log device. 7792 */ 7793 spa->spa_import_flags |= ZFS_IMPORT_MISSING_LOG; 7794 7795 error = spa_load(spa, SPA_LOAD_TRYIMPORT, SPA_IMPORT_EXISTING); 7796 7797 /* 7798 * If 'tryconfig' was at least parsable, return the current config. 7799 */ 7800 if (spa->spa_root_vdev != NULL) { 7801 config = spa_config_generate(spa, NULL, -1ULL, B_TRUE); 7802 fnvlist_add_string(config, ZPOOL_CONFIG_POOL_NAME, 7803 spa_load_name(spa)); 7804 fnvlist_add_uint64(config, ZPOOL_CONFIG_POOL_STATE, state); 7805 fnvlist_add_uint64(config, ZPOOL_CONFIG_TIMESTAMP, 7806 spa->spa_uberblock.ub_timestamp); 7807 fnvlist_add_nvlist(config, ZPOOL_CONFIG_LOAD_INFO, 7808 spa->spa_load_info); 7809 fnvlist_add_uint64(config, ZPOOL_CONFIG_ERRATA, 7810 spa->spa_errata); 7811 7812 /* 7813 * If the bootfs property exists on this pool then we 7814 * copy it out so that external consumers can tell which 7815 * pools are bootable. 7816 */ 7817 if ((!error || error == EEXIST) && spa->spa_bootfs) { 7818 char *tmpname = kmem_alloc(MAXPATHLEN, KM_SLEEP); 7819 7820 /* 7821 * We have to play games with the name since the 7822 * pool was opened as TRYIMPORT_NAME. 7823 */ 7824 if (dsl_dsobj_to_dsname(spa_name(spa), 7825 spa->spa_bootfs, tmpname) == 0) { 7826 char *cp; 7827 char *dsname; 7828 7829 dsname = kmem_alloc(MAXPATHLEN, KM_SLEEP); 7830 7831 cp = strchr(tmpname, '/'); 7832 if (cp == NULL) { 7833 (void) strlcpy(dsname, tmpname, 7834 MAXPATHLEN); 7835 } else { 7836 (void) snprintf(dsname, MAXPATHLEN, 7837 "%s/%s", spa_load_name(spa), ++cp); 7838 } 7839 fnvlist_add_string(config, ZPOOL_CONFIG_BOOTFS, 7840 dsname); 7841 kmem_free(dsname, MAXPATHLEN); 7842 } 7843 kmem_free(tmpname, MAXPATHLEN); 7844 } 7845 7846 /* 7847 * Add the list of hot spares and level 2 cache devices. 7848 */ 7849 spa_config_enter(spa, SCL_CONFIG, FTAG, RW_READER); 7850 spa_add_spares(spa, config); 7851 spa_add_l2cache(spa, config); 7852 spa_config_exit(spa, SCL_CONFIG, FTAG); 7853 } 7854 7855 spa_unload(spa); 7856 spa_deactivate(spa); 7857 spa_remove(spa); 7858 spa_namespace_exit(FTAG); 7859 7860 return (config); 7861 } 7862 7863 /* 7864 * Pool export/destroy 7865 * 7866 * The act of destroying or exporting a pool is very simple. We make sure there 7867 * is no more pending I/O and any references to the pool are gone. Then, we 7868 * update the pool state and sync all the labels to disk, removing the 7869 * configuration from the cache afterwards. If the 'hardforce' flag is set, then 7870 * we don't sync the labels or remove the configuration cache. 7871 */ 7872 static int 7873 spa_export_common(const char *pool, int new_state, nvlist_t **oldconfig, 7874 boolean_t force, boolean_t hardforce) 7875 { 7876 int error = 0; 7877 spa_t *spa; 7878 hrtime_t export_start = gethrtime(); 7879 7880 if (oldconfig) 7881 *oldconfig = NULL; 7882 7883 if (!(spa_mode_global & SPA_MODE_WRITE)) 7884 return (SET_ERROR(EROFS)); 7885 7886 spa_namespace_enter(FTAG); 7887 if ((spa = spa_lookup(pool)) == NULL) { 7888 spa_namespace_exit(FTAG); 7889 return (SET_ERROR(ENOENT)); 7890 } 7891 7892 if (spa->spa_is_exporting) { 7893 /* the pool is being exported by another thread */ 7894 spa_namespace_exit(FTAG); 7895 return (SET_ERROR(ZFS_ERR_EXPORT_IN_PROGRESS)); 7896 } 7897 spa->spa_is_exporting = B_TRUE; 7898 7899 /* 7900 * Put a hold on the pool, drop the namespace lock, stop async tasks 7901 * and see if we can export. 7902 */ 7903 spa_open_ref(spa, FTAG); 7904 spa_namespace_exit(FTAG); 7905 #ifdef ZFS_DEBUG 7906 spa_condense_debug_cancel(spa); 7907 #endif 7908 spa_async_suspend(spa); 7909 7910 spa_namespace_enter(FTAG); 7911 spa->spa_export_thread = curthread; 7912 spa_close(spa, FTAG); 7913 7914 if (spa->spa_state == POOL_STATE_UNINITIALIZED) { 7915 spa_namespace_exit(FTAG); 7916 goto export_spa; 7917 } 7918 7919 /* 7920 * The pool will be in core if it's openable, in which case we can 7921 * modify its state. Objsets may be open only because they're dirty, 7922 * so we have to force it to sync before checking spa_refcnt. 7923 */ 7924 if (spa->spa_sync_on) { 7925 txg_wait_synced(spa->spa_dsl_pool, 0); 7926 spa_evicting_os_wait(spa); 7927 } 7928 7929 /* 7930 * A pool cannot be exported or destroyed if there are active 7931 * references. If we are resetting a pool, allow references by 7932 * fault injection handlers. 7933 */ 7934 if (!spa_refcount_zero(spa) || (spa->spa_inject_ref != 0)) { 7935 error = SET_ERROR(EBUSY); 7936 goto fail; 7937 } 7938 7939 spa_namespace_exit(FTAG); 7940 /* 7941 * At this point we no longer hold the spa_namespace_lock and 7942 * there were no references on the spa. Future spa_lookups will 7943 * notice the spa->spa_export_thread and wait until we signal 7944 * that we are finshed. 7945 */ 7946 7947 if (spa->spa_zvol_taskq) { 7948 zvol_remove_minors(spa, spa_name(spa), B_TRUE); 7949 taskq_wait(spa->spa_zvol_taskq); 7950 } 7951 7952 if (spa->spa_sync_on) { 7953 vdev_t *rvd = spa->spa_root_vdev; 7954 /* 7955 * A pool cannot be exported if it has an active shared spare. 7956 * This is to prevent other pools stealing the active spare 7957 * from an exported pool. At user's own will, such pool can 7958 * be forcedly exported. 7959 */ 7960 if (!force && new_state == POOL_STATE_EXPORTED && 7961 spa_has_active_shared_spare(spa)) { 7962 error = SET_ERROR(EXDEV); 7963 spa_namespace_enter(FTAG); 7964 goto fail; 7965 } 7966 7967 /* 7968 * We're about to export or destroy this pool. Make sure 7969 * we stop all initialization and trim activity here before 7970 * we set the spa_final_txg. This will ensure that all 7971 * dirty data resulting from the initialization is 7972 * committed to disk before we unload the pool. 7973 */ 7974 vdev_initialize_stop_all(rvd, VDEV_INITIALIZE_ACTIVE); 7975 vdev_trim_stop_all(rvd, VDEV_TRIM_ACTIVE); 7976 vdev_autotrim_stop_all(spa); 7977 vdev_rebuild_stop_all(spa); 7978 l2arc_spa_rebuild_stop(spa); 7979 7980 /* 7981 * We want this to be reflected on every label, 7982 * so mark them all dirty. spa_unload() will do the 7983 * final sync that pushes these changes out. 7984 */ 7985 if (new_state != POOL_STATE_UNINITIALIZED && !hardforce) { 7986 spa_config_enter(spa, SCL_ALL, FTAG, RW_WRITER); 7987 spa->spa_state = new_state; 7988 vdev_config_dirty(rvd); 7989 spa_config_exit(spa, SCL_ALL, FTAG); 7990 } 7991 7992 if (spa_should_sync_time_logger_on_unload(spa)) 7993 spa_unload_sync_time_logger(spa); 7994 7995 /* 7996 * If the log space map feature is enabled and the pool is 7997 * getting exported (but not destroyed), we want to spend some 7998 * time flushing as many metaslabs as we can in an attempt to 7999 * destroy log space maps and save import time. This has to be 8000 * done before we set the spa_final_txg, otherwise 8001 * spa_sync() -> spa_flush_metaslabs() may dirty the final TXGs. 8002 * spa_should_flush_logs_on_unload() should be called after 8003 * spa_state has been set to the new_state. 8004 */ 8005 if (spa_should_flush_logs_on_unload(spa)) 8006 spa_unload_log_sm_flush_all(spa); 8007 8008 if (new_state != POOL_STATE_UNINITIALIZED && !hardforce) { 8009 spa_config_enter(spa, SCL_ALL, FTAG, RW_WRITER); 8010 spa->spa_final_txg = spa_last_synced_txg(spa) + 8011 TXG_DEFER_SIZE + 1; 8012 spa_config_exit(spa, SCL_ALL, FTAG); 8013 } 8014 } 8015 8016 export_spa: 8017 spa_export_os(spa); 8018 8019 if (new_state == POOL_STATE_DESTROYED) 8020 spa_event_notify(spa, NULL, NULL, ESC_ZFS_POOL_DESTROY); 8021 else if (new_state == POOL_STATE_EXPORTED) 8022 spa_event_notify(spa, NULL, NULL, ESC_ZFS_POOL_EXPORT); 8023 8024 if (spa->spa_state != POOL_STATE_UNINITIALIZED) { 8025 spa_unload(spa); 8026 spa_deactivate(spa); 8027 } 8028 8029 if (oldconfig && spa->spa_config) 8030 *oldconfig = fnvlist_dup(spa->spa_config); 8031 8032 if (new_state == POOL_STATE_EXPORTED) 8033 zio_handle_export_delay(spa, gethrtime() - export_start); 8034 8035 /* 8036 * Take the namespace lock for the actual spa_t removal 8037 */ 8038 spa_namespace_enter(FTAG); 8039 if (new_state != POOL_STATE_UNINITIALIZED) { 8040 if (!hardforce) 8041 spa_write_cachefile(spa, B_TRUE, B_TRUE, B_FALSE); 8042 spa_remove(spa); 8043 } else { 8044 /* 8045 * If spa_remove() is not called for this spa_t and 8046 * there is any possibility that it can be reused, 8047 * we make sure to reset the exporting flag. 8048 */ 8049 spa->spa_is_exporting = B_FALSE; 8050 spa->spa_export_thread = NULL; 8051 } 8052 8053 /* 8054 * Wake up any waiters in spa_lookup() 8055 */ 8056 spa_namespace_broadcast(); 8057 spa_namespace_exit(FTAG); 8058 return (0); 8059 8060 fail: 8061 spa->spa_is_exporting = B_FALSE; 8062 spa->spa_export_thread = NULL; 8063 8064 spa_async_resume(spa); 8065 /* 8066 * Wake up any waiters in spa_lookup() 8067 */ 8068 spa_namespace_broadcast(); 8069 spa_namespace_exit(FTAG); 8070 return (error); 8071 } 8072 8073 /* 8074 * Destroy a storage pool. 8075 */ 8076 int 8077 spa_destroy(const char *pool) 8078 { 8079 return (spa_export_common(pool, POOL_STATE_DESTROYED, NULL, 8080 B_FALSE, B_FALSE)); 8081 } 8082 8083 /* 8084 * Export a storage pool. 8085 */ 8086 int 8087 spa_export(const char *pool, nvlist_t **oldconfig, boolean_t force, 8088 boolean_t hardforce) 8089 { 8090 return (spa_export_common(pool, POOL_STATE_EXPORTED, oldconfig, 8091 force, hardforce)); 8092 } 8093 8094 /* 8095 * Similar to spa_export(), this unloads the spa_t without actually removing it 8096 * from the namespace in any way. 8097 */ 8098 int 8099 spa_reset(const char *pool) 8100 { 8101 return (spa_export_common(pool, POOL_STATE_UNINITIALIZED, NULL, 8102 B_FALSE, B_FALSE)); 8103 } 8104 8105 /* 8106 * ========================================================================== 8107 * Device manipulation 8108 * ========================================================================== 8109 */ 8110 8111 /* 8112 * This is called as a synctask to increment the draid feature flag 8113 */ 8114 static void 8115 spa_draid_feature_incr(void *arg, dmu_tx_t *tx) 8116 { 8117 spa_t *spa = dmu_tx_pool(tx)->dp_spa; 8118 int draid = (int)(uintptr_t)arg; 8119 8120 for (int c = 0; c < draid; c++) 8121 spa_feature_incr(spa, SPA_FEATURE_DRAID, tx); 8122 } 8123 8124 /* 8125 * This is called as a synctask to increment the draid_fail_domains feature flag 8126 */ 8127 static void 8128 spa_draid_fdomains_feature_incr(void *arg, dmu_tx_t *tx) 8129 { 8130 spa_t *spa = dmu_tx_pool(tx)->dp_spa; 8131 int nfgrp = (int)(uintptr_t)arg; 8132 8133 for (int c = 0; c < nfgrp; c++) 8134 spa_feature_incr(spa, SPA_FEATURE_DRAID_FAIL_DOMAINS, tx); 8135 } 8136 8137 /* 8138 * Add a device to a storage pool. 8139 */ 8140 int 8141 spa_vdev_add(spa_t *spa, nvlist_t *nvroot, boolean_t check_ashift) 8142 { 8143 uint64_t txg, ndraid = 0, draid_nfgroup = 0; 8144 int error; 8145 vdev_t *rvd = spa->spa_root_vdev; 8146 vdev_t *vd, *tvd; 8147 nvlist_t **spares, **l2cache; 8148 uint_t nspares, nl2cache; 8149 8150 ASSERT(spa_writeable(spa)); 8151 8152 txg = spa_vdev_enter(spa); 8153 8154 if ((error = spa_config_parse(spa, &vd, nvroot, NULL, 0, 8155 VDEV_ALLOC_ADD)) != 0) 8156 return (spa_vdev_exit(spa, NULL, txg, error)); 8157 8158 spa->spa_pending_vdev = vd; /* spa_vdev_exit() will clear this */ 8159 8160 if (nvlist_lookup_nvlist_array(nvroot, ZPOOL_CONFIG_SPARES, &spares, 8161 &nspares) != 0) 8162 nspares = 0; 8163 8164 if (nvlist_lookup_nvlist_array(nvroot, ZPOOL_CONFIG_L2CACHE, &l2cache, 8165 &nl2cache) != 0) 8166 nl2cache = 0; 8167 8168 if (vd->vdev_children == 0 && nspares == 0 && nl2cache == 0) 8169 return (spa_vdev_exit(spa, vd, txg, EINVAL)); 8170 8171 if (vd->vdev_children != 0 && 8172 (error = vdev_create(vd, txg, B_FALSE)) != 0) { 8173 return (spa_vdev_exit(spa, vd, txg, error)); 8174 } 8175 8176 /* 8177 * The virtual dRAID spares must be added after vdev tree is created 8178 * and the vdev guids are generated. The guid of their associated 8179 * dRAID is stored in the config and used when opening the spare. 8180 */ 8181 if ((error = vdev_draid_spare_create(nvroot, vd, &ndraid, 8182 &draid_nfgroup, rvd->vdev_children)) == 0) { 8183 8184 if (ndraid > 0 && nvlist_lookup_nvlist_array(nvroot, 8185 ZPOOL_CONFIG_SPARES, &spares, &nspares) != 0) 8186 nspares = 0; 8187 8188 if (draid_nfgroup > 0 && !spa_feature_is_enabled(spa, 8189 SPA_FEATURE_DRAID_FAIL_DOMAINS)) 8190 return (spa_vdev_exit(spa, vd, txg, ENOTSUP)); 8191 } else { 8192 return (spa_vdev_exit(spa, vd, txg, error)); 8193 } 8194 8195 /* 8196 * We must validate the spares and l2cache devices after checking the 8197 * children. Otherwise, vdev_inuse() will blindly overwrite the spare. 8198 */ 8199 if ((error = spa_validate_aux(spa, nvroot, txg, VDEV_ALLOC_ADD)) != 0) 8200 return (spa_vdev_exit(spa, vd, txg, error)); 8201 8202 /* 8203 * If we are in the middle of a device removal, we can only add 8204 * devices which match the existing devices in the pool. 8205 * If we are in the middle of a removal, or have some indirect 8206 * vdevs, we can not add raidz or dRAID top levels. 8207 */ 8208 if (spa->spa_vdev_removal != NULL || 8209 spa->spa_removing_phys.sr_prev_indirect_vdev != -1) { 8210 for (int c = 0; c < vd->vdev_children; c++) { 8211 tvd = vd->vdev_child[c]; 8212 if (spa->spa_vdev_removal != NULL && 8213 tvd->vdev_ashift != spa->spa_max_ashift) { 8214 return (spa_vdev_exit(spa, vd, txg, EINVAL)); 8215 } 8216 /* Fail if top level vdev is raidz or a dRAID */ 8217 if (vdev_get_nparity(tvd) != 0) 8218 return (spa_vdev_exit(spa, vd, txg, EINVAL)); 8219 8220 /* 8221 * Need the top level mirror to be 8222 * a mirror of leaf vdevs only 8223 */ 8224 if (tvd->vdev_ops == &vdev_mirror_ops) { 8225 for (uint64_t cid = 0; 8226 cid < tvd->vdev_children; cid++) { 8227 vdev_t *cvd = tvd->vdev_child[cid]; 8228 if (!cvd->vdev_ops->vdev_op_leaf) { 8229 return (spa_vdev_exit(spa, vd, 8230 txg, EINVAL)); 8231 } 8232 } 8233 } 8234 } 8235 } 8236 8237 if (check_ashift && spa->spa_max_ashift == spa->spa_min_ashift) { 8238 for (int c = 0; c < vd->vdev_children; c++) { 8239 tvd = vd->vdev_child[c]; 8240 if (tvd->vdev_ashift != spa->spa_max_ashift) { 8241 return (spa_vdev_exit(spa, vd, txg, 8242 ZFS_ERR_ASHIFT_MISMATCH)); 8243 } 8244 } 8245 } 8246 8247 for (int c = 0; c < vd->vdev_children; c++) { 8248 tvd = vd->vdev_child[c]; 8249 vdev_remove_child(vd, tvd); 8250 tvd->vdev_id = rvd->vdev_children; 8251 vdev_add_child(rvd, tvd); 8252 vdev_config_dirty(tvd); 8253 } 8254 8255 if (nspares != 0) { 8256 spa_set_aux_vdevs(&spa->spa_spares, spares, nspares, 8257 ZPOOL_CONFIG_SPARES); 8258 spa_load_spares(spa); 8259 spa->spa_spares.sav_sync = B_TRUE; 8260 } 8261 8262 if (nl2cache != 0) { 8263 spa_set_aux_vdevs(&spa->spa_l2cache, l2cache, nl2cache, 8264 ZPOOL_CONFIG_L2CACHE); 8265 spa_load_l2cache(spa); 8266 spa->spa_l2cache.sav_sync = B_TRUE; 8267 } 8268 8269 /* 8270 * We can't increment a feature while holding spa_vdev so we 8271 * have to do it in a synctask. 8272 */ 8273 if (ndraid != 0) { 8274 dmu_tx_t *tx; 8275 8276 tx = dmu_tx_create_assigned(spa->spa_dsl_pool, txg); 8277 8278 dsl_sync_task_nowait(spa->spa_dsl_pool, spa_draid_feature_incr, 8279 (void *)(uintptr_t)ndraid, tx); 8280 8281 if (draid_nfgroup > 0) 8282 dsl_sync_task_nowait(spa->spa_dsl_pool, 8283 spa_draid_fdomains_feature_incr, 8284 (void *)(uintptr_t)draid_nfgroup, tx); 8285 8286 dmu_tx_commit(tx); 8287 } 8288 8289 /* 8290 * We have to be careful when adding new vdevs to an existing pool. 8291 * If other threads start allocating from these vdevs before we 8292 * sync the config cache, and we lose power, then upon reboot we may 8293 * fail to open the pool because there are DVAs that the config cache 8294 * can't translate. Therefore, we first add the vdevs without 8295 * initializing metaslabs; sync the config cache (via spa_vdev_exit()); 8296 * and then let spa_config_update() initialize the new metaslabs. 8297 * 8298 * spa_load() checks for added-but-not-initialized vdevs, so that 8299 * if we lose power at any point in this sequence, the remaining 8300 * steps will be completed the next time we load the pool. 8301 */ 8302 (void) spa_vdev_exit(spa, vd, txg, 0); 8303 8304 spa_namespace_enter(FTAG); 8305 spa_config_update(spa, SPA_CONFIG_UPDATE_POOL); 8306 spa_event_notify(spa, NULL, NULL, ESC_ZFS_VDEV_ADD); 8307 spa_namespace_exit(FTAG); 8308 8309 return (0); 8310 } 8311 8312 /* 8313 * Given a vdev to be replaced and its parent, check for a possible 8314 * "double spare" condition if a vdev is to be replaced by a spare. When this 8315 * happens, you can get two spares assigned to one failed vdev. 8316 * 8317 * To trigger a double spare condition: 8318 * 8319 * 1. disk1 fails 8320 * 2. 1st spare is kicked in for disk1 and it resilvers 8321 * 3. Someone replaces disk1 with a new blank disk 8322 * 4. New blank disk starts resilvering 8323 * 5. While resilvering, new blank disk has IO errors and faults 8324 * 6. 2nd spare is kicked in for new blank disk 8325 * 7. At this point two spares are kicked in for the original disk1. 8326 * 8327 * It looks like this: 8328 * 8329 * NAME STATE READ WRITE CKSUM 8330 * tank2 DEGRADED 0 0 0 8331 * draid2:6d:10c:2s-0 DEGRADED 0 0 0 8332 * scsi-0QEMU_QEMU_HARDDISK_d1 ONLINE 0 0 0 8333 * scsi-0QEMU_QEMU_HARDDISK_d2 ONLINE 0 0 0 8334 * scsi-0QEMU_QEMU_HARDDISK_d3 ONLINE 0 0 0 8335 * scsi-0QEMU_QEMU_HARDDISK_d4 ONLINE 0 0 0 8336 * scsi-0QEMU_QEMU_HARDDISK_d5 ONLINE 0 0 0 8337 * scsi-0QEMU_QEMU_HARDDISK_d6 ONLINE 0 0 0 8338 * scsi-0QEMU_QEMU_HARDDISK_d7 ONLINE 0 0 0 8339 * scsi-0QEMU_QEMU_HARDDISK_d8 ONLINE 0 0 0 8340 * scsi-0QEMU_QEMU_HARDDISK_d9 ONLINE 0 0 0 8341 * spare-9 DEGRADED 0 0 0 8342 * replacing-0 DEGRADED 0 93 0 8343 * scsi-0QEMU_QEMU_HARDDISK_d10-part1/old UNAVAIL 0 0 0 8344 * spare-1 DEGRADED 0 0 0 8345 * scsi-0QEMU_QEMU_HARDDISK_d10 REMOVED 0 0 0 8346 * draid2-0-0 ONLINE 0 0 0 8347 * draid2-0-1 ONLINE 0 0 0 8348 * spares 8349 * draid2-0-0 INUSE currently in use 8350 * draid2-0-1 INUSE currently in use 8351 * 8352 * ARGS: 8353 * 8354 * newvd: New spare disk 8355 * pvd: Parent vdev_t the spare should attach to 8356 * 8357 * This function returns B_TRUE if adding the new vdev would create a double 8358 * spare condition, B_FALSE otherwise. 8359 */ 8360 static boolean_t 8361 spa_vdev_new_spare_would_cause_double_spares(vdev_t *newvd, vdev_t *pvd) 8362 { 8363 vdev_t *ppvd; 8364 8365 ppvd = pvd->vdev_parent; 8366 if (ppvd == NULL) 8367 return (B_FALSE); 8368 8369 /* 8370 * To determine if this configuration would cause a double spare, we 8371 * look at the vdev_op of the parent vdev, and of the parent's parent 8372 * vdev. We also look at vdev_isspare on the new disk. A double spare 8373 * condition looks like this: 8374 * 8375 * 1. parent of parent's op is a spare or draid spare 8376 * 2. parent's op is replacing 8377 * 3. new disk is a spare 8378 */ 8379 if ((ppvd->vdev_ops == &vdev_spare_ops) || 8380 (ppvd->vdev_ops == &vdev_draid_spare_ops)) 8381 if (pvd->vdev_ops == &vdev_replacing_ops) 8382 if (newvd->vdev_isspare) 8383 return (B_TRUE); 8384 8385 return (B_FALSE); 8386 } 8387 8388 /* 8389 * Attach a device to a vdev specified by its guid. The vdev type can be 8390 * a mirror, a raidz, or a leaf device that is also a top-level (e.g. a 8391 * single device). When the vdev is a single device, a mirror vdev will be 8392 * automatically inserted. 8393 * 8394 * If 'replacing' is specified, the new device is intended to replace the 8395 * existing device; in this case the two devices are made into their own 8396 * mirror using the 'replacing' vdev, which is functionally identical to 8397 * the mirror vdev (it actually reuses all the same ops) but has a few 8398 * extra rules: you can't attach to it after it's been created, and upon 8399 * completion of resilvering, the first disk (the one being replaced) 8400 * is automatically detached. 8401 * 8402 * If 'rebuild' is specified, then sequential reconstruction (a.ka. rebuild) 8403 * should be performed instead of traditional healing reconstruction. From 8404 * an administrators perspective these are both resilver operations. 8405 */ 8406 int 8407 spa_vdev_attach(spa_t *spa, uint64_t guid, nvlist_t *nvroot, int replacing, 8408 int rebuild) 8409 { 8410 uint64_t txg, dtl_max_txg; 8411 vdev_t *rvd = spa->spa_root_vdev; 8412 vdev_t *oldvd, *newvd, *newrootvd, *pvd, *tvd; 8413 vdev_ops_t *pvops; 8414 char *oldvdpath, *newvdpath; 8415 int newvd_isspare = B_FALSE; 8416 int error; 8417 8418 ASSERT(spa_writeable(spa)); 8419 8420 txg = spa_vdev_enter(spa); 8421 8422 oldvd = spa_lookup_by_guid(spa, guid, B_FALSE); 8423 8424 ASSERT(spa_namespace_held()); 8425 if (spa_feature_is_active(spa, SPA_FEATURE_POOL_CHECKPOINT)) { 8426 error = (spa_has_checkpoint(spa)) ? 8427 ZFS_ERR_CHECKPOINT_EXISTS : ZFS_ERR_DISCARDING_CHECKPOINT; 8428 return (spa_vdev_exit(spa, NULL, txg, error)); 8429 } 8430 8431 if (rebuild) { 8432 if (!spa_feature_is_enabled(spa, SPA_FEATURE_DEVICE_REBUILD)) 8433 return (spa_vdev_exit(spa, NULL, txg, ENOTSUP)); 8434 8435 if (dsl_scan_resilvering(spa_get_dsl(spa)) || 8436 dsl_scan_resilver_scheduled(spa_get_dsl(spa))) { 8437 return (spa_vdev_exit(spa, NULL, txg, 8438 ZFS_ERR_RESILVER_IN_PROGRESS)); 8439 } 8440 } else { 8441 if (vdev_rebuild_active(rvd)) 8442 return (spa_vdev_exit(spa, NULL, txg, 8443 ZFS_ERR_REBUILD_IN_PROGRESS)); 8444 } 8445 8446 if (spa->spa_vdev_removal != NULL) { 8447 return (spa_vdev_exit(spa, NULL, txg, 8448 ZFS_ERR_DEVRM_IN_PROGRESS)); 8449 } 8450 8451 if (oldvd == NULL) 8452 return (spa_vdev_exit(spa, NULL, txg, ENODEV)); 8453 8454 boolean_t raidz = oldvd->vdev_ops == &vdev_raidz_ops; 8455 8456 if (raidz) { 8457 if (!spa_feature_is_enabled(spa, SPA_FEATURE_RAIDZ_EXPANSION)) 8458 return (spa_vdev_exit(spa, NULL, txg, ENOTSUP)); 8459 8460 /* 8461 * Can't expand a raidz while prior expand is in progress. 8462 */ 8463 if (spa->spa_raidz_expand != NULL) { 8464 return (spa_vdev_exit(spa, NULL, txg, 8465 ZFS_ERR_RAIDZ_EXPAND_IN_PROGRESS)); 8466 } 8467 } else if (!oldvd->vdev_ops->vdev_op_leaf) { 8468 return (spa_vdev_exit(spa, NULL, txg, ENOTSUP)); 8469 } 8470 8471 if (raidz) 8472 pvd = oldvd; 8473 else 8474 pvd = oldvd->vdev_parent; 8475 8476 if (spa_config_parse(spa, &newrootvd, nvroot, NULL, 0, 8477 VDEV_ALLOC_ATTACH) != 0) 8478 return (spa_vdev_exit(spa, NULL, txg, EINVAL)); 8479 8480 if (newrootvd->vdev_children != 1) 8481 return (spa_vdev_exit(spa, newrootvd, txg, EINVAL)); 8482 8483 newvd = newrootvd->vdev_child[0]; 8484 8485 if (!newvd->vdev_ops->vdev_op_leaf) 8486 return (spa_vdev_exit(spa, newrootvd, txg, EINVAL)); 8487 8488 if ((error = vdev_create(newrootvd, txg, replacing)) != 0) 8489 return (spa_vdev_exit(spa, newrootvd, txg, error)); 8490 8491 /* 8492 * Spares can't replace logs 8493 */ 8494 if (oldvd->vdev_top->vdev_islog && newvd->vdev_isspare) 8495 return (spa_vdev_exit(spa, newrootvd, txg, ENOTSUP)); 8496 8497 /* 8498 * For special and dedup vdevs a spare must have matching rotational 8499 * characteristics. A rotating spare replacing a non-rotating vdev 8500 * would silently degrade pool performance, so we reject the mismatch. 8501 */ 8502 if (newvd->vdev_isspare && 8503 oldvd->vdev_top->vdev_alloc_bias != VDEV_BIAS_NONE && 8504 newvd->vdev_nonrot != oldvd->vdev_nonrot) 8505 return (spa_vdev_exit(spa, newrootvd, txg, ENOTSUP)); 8506 8507 /* 8508 * A dRAID spare can only replace a child of its parent dRAID vdev. 8509 */ 8510 if (newvd->vdev_ops == &vdev_draid_spare_ops && 8511 oldvd->vdev_top != vdev_draid_spare_get_parent(newvd)) { 8512 return (spa_vdev_exit(spa, newrootvd, txg, ENOTSUP)); 8513 } 8514 8515 if (rebuild) { 8516 /* 8517 * For rebuilds, the top vdev must support reconstruction 8518 * using only space maps. This means the only allowable 8519 * vdevs types are the root vdev, a mirror, or dRAID. 8520 */ 8521 tvd = pvd; 8522 if (pvd->vdev_top != NULL) 8523 tvd = pvd->vdev_top; 8524 8525 if (tvd->vdev_ops != &vdev_mirror_ops && 8526 tvd->vdev_ops != &vdev_root_ops && 8527 tvd->vdev_ops != &vdev_draid_ops) { 8528 return (spa_vdev_exit(spa, newrootvd, txg, ENOTSUP)); 8529 } 8530 } 8531 8532 if (!replacing) { 8533 /* 8534 * For attach, the only allowable parent is a mirror or 8535 * the root vdev. A raidz vdev can be attached to, but 8536 * you cannot attach to a raidz child. 8537 */ 8538 if (pvd->vdev_ops != &vdev_mirror_ops && 8539 pvd->vdev_ops != &vdev_root_ops && 8540 !raidz) 8541 return (spa_vdev_exit(spa, newrootvd, txg, ENOTSUP)); 8542 8543 pvops = &vdev_mirror_ops; 8544 } else { 8545 /* 8546 * Active hot spares can only be replaced by inactive hot 8547 * spares. 8548 */ 8549 if (pvd->vdev_ops == &vdev_spare_ops && 8550 oldvd->vdev_isspare && 8551 !spa_has_spare(spa, newvd->vdev_guid)) 8552 return (spa_vdev_exit(spa, newrootvd, txg, ENOTSUP)); 8553 8554 /* 8555 * If the source is a hot spare, and the parent isn't already a 8556 * spare, then we want to create a new hot spare. Otherwise, we 8557 * want to create a replacing vdev. The user is not allowed to 8558 * attach to a spared vdev child unless the 'isspare' state is 8559 * the same (spare replaces spare, non-spare replaces 8560 * non-spare). 8561 */ 8562 if (pvd->vdev_ops == &vdev_replacing_ops && 8563 spa_version(spa) < SPA_VERSION_MULTI_REPLACE) { 8564 return (spa_vdev_exit(spa, newrootvd, txg, ENOTSUP)); 8565 } else if (pvd->vdev_ops == &vdev_spare_ops && 8566 newvd->vdev_isspare != oldvd->vdev_isspare) { 8567 return (spa_vdev_exit(spa, newrootvd, txg, ENOTSUP)); 8568 } 8569 8570 if (spa_vdev_new_spare_would_cause_double_spares(newvd, pvd)) { 8571 vdev_dbgmsg(newvd, 8572 "disk would create double spares, ignore."); 8573 return (spa_vdev_exit(spa, newrootvd, txg, EEXIST)); 8574 } 8575 8576 if (newvd->vdev_isspare) 8577 pvops = &vdev_spare_ops; 8578 else 8579 pvops = &vdev_replacing_ops; 8580 } 8581 8582 /* 8583 * Make sure the new device is big enough. 8584 */ 8585 vdev_t *min_vdev = raidz ? oldvd->vdev_child[0] : oldvd; 8586 if (newvd->vdev_asize < vdev_get_min_asize(min_vdev)) 8587 return (spa_vdev_exit(spa, newrootvd, txg, EOVERFLOW)); 8588 8589 /* 8590 * The new device cannot have a higher alignment requirement 8591 * than the top-level vdev. 8592 */ 8593 if (newvd->vdev_ashift > oldvd->vdev_top->vdev_ashift) { 8594 return (spa_vdev_exit(spa, newrootvd, txg, 8595 ZFS_ERR_ASHIFT_MISMATCH)); 8596 } 8597 8598 /* 8599 * RAIDZ-expansion-specific checks. 8600 */ 8601 if (raidz) { 8602 if (vdev_raidz_attach_check(newvd) != 0) 8603 return (spa_vdev_exit(spa, newrootvd, txg, ENOTSUP)); 8604 8605 /* 8606 * Fail early if a child is not healthy or being replaced 8607 */ 8608 for (int i = 0; i < oldvd->vdev_children; i++) { 8609 if (vdev_is_dead(oldvd->vdev_child[i]) || 8610 !oldvd->vdev_child[i]->vdev_ops->vdev_op_leaf) { 8611 return (spa_vdev_exit(spa, newrootvd, txg, 8612 ENXIO)); 8613 } 8614 /* Also fail if reserved boot area is in-use */ 8615 if (vdev_check_boot_reserve(spa, oldvd->vdev_child[i]) 8616 != 0) { 8617 return (spa_vdev_exit(spa, newrootvd, txg, 8618 EADDRINUSE)); 8619 } 8620 } 8621 } 8622 8623 if (raidz) { 8624 /* 8625 * Note: oldvdpath is freed by spa_strfree(), but 8626 * kmem_asprintf() is freed by kmem_strfree(), so we have to 8627 * move it to a spa_strdup-ed string. 8628 */ 8629 char *tmp = kmem_asprintf("raidz%u-%u", 8630 (uint_t)vdev_get_nparity(oldvd), (uint_t)oldvd->vdev_id); 8631 oldvdpath = spa_strdup(tmp); 8632 kmem_strfree(tmp); 8633 } else { 8634 oldvdpath = spa_strdup(oldvd->vdev_path); 8635 } 8636 newvdpath = spa_strdup(newvd->vdev_path); 8637 8638 /* 8639 * If this is an in-place replacement, update oldvd's path and devid 8640 * to make it distinguishable from newvd, and unopenable from now on. 8641 */ 8642 if (strcmp(oldvdpath, newvdpath) == 0) { 8643 spa_strfree(oldvd->vdev_path); 8644 oldvd->vdev_path = kmem_alloc(strlen(newvdpath) + 5, 8645 KM_SLEEP); 8646 (void) sprintf(oldvd->vdev_path, "%s/old", 8647 newvdpath); 8648 if (oldvd->vdev_devid != NULL) { 8649 spa_strfree(oldvd->vdev_devid); 8650 oldvd->vdev_devid = NULL; 8651 } 8652 spa_strfree(oldvdpath); 8653 oldvdpath = spa_strdup(oldvd->vdev_path); 8654 } 8655 8656 /* 8657 * If the parent is not a mirror, or if we're replacing, insert the new 8658 * mirror/replacing/spare vdev above oldvd. 8659 */ 8660 if (!raidz && pvd->vdev_ops != pvops) { 8661 pvd = vdev_add_parent(oldvd, pvops); 8662 ASSERT(pvd->vdev_ops == pvops); 8663 ASSERT(oldvd->vdev_parent == pvd); 8664 } 8665 8666 ASSERT(pvd->vdev_top->vdev_parent == rvd); 8667 8668 /* 8669 * Extract the new device from its root and add it to pvd. 8670 */ 8671 vdev_remove_child(newrootvd, newvd); 8672 newvd->vdev_id = pvd->vdev_children; 8673 newvd->vdev_crtxg = oldvd->vdev_crtxg; 8674 vdev_add_child(pvd, newvd); 8675 8676 /* 8677 * Reevaluate the parent vdev state. 8678 */ 8679 vdev_propagate_state(pvd); 8680 8681 tvd = newvd->vdev_top; 8682 ASSERT(pvd->vdev_top == tvd); 8683 ASSERT(tvd->vdev_parent == rvd); 8684 8685 vdev_config_dirty(tvd); 8686 8687 /* 8688 * Set newvd's DTL to [TXG_INITIAL, dtl_max_txg) so that we account 8689 * for any dmu_sync-ed blocks. It will propagate upward when 8690 * spa_vdev_exit() calls vdev_dtl_reassess(). 8691 */ 8692 dtl_max_txg = txg + TXG_CONCURRENT_STATES; 8693 8694 if (raidz) { 8695 dmu_tx_t *tx = dmu_tx_create_assigned(spa->spa_dsl_pool, 8696 txg); 8697 dsl_sync_task_nowait(spa->spa_dsl_pool, vdev_raidz_attach_sync, 8698 newvd, tx); 8699 dmu_tx_commit(tx); 8700 8701 /* 8702 * Wait for the youngest allocations and frees to sync, 8703 * and then wait for the deferral of those frees to finish. 8704 */ 8705 spa_vdev_config_exit(spa, NULL, 8706 txg + TXG_CONCURRENT_STATES + TXG_DEFER_SIZE, 0, FTAG); 8707 8708 vdev_initialize_stop_all(tvd, VDEV_INITIALIZE_ACTIVE); 8709 vdev_trim_stop_all(tvd, VDEV_TRIM_ACTIVE); 8710 vdev_autotrim_stop_wait(tvd); 8711 8712 dtl_max_txg = spa_vdev_config_enter(spa); 8713 8714 tvd->vdev_rz_expanding = B_TRUE; 8715 8716 vdev_dirty_leaves(tvd, VDD_DTL, dtl_max_txg); 8717 vdev_config_dirty(tvd); 8718 zthr_wakeup(spa->spa_raidz_expand_zthr); 8719 } else { 8720 vdev_dtl_dirty(newvd, DTL_MISSING, TXG_INITIAL, 8721 dtl_max_txg - TXG_INITIAL); 8722 8723 if (newvd->vdev_isspare) { 8724 spa_spare_activate(newvd); 8725 spa_event_notify(spa, newvd, NULL, ESC_ZFS_VDEV_SPARE); 8726 } 8727 8728 newvd_isspare = newvd->vdev_isspare; 8729 8730 /* 8731 * Mark newvd's DTL dirty in this txg. 8732 */ 8733 vdev_dirty(tvd, VDD_DTL, newvd, txg); 8734 8735 /* 8736 * Schedule the resilver or rebuild to restart in the future. 8737 * We do this to ensure that dmu_sync-ed blocks have been 8738 * stitched into the respective datasets. 8739 */ 8740 if (rebuild) { 8741 newvd->vdev_rebuild_txg = txg; 8742 8743 vdev_rebuild(tvd, txg); 8744 } else { 8745 newvd->vdev_resilver_txg = txg; 8746 8747 if (dsl_scan_resilvering(spa_get_dsl(spa)) && 8748 spa_feature_is_enabled(spa, 8749 SPA_FEATURE_RESILVER_DEFER)) { 8750 vdev_defer_resilver(newvd); 8751 } else { 8752 dsl_scan_restart_resilver(spa->spa_dsl_pool, 8753 dtl_max_txg); 8754 } 8755 } 8756 } 8757 8758 if (spa->spa_bootfs) 8759 spa_event_notify(spa, newvd, NULL, ESC_ZFS_BOOTFS_VDEV_ATTACH); 8760 8761 spa_event_notify(spa, newvd, NULL, ESC_ZFS_VDEV_ATTACH); 8762 8763 /* 8764 * Commit the config 8765 */ 8766 (void) spa_vdev_exit(spa, newrootvd, dtl_max_txg, 0); 8767 8768 spa_history_log_internal(spa, "vdev attach", NULL, 8769 "%s vdev=%s %s vdev=%s", 8770 replacing && newvd_isspare ? "spare in" : 8771 replacing ? "replace" : "attach", newvdpath, 8772 replacing ? "for" : "to", oldvdpath); 8773 8774 spa_strfree(oldvdpath); 8775 spa_strfree(newvdpath); 8776 8777 return (0); 8778 } 8779 8780 /* 8781 * Detach a device from a mirror or replacing vdev. 8782 * 8783 * If 'replace_done' is specified, only detach if the parent 8784 * is a replacing or a spare vdev. 8785 */ 8786 int 8787 spa_vdev_detach(spa_t *spa, uint64_t guid, uint64_t pguid, int replace_done) 8788 { 8789 uint64_t txg; 8790 int error; 8791 vdev_t *rvd __maybe_unused = spa->spa_root_vdev; 8792 vdev_t *vd, *pvd, *cvd, *tvd; 8793 boolean_t unspare = B_FALSE; 8794 uint64_t unspare_guid = 0; 8795 char *vdpath; 8796 8797 ASSERT(spa_writeable(spa)); 8798 8799 txg = spa_vdev_detach_enter(spa, guid); 8800 8801 vd = spa_lookup_by_guid(spa, guid, B_FALSE); 8802 8803 /* 8804 * Besides being called directly from the userland through the 8805 * ioctl interface, spa_vdev_detach() can be potentially called 8806 * at the end of spa_vdev_resilver_done(). 8807 * 8808 * In the regular case, when we have a checkpoint this shouldn't 8809 * happen as we never empty the DTLs of a vdev during the scrub 8810 * [see comment in dsl_scan_done()]. Thus spa_vdev_resilvering_done() 8811 * should never get here when we have a checkpoint. 8812 * 8813 * That said, even in a case when we checkpoint the pool exactly 8814 * as spa_vdev_resilver_done() calls this function everything 8815 * should be fine as the resilver will return right away. 8816 */ 8817 ASSERT(spa_namespace_held()); 8818 if (spa_feature_is_active(spa, SPA_FEATURE_POOL_CHECKPOINT)) { 8819 error = (spa_has_checkpoint(spa)) ? 8820 ZFS_ERR_CHECKPOINT_EXISTS : ZFS_ERR_DISCARDING_CHECKPOINT; 8821 return (spa_vdev_exit(spa, NULL, txg, error)); 8822 } 8823 8824 if (vd == NULL) 8825 return (spa_vdev_exit(spa, NULL, txg, ENODEV)); 8826 8827 if (!vd->vdev_ops->vdev_op_leaf) 8828 return (spa_vdev_exit(spa, NULL, txg, ENOTSUP)); 8829 8830 pvd = vd->vdev_parent; 8831 8832 /* 8833 * If the parent/child relationship is not as expected, don't do it. 8834 * Consider M(A,R(B,C)) -- that is, a mirror of A with a replacing 8835 * vdev that's replacing B with C. The user's intent in replacing 8836 * is to go from M(A,B) to M(A,C). If the user decides to cancel 8837 * the replace by detaching C, the expected behavior is to end up 8838 * M(A,B). But suppose that right after deciding to detach C, 8839 * the replacement of B completes. We would have M(A,C), and then 8840 * ask to detach C, which would leave us with just A -- not what 8841 * the user wanted. To prevent this, we make sure that the 8842 * parent/child relationship hasn't changed -- in this example, 8843 * that C's parent is still the replacing vdev R. 8844 */ 8845 if (pvd->vdev_guid != pguid && pguid != 0) 8846 return (spa_vdev_exit(spa, NULL, txg, EBUSY)); 8847 8848 /* 8849 * Only 'replacing' or 'spare' vdevs can be replaced. 8850 */ 8851 if (replace_done && pvd->vdev_ops != &vdev_replacing_ops && 8852 pvd->vdev_ops != &vdev_spare_ops) 8853 return (spa_vdev_exit(spa, NULL, txg, ENOTSUP)); 8854 8855 ASSERT(pvd->vdev_ops != &vdev_spare_ops || 8856 spa_version(spa) >= SPA_VERSION_SPARES); 8857 8858 /* 8859 * Only mirror, replacing, and spare vdevs support detach. 8860 */ 8861 if (pvd->vdev_ops != &vdev_replacing_ops && 8862 pvd->vdev_ops != &vdev_mirror_ops && 8863 pvd->vdev_ops != &vdev_spare_ops) 8864 return (spa_vdev_exit(spa, NULL, txg, ENOTSUP)); 8865 8866 /* 8867 * If this device has the only valid copy of some data, 8868 * we cannot safely detach it. 8869 */ 8870 if (vdev_dtl_required(vd)) 8871 return (spa_vdev_exit(spa, NULL, txg, EBUSY)); 8872 8873 ASSERT(pvd->vdev_children >= 2); 8874 8875 /* 8876 * If we are detaching the second disk from a replacing vdev, then 8877 * check to see if we changed the original vdev's path to have "/old" 8878 * at the end in spa_vdev_attach(). If so, undo that change now. 8879 */ 8880 if (pvd->vdev_ops == &vdev_replacing_ops && vd->vdev_id > 0 && 8881 vd->vdev_path != NULL) { 8882 size_t len = strlen(vd->vdev_path); 8883 8884 for (int c = 0; c < pvd->vdev_children; c++) { 8885 cvd = pvd->vdev_child[c]; 8886 8887 if (cvd == vd || cvd->vdev_path == NULL) 8888 continue; 8889 8890 if (strncmp(cvd->vdev_path, vd->vdev_path, len) == 0 && 8891 strcmp(cvd->vdev_path + len, "/old") == 0) { 8892 spa_strfree(cvd->vdev_path); 8893 cvd->vdev_path = spa_strdup(vd->vdev_path); 8894 break; 8895 } 8896 } 8897 } 8898 8899 /* 8900 * If we are detaching the original disk from a normal spare, then it 8901 * implies that the spare should become a real disk, and be removed 8902 * from the active spare list for the pool. dRAID spares on the 8903 * other hand are coupled to the pool and thus should never be removed 8904 * from the spares list. 8905 */ 8906 if (pvd->vdev_ops == &vdev_spare_ops && vd->vdev_id == 0) { 8907 vdev_t *last_cvd = pvd->vdev_child[pvd->vdev_children - 1]; 8908 8909 if (last_cvd->vdev_isspare && 8910 last_cvd->vdev_ops != &vdev_draid_spare_ops) { 8911 unspare = B_TRUE; 8912 } 8913 } 8914 8915 /* 8916 * Erase the disk labels so the disk can be used for other things. 8917 * This must be done after all other error cases are handled, 8918 * but before we disembowel vd (so we can still do I/O to it). 8919 * But if we can't do it, don't treat the error as fatal -- 8920 * it may be that the unwritability of the disk is the reason 8921 * it's being detached! 8922 */ 8923 (void) vdev_label_init(vd, 0, VDEV_LABEL_REMOVE); 8924 8925 /* 8926 * Remove vd from its parent and compact the parent's children. 8927 */ 8928 vdev_remove_child(pvd, vd); 8929 vdev_compact_children(pvd); 8930 8931 /* 8932 * Remember one of the remaining children so we can get tvd below. 8933 */ 8934 cvd = pvd->vdev_child[pvd->vdev_children - 1]; 8935 8936 /* 8937 * If we need to remove the remaining child from the list of hot spares, 8938 * do it now, marking the vdev as no longer a spare in the process. 8939 * We must do this before vdev_remove_parent(), because that can 8940 * change the GUID if it creates a new toplevel GUID. For a similar 8941 * reason, we must remove the spare now, in the same txg as the detach; 8942 * otherwise someone could attach a new sibling, change the GUID, and 8943 * the subsequent attempt to spa_vdev_remove(unspare_guid) would fail. 8944 */ 8945 if (unspare) { 8946 ASSERT(cvd->vdev_isspare); 8947 spa_spare_remove(cvd); 8948 unspare_guid = cvd->vdev_guid; 8949 (void) spa_vdev_remove(spa, unspare_guid, B_TRUE); 8950 cvd->vdev_unspare = B_TRUE; 8951 } 8952 8953 /* 8954 * If the parent mirror/replacing vdev only has one child, 8955 * the parent is no longer needed. Remove it from the tree. 8956 */ 8957 if (pvd->vdev_children == 1) { 8958 if (pvd->vdev_ops == &vdev_spare_ops) 8959 cvd->vdev_unspare = B_FALSE; 8960 vdev_remove_parent(cvd); 8961 } 8962 8963 /* 8964 * We don't set tvd until now because the parent we just removed 8965 * may have been the previous top-level vdev. 8966 */ 8967 tvd = cvd->vdev_top; 8968 ASSERT(tvd->vdev_parent == rvd); 8969 8970 /* 8971 * Reevaluate the parent vdev state. 8972 */ 8973 vdev_propagate_state(cvd); 8974 8975 /* 8976 * If the 'autoexpand' property is set on the pool then automatically 8977 * try to expand the size of the pool. For example if the device we 8978 * just detached was smaller than the others, it may be possible to 8979 * add metaslabs (i.e. grow the pool). We need to reopen the vdev 8980 * first so that we can obtain the updated sizes of the leaf vdevs. 8981 */ 8982 if (spa->spa_autoexpand) { 8983 vdev_reopen(tvd); 8984 vdev_expand(tvd, txg); 8985 } 8986 8987 vdev_config_dirty(tvd); 8988 8989 /* 8990 * Mark vd's DTL as dirty in this txg. vdev_dtl_sync() will see that 8991 * vd->vdev_detached is set and free vd's DTL object in syncing context. 8992 * But first make sure we're not on any *other* txg's DTL list, to 8993 * prevent vd from being accessed after it's freed. 8994 */ 8995 vdpath = spa_strdup(vd->vdev_path ? vd->vdev_path : "none"); 8996 for (int t = 0; t < TXG_SIZE; t++) 8997 (void) txg_list_remove_this(&tvd->vdev_dtl_list, vd, t); 8998 vd->vdev_detached = B_TRUE; 8999 vdev_dirty(tvd, VDD_DTL, vd, txg); 9000 9001 spa_event_notify(spa, vd, NULL, ESC_ZFS_VDEV_REMOVE); 9002 spa_notify_waiters(spa); 9003 9004 /* hang on to the spa before we release the lock */ 9005 spa_open_ref(spa, FTAG); 9006 9007 error = spa_vdev_exit(spa, vd, txg, 0); 9008 9009 spa_history_log_internal(spa, "detach", NULL, 9010 "vdev=%s", vdpath); 9011 spa_strfree(vdpath); 9012 9013 /* 9014 * If this was the removal of the original device in a hot spare vdev, 9015 * then we want to go through and remove the device from the hot spare 9016 * list of every other pool. 9017 */ 9018 if (unspare) { 9019 spa_t *altspa = NULL; 9020 9021 spa_namespace_enter(FTAG); 9022 while ((altspa = spa_next(altspa)) != NULL) { 9023 if (altspa->spa_state != POOL_STATE_ACTIVE || 9024 altspa == spa) 9025 continue; 9026 9027 spa_open_ref(altspa, FTAG); 9028 spa_namespace_exit(FTAG); 9029 (void) spa_vdev_remove(altspa, unspare_guid, B_TRUE); 9030 spa_namespace_enter(FTAG); 9031 spa_close(altspa, FTAG); 9032 } 9033 spa_namespace_exit(FTAG); 9034 9035 /* search the rest of the vdevs for spares to remove */ 9036 spa_vdev_resilver_done(spa); 9037 } 9038 9039 /* all done with the spa; OK to release */ 9040 spa_namespace_enter(FTAG); 9041 spa_close(spa, FTAG); 9042 spa_namespace_exit(FTAG); 9043 9044 return (error); 9045 } 9046 9047 static int 9048 spa_vdev_initialize_impl(spa_t *spa, uint64_t guid, uint64_t cmd_type, 9049 uint64_t value, boolean_t value_provided, list_t *vd_list) 9050 { 9051 ASSERT(spa_namespace_held()); 9052 9053 spa_config_enter(spa, SCL_CONFIG | SCL_STATE, FTAG, RW_READER); 9054 9055 /* Look up vdev and ensure it's a leaf. */ 9056 vdev_t *vd = spa_lookup_by_guid(spa, guid, B_FALSE); 9057 if (vd == NULL || vd->vdev_detached) { 9058 spa_config_exit(spa, SCL_CONFIG | SCL_STATE, FTAG); 9059 return (SET_ERROR(ENODEV)); 9060 } else if (!vd->vdev_ops->vdev_op_leaf || !vdev_is_concrete(vd)) { 9061 spa_config_exit(spa, SCL_CONFIG | SCL_STATE, FTAG); 9062 return (SET_ERROR(EINVAL)); 9063 } else if (!vdev_writeable(vd)) { 9064 spa_config_exit(spa, SCL_CONFIG | SCL_STATE, FTAG); 9065 return (SET_ERROR(EROFS)); 9066 } 9067 mutex_enter(&vd->vdev_initialize_lock); 9068 spa_config_exit(spa, SCL_CONFIG | SCL_STATE, FTAG); 9069 9070 /* 9071 * When we activate an initialize action we check to see 9072 * if the vdev_initialize_thread is NULL. We do this instead 9073 * of using the vdev_initialize_state since there might be 9074 * a previous initialization process which has completed but 9075 * the thread is not exited. 9076 */ 9077 if (cmd_type == POOL_INITIALIZE_START && 9078 (vd->vdev_initialize_thread != NULL || 9079 vd->vdev_top->vdev_removing || vd->vdev_top->vdev_rz_expanding)) { 9080 mutex_exit(&vd->vdev_initialize_lock); 9081 return (SET_ERROR(EBUSY)); 9082 } else if (cmd_type == POOL_INITIALIZE_CANCEL && 9083 (vd->vdev_initialize_state != VDEV_INITIALIZE_ACTIVE && 9084 vd->vdev_initialize_state != VDEV_INITIALIZE_SUSPENDED)) { 9085 mutex_exit(&vd->vdev_initialize_lock); 9086 return (SET_ERROR(ESRCH)); 9087 } else if (cmd_type == POOL_INITIALIZE_SUSPEND && 9088 vd->vdev_initialize_state != VDEV_INITIALIZE_ACTIVE) { 9089 mutex_exit(&vd->vdev_initialize_lock); 9090 return (SET_ERROR(ESRCH)); 9091 } else if (cmd_type == POOL_INITIALIZE_UNINIT && 9092 vd->vdev_initialize_thread != NULL) { 9093 mutex_exit(&vd->vdev_initialize_lock); 9094 return (SET_ERROR(EBUSY)); 9095 } 9096 9097 switch (cmd_type) { 9098 case POOL_INITIALIZE_START: 9099 vdev_initialize(vd, value, value_provided); 9100 break; 9101 case POOL_INITIALIZE_CANCEL: 9102 vdev_initialize_stop(vd, VDEV_INITIALIZE_CANCELED, vd_list); 9103 break; 9104 case POOL_INITIALIZE_SUSPEND: 9105 vdev_initialize_stop(vd, VDEV_INITIALIZE_SUSPENDED, vd_list); 9106 break; 9107 case POOL_INITIALIZE_UNINIT: 9108 vdev_uninitialize(vd); 9109 break; 9110 default: 9111 panic("invalid cmd_type %llu", (unsigned long long)cmd_type); 9112 } 9113 mutex_exit(&vd->vdev_initialize_lock); 9114 9115 return (0); 9116 } 9117 9118 int 9119 spa_vdev_initialize(spa_t *spa, nvlist_t *nv, uint64_t cmd_type, 9120 uint64_t value, boolean_t value_provided, nvlist_t *vdev_errlist) 9121 { 9122 int total_errors = 0; 9123 list_t vd_list; 9124 9125 list_create(&vd_list, sizeof (vdev_t), 9126 offsetof(vdev_t, vdev_initialize_node)); 9127 9128 /* 9129 * We hold the namespace lock through the whole function 9130 * to prevent any changes to the pool while we're starting or 9131 * stopping initialization. The config and state locks are held so that 9132 * we can properly assess the vdev state before we commit to 9133 * the initializing operation. 9134 */ 9135 spa_namespace_enter(FTAG); 9136 9137 for (nvpair_t *pair = nvlist_next_nvpair(nv, NULL); 9138 pair != NULL; pair = nvlist_next_nvpair(nv, pair)) { 9139 uint64_t vdev_guid = fnvpair_value_uint64(pair); 9140 9141 int error = spa_vdev_initialize_impl(spa, vdev_guid, cmd_type, 9142 value, value_provided, &vd_list); 9143 if (error != 0) { 9144 char guid_as_str[MAXNAMELEN]; 9145 9146 (void) snprintf(guid_as_str, sizeof (guid_as_str), 9147 "%llu", (unsigned long long)vdev_guid); 9148 fnvlist_add_int64(vdev_errlist, guid_as_str, error); 9149 total_errors++; 9150 } 9151 } 9152 9153 /* Wait for all initialize threads to stop. */ 9154 vdev_initialize_stop_wait(spa, &vd_list); 9155 9156 /* Sync out the initializing state */ 9157 txg_wait_synced(spa->spa_dsl_pool, 0); 9158 spa_namespace_exit(FTAG); 9159 9160 list_destroy(&vd_list); 9161 9162 return (total_errors); 9163 } 9164 9165 static int 9166 spa_vdev_trim_impl(spa_t *spa, uint64_t guid, uint64_t cmd_type, 9167 uint64_t rate, boolean_t partial, boolean_t secure, list_t *vd_list) 9168 { 9169 ASSERT(spa_namespace_held()); 9170 9171 spa_config_enter(spa, SCL_CONFIG | SCL_STATE, FTAG, RW_READER); 9172 9173 /* Look up vdev and ensure it's a leaf. */ 9174 vdev_t *vd = spa_lookup_by_guid(spa, guid, B_FALSE); 9175 if (vd == NULL || vd->vdev_detached) { 9176 spa_config_exit(spa, SCL_CONFIG | SCL_STATE, FTAG); 9177 return (SET_ERROR(ENODEV)); 9178 } else if (!vd->vdev_ops->vdev_op_leaf || !vdev_is_concrete(vd)) { 9179 spa_config_exit(spa, SCL_CONFIG | SCL_STATE, FTAG); 9180 return (SET_ERROR(EINVAL)); 9181 } else if (!vdev_writeable(vd)) { 9182 spa_config_exit(spa, SCL_CONFIG | SCL_STATE, FTAG); 9183 return (SET_ERROR(EROFS)); 9184 } else if (!vd->vdev_has_trim) { 9185 spa_config_exit(spa, SCL_CONFIG | SCL_STATE, FTAG); 9186 return (SET_ERROR(EOPNOTSUPP)); 9187 } else if (secure && !vd->vdev_has_securetrim) { 9188 spa_config_exit(spa, SCL_CONFIG | SCL_STATE, FTAG); 9189 return (SET_ERROR(EOPNOTSUPP)); 9190 } 9191 mutex_enter(&vd->vdev_trim_lock); 9192 spa_config_exit(spa, SCL_CONFIG | SCL_STATE, FTAG); 9193 9194 /* 9195 * When we activate a TRIM action we check to see if the 9196 * vdev_trim_thread is NULL. We do this instead of using the 9197 * vdev_trim_state since there might be a previous TRIM process 9198 * which has completed but the thread is not exited. 9199 */ 9200 if (cmd_type == POOL_TRIM_START && 9201 (vd->vdev_trim_thread != NULL || vd->vdev_top->vdev_removing || 9202 vd->vdev_top->vdev_rz_expanding)) { 9203 mutex_exit(&vd->vdev_trim_lock); 9204 return (SET_ERROR(EBUSY)); 9205 } else if (cmd_type == POOL_TRIM_CANCEL && 9206 (vd->vdev_trim_state != VDEV_TRIM_ACTIVE && 9207 vd->vdev_trim_state != VDEV_TRIM_SUSPENDED)) { 9208 mutex_exit(&vd->vdev_trim_lock); 9209 return (SET_ERROR(ESRCH)); 9210 } else if (cmd_type == POOL_TRIM_SUSPEND && 9211 vd->vdev_trim_state != VDEV_TRIM_ACTIVE) { 9212 mutex_exit(&vd->vdev_trim_lock); 9213 return (SET_ERROR(ESRCH)); 9214 } 9215 9216 switch (cmd_type) { 9217 case POOL_TRIM_START: 9218 vdev_trim(vd, rate, partial, secure); 9219 break; 9220 case POOL_TRIM_CANCEL: 9221 vdev_trim_stop(vd, VDEV_TRIM_CANCELED, vd_list); 9222 break; 9223 case POOL_TRIM_SUSPEND: 9224 vdev_trim_stop(vd, VDEV_TRIM_SUSPENDED, vd_list); 9225 break; 9226 default: 9227 panic("invalid cmd_type %llu", (unsigned long long)cmd_type); 9228 } 9229 mutex_exit(&vd->vdev_trim_lock); 9230 9231 return (0); 9232 } 9233 9234 /* 9235 * Initiates a manual TRIM for the requested vdevs. This kicks off individual 9236 * TRIM threads for each child vdev. These threads pass over all of the free 9237 * space in the vdev's metaslabs and issues TRIM commands for that space. 9238 */ 9239 int 9240 spa_vdev_trim(spa_t *spa, nvlist_t *nv, uint64_t cmd_type, uint64_t rate, 9241 boolean_t partial, boolean_t secure, nvlist_t *vdev_errlist) 9242 { 9243 int total_errors = 0; 9244 list_t vd_list; 9245 9246 list_create(&vd_list, sizeof (vdev_t), 9247 offsetof(vdev_t, vdev_trim_node)); 9248 9249 /* 9250 * We hold the namespace lock through the whole function 9251 * to prevent any changes to the pool while we're starting or 9252 * stopping TRIM. The config and state locks are held so that 9253 * we can properly assess the vdev state before we commit to 9254 * the TRIM operation. 9255 */ 9256 spa_namespace_enter(FTAG); 9257 9258 for (nvpair_t *pair = nvlist_next_nvpair(nv, NULL); 9259 pair != NULL; pair = nvlist_next_nvpair(nv, pair)) { 9260 uint64_t vdev_guid = fnvpair_value_uint64(pair); 9261 9262 int error = spa_vdev_trim_impl(spa, vdev_guid, cmd_type, 9263 rate, partial, secure, &vd_list); 9264 if (error != 0) { 9265 char guid_as_str[MAXNAMELEN]; 9266 9267 (void) snprintf(guid_as_str, sizeof (guid_as_str), 9268 "%llu", (unsigned long long)vdev_guid); 9269 fnvlist_add_int64(vdev_errlist, guid_as_str, error); 9270 total_errors++; 9271 } 9272 } 9273 9274 /* Wait for all TRIM threads to stop. */ 9275 vdev_trim_stop_wait(spa, &vd_list); 9276 9277 /* Sync out the TRIM state */ 9278 txg_wait_synced(spa->spa_dsl_pool, 0); 9279 spa_namespace_exit(FTAG); 9280 9281 list_destroy(&vd_list); 9282 9283 return (total_errors); 9284 } 9285 9286 typedef struct spa_split_dtl_arg { 9287 spa_t *ssda_spa; /* the new pool */ 9288 uint64_t *ssda_objs; /* original DTL space map objects */ 9289 uint_t ssda_count; /* nitems in ssda_objs */ 9290 } spa_split_dtl_arg_t; 9291 9292 /* 9293 * Record the DTL space map object of every leaf that has one into objs[], 9294 * advancing *idxp. These are the objects that will be carried, via the 9295 * copied MOS, onto the split disks. 9296 */ 9297 static void 9298 spa_split_collect_dtl(vdev_t *vd, uint64_t *objs, uint_t *idxp) 9299 { 9300 if (vd->vdev_ops->vdev_op_leaf) { 9301 if (vd->vdev_dtl_sm != NULL) 9302 objs[(*idxp)++] = space_map_object(vd->vdev_dtl_sm); 9303 return; 9304 } 9305 for (uint64_t c = 0; c < vd->vdev_children; c++) 9306 spa_split_collect_dtl(vd->vdev_child[c], objs, idxp); 9307 } 9308 9309 /* 9310 * Callback that frees the inherited DTL space map objects from the new 9311 * pool MOS. The new pool MOS is a byte copy of the original pool, so it 9312 * contains a DTL space map object for every leaf of the original pool. 9313 * The new pool references none of them because split leaves 9314 * start with an empty DTL and allocate their own on demand. 9315 */ 9316 static void 9317 spa_split_dtl_free_sync(void *arg, dmu_tx_t *tx) 9318 { 9319 spa_split_dtl_arg_t *ssda = arg; 9320 objset_t *mos = ssda->ssda_spa->spa_meta_objset; 9321 9322 for (uint_t i = 0; i < ssda->ssda_count; i++) 9323 space_map_free_obj(mos, ssda->ssda_objs[i], tx); 9324 } 9325 9326 /* 9327 * Split a set of devices from their mirrors, and create a new pool from them. 9328 */ 9329 int 9330 spa_vdev_split_mirror(spa_t *spa, const char *newname, nvlist_t *config, 9331 nvlist_t *props, boolean_t exp) 9332 { 9333 int error = 0; 9334 uint64_t txg, *glist; 9335 spa_t *newspa; 9336 uint_t c, children, lastlog; 9337 nvlist_t **child, *nvl, *tmp; 9338 dmu_tx_t *tx; 9339 const char *altroot = NULL; 9340 vdev_t *rvd, **vml = NULL; /* vdev modify list */ 9341 uint64_t *dtl_objs = NULL; /* DTL objs from original pool */ 9342 uint_t ndtl = 0, nleaves; 9343 boolean_t activate_slog; 9344 9345 ASSERT(spa_writeable(spa)); 9346 9347 txg = spa_vdev_enter(spa); 9348 9349 ASSERT(spa_namespace_held()); 9350 if (spa_feature_is_active(spa, SPA_FEATURE_POOL_CHECKPOINT)) { 9351 error = (spa_has_checkpoint(spa)) ? 9352 ZFS_ERR_CHECKPOINT_EXISTS : ZFS_ERR_DISCARDING_CHECKPOINT; 9353 return (spa_vdev_exit(spa, NULL, txg, error)); 9354 } 9355 9356 /* clear the log and flush everything up to now */ 9357 activate_slog = spa_passivate_log(spa); 9358 (void) spa_vdev_config_exit(spa, NULL, txg, 0, FTAG); 9359 error = spa_reset_logs(spa); 9360 txg = spa_vdev_config_enter(spa); 9361 9362 if (activate_slog) 9363 spa_activate_log(spa); 9364 9365 if (error != 0) 9366 return (spa_vdev_exit(spa, NULL, txg, error)); 9367 9368 /* check new spa name before going any further */ 9369 if (spa_lookup(newname) != NULL) 9370 return (spa_vdev_exit(spa, NULL, txg, EEXIST)); 9371 9372 /* 9373 * scan through all the children to ensure they're all mirrors 9374 */ 9375 if (nvlist_lookup_nvlist(config, ZPOOL_CONFIG_VDEV_TREE, &nvl) != 0 || 9376 nvlist_lookup_nvlist_array(nvl, ZPOOL_CONFIG_CHILDREN, &child, 9377 &children) != 0) 9378 return (spa_vdev_exit(spa, NULL, txg, EINVAL)); 9379 9380 /* first, check to ensure we've got the right child count */ 9381 rvd = spa->spa_root_vdev; 9382 lastlog = 0; 9383 for (c = 0; c < rvd->vdev_children; c++) { 9384 vdev_t *vd = rvd->vdev_child[c]; 9385 9386 /* don't count the holes & logs as children */ 9387 if (vd->vdev_islog || (vd->vdev_ops != &vdev_indirect_ops && 9388 !vdev_is_concrete(vd))) { 9389 if (lastlog == 0) 9390 lastlog = c; 9391 continue; 9392 } 9393 9394 lastlog = 0; 9395 } 9396 if (children != (lastlog != 0 ? lastlog : rvd->vdev_children)) 9397 return (spa_vdev_exit(spa, NULL, txg, EINVAL)); 9398 9399 /* next, ensure no spare or cache devices are part of the split */ 9400 if (nvlist_lookup_nvlist(nvl, ZPOOL_CONFIG_SPARES, &tmp) == 0 || 9401 nvlist_lookup_nvlist(nvl, ZPOOL_CONFIG_L2CACHE, &tmp) == 0) 9402 return (spa_vdev_exit(spa, NULL, txg, EINVAL)); 9403 9404 vml = kmem_zalloc(children * sizeof (vdev_t *), KM_SLEEP); 9405 glist = kmem_zalloc(children * sizeof (uint64_t), KM_SLEEP); 9406 9407 /* then, loop over each vdev and validate it */ 9408 for (c = 0; c < children; c++) { 9409 uint64_t is_hole = 0; 9410 9411 (void) nvlist_lookup_uint64(child[c], ZPOOL_CONFIG_IS_HOLE, 9412 &is_hole); 9413 9414 if (is_hole != 0) { 9415 if (spa->spa_root_vdev->vdev_child[c]->vdev_ishole || 9416 spa->spa_root_vdev->vdev_child[c]->vdev_islog) { 9417 continue; 9418 } else { 9419 error = SET_ERROR(EINVAL); 9420 break; 9421 } 9422 } 9423 9424 /* deal with indirect vdevs */ 9425 if (spa->spa_root_vdev->vdev_child[c]->vdev_ops == 9426 &vdev_indirect_ops) 9427 continue; 9428 9429 /* which disk is going to be split? */ 9430 if (nvlist_lookup_uint64(child[c], ZPOOL_CONFIG_GUID, 9431 &glist[c]) != 0) { 9432 error = SET_ERROR(EINVAL); 9433 break; 9434 } 9435 9436 /* look it up in the spa */ 9437 vml[c] = spa_lookup_by_guid(spa, glist[c], B_FALSE); 9438 if (vml[c] == NULL) { 9439 error = SET_ERROR(ENODEV); 9440 break; 9441 } 9442 9443 /* make sure there's nothing stopping the split */ 9444 if (vml[c]->vdev_parent->vdev_ops != &vdev_mirror_ops || 9445 vml[c]->vdev_islog || 9446 !vdev_is_concrete(vml[c]) || 9447 vml[c]->vdev_isspare || 9448 vml[c]->vdev_isl2cache || 9449 !vdev_writeable(vml[c]) || 9450 vml[c]->vdev_children != 0 || 9451 vml[c]->vdev_state != VDEV_STATE_HEALTHY || 9452 c != spa->spa_root_vdev->vdev_child[c]->vdev_id) { 9453 error = SET_ERROR(EINVAL); 9454 break; 9455 } 9456 9457 if (vdev_dtl_required(vml[c]) || 9458 vdev_resilver_needed(vml[c], NULL, NULL)) { 9459 error = SET_ERROR(EBUSY); 9460 break; 9461 } 9462 9463 /* we need certain info from the top level */ 9464 fnvlist_add_uint64(child[c], ZPOOL_CONFIG_METASLAB_ARRAY, 9465 vml[c]->vdev_top->vdev_ms_array); 9466 fnvlist_add_uint64(child[c], ZPOOL_CONFIG_METASLAB_SHIFT, 9467 vml[c]->vdev_top->vdev_ms_shift); 9468 fnvlist_add_uint64(child[c], ZPOOL_CONFIG_ASIZE, 9469 vml[c]->vdev_top->vdev_asize); 9470 fnvlist_add_uint64(child[c], ZPOOL_CONFIG_ASHIFT, 9471 vml[c]->vdev_top->vdev_ashift); 9472 9473 /* transfer per-vdev ZAPs */ 9474 ASSERT3U(vml[c]->vdev_leaf_zap, !=, 0); 9475 VERIFY0(nvlist_add_uint64(child[c], 9476 ZPOOL_CONFIG_VDEV_LEAF_ZAP, vml[c]->vdev_leaf_zap)); 9477 9478 ASSERT3U(vml[c]->vdev_top->vdev_top_zap, !=, 0); 9479 VERIFY0(nvlist_add_uint64(child[c], 9480 ZPOOL_CONFIG_VDEV_TOP_ZAP, 9481 vml[c]->vdev_parent->vdev_top_zap)); 9482 } 9483 9484 if (error != 0) { 9485 kmem_free(vml, children * sizeof (vdev_t *)); 9486 kmem_free(glist, children * sizeof (uint64_t)); 9487 return (spa_vdev_exit(spa, NULL, txg, error)); 9488 } 9489 9490 /* Create array of DTL objects. */ 9491 nleaves = vdev_count_leaves(spa); 9492 dtl_objs = kmem_zalloc(nleaves * sizeof (uint64_t), KM_SLEEP); 9493 spa_split_collect_dtl(spa->spa_root_vdev, dtl_objs, &ndtl); 9494 9495 /* stop writers from using the disks */ 9496 for (c = 0; c < children; c++) { 9497 if (vml[c] != NULL) 9498 vml[c]->vdev_offline = B_TRUE; 9499 } 9500 vdev_reopen(spa->spa_root_vdev); 9501 9502 /* 9503 * Temporarily record the splitting vdevs in the spa config. This 9504 * will disappear once the config is regenerated. 9505 */ 9506 nvl = fnvlist_alloc(); 9507 fnvlist_add_uint64_array(nvl, ZPOOL_CONFIG_SPLIT_LIST, glist, children); 9508 kmem_free(glist, children * sizeof (uint64_t)); 9509 9510 mutex_enter(&spa->spa_props_lock); 9511 fnvlist_add_nvlist(spa->spa_config, ZPOOL_CONFIG_SPLIT, nvl); 9512 mutex_exit(&spa->spa_props_lock); 9513 spa->spa_config_splitting = nvl; 9514 vdev_config_dirty(spa->spa_root_vdev); 9515 9516 /* configure and create the new pool */ 9517 fnvlist_add_string(config, ZPOOL_CONFIG_POOL_NAME, newname); 9518 fnvlist_add_uint64(config, ZPOOL_CONFIG_POOL_STATE, 9519 exp ? POOL_STATE_EXPORTED : POOL_STATE_ACTIVE); 9520 fnvlist_add_uint64(config, ZPOOL_CONFIG_VERSION, spa_version(spa)); 9521 fnvlist_add_uint64(config, ZPOOL_CONFIG_POOL_TXG, spa->spa_config_txg); 9522 fnvlist_add_uint64(config, ZPOOL_CONFIG_POOL_GUID, 9523 spa_generate_guid(NULL)); 9524 VERIFY0(nvlist_add_boolean(config, ZPOOL_CONFIG_HAS_PER_VDEV_ZAPS)); 9525 (void) nvlist_lookup_string(props, 9526 zpool_prop_to_name(ZPOOL_PROP_ALTROOT), &altroot); 9527 9528 /* add the new pool to the namespace */ 9529 newspa = spa_add(newname, config, altroot); 9530 newspa->spa_avz_action = AVZ_ACTION_REBUILD; 9531 newspa->spa_config_txg = spa->spa_config_txg; 9532 spa_set_log_state(newspa, SPA_LOG_CLEAR); 9533 9534 /* release the spa config lock, retaining the namespace lock */ 9535 spa_vdev_config_exit(spa, NULL, txg, 0, FTAG); 9536 9537 if (zio_injection_enabled) 9538 zio_handle_panic_injection(spa, FTAG, 1); 9539 9540 spa_activate(newspa, spa_mode_global); 9541 spa_async_suspend(newspa); 9542 9543 /* 9544 * Temporarily stop the initializing and TRIM activity. We set the 9545 * state to ACTIVE so that we know to resume initializing or TRIM 9546 * once the split has completed. 9547 */ 9548 list_t vd_initialize_list; 9549 list_create(&vd_initialize_list, sizeof (vdev_t), 9550 offsetof(vdev_t, vdev_initialize_node)); 9551 9552 list_t vd_trim_list; 9553 list_create(&vd_trim_list, sizeof (vdev_t), 9554 offsetof(vdev_t, vdev_trim_node)); 9555 9556 for (c = 0; c < children; c++) { 9557 if (vml[c] != NULL && vml[c]->vdev_ops != &vdev_indirect_ops) { 9558 mutex_enter(&vml[c]->vdev_initialize_lock); 9559 vdev_initialize_stop(vml[c], 9560 VDEV_INITIALIZE_ACTIVE, &vd_initialize_list); 9561 mutex_exit(&vml[c]->vdev_initialize_lock); 9562 9563 mutex_enter(&vml[c]->vdev_trim_lock); 9564 vdev_trim_stop(vml[c], VDEV_TRIM_ACTIVE, &vd_trim_list); 9565 mutex_exit(&vml[c]->vdev_trim_lock); 9566 } 9567 } 9568 9569 vdev_initialize_stop_wait(spa, &vd_initialize_list); 9570 vdev_trim_stop_wait(spa, &vd_trim_list); 9571 9572 list_destroy(&vd_initialize_list); 9573 list_destroy(&vd_trim_list); 9574 9575 newspa->spa_config_source = SPA_CONFIG_SRC_SPLIT; 9576 newspa->spa_is_splitting = B_TRUE; 9577 9578 /* create the new pool from the disks of the original pool */ 9579 error = spa_load(newspa, SPA_LOAD_IMPORT, SPA_IMPORT_ASSEMBLE); 9580 if (error) 9581 goto out; 9582 9583 /* if that worked, generate a real config for the new pool */ 9584 if (newspa->spa_root_vdev != NULL) { 9585 newspa->spa_config_splitting = fnvlist_alloc(); 9586 fnvlist_add_uint64(newspa->spa_config_splitting, 9587 ZPOOL_CONFIG_SPLIT_GUID, spa_guid(spa)); 9588 spa_config_set(newspa, spa_config_generate(newspa, NULL, -1ULL, 9589 B_TRUE)); 9590 } 9591 9592 /* 9593 * Free the DTL space map objects inherited from the original pool 9594 * MOS so we won't leak them. 9595 */ 9596 if (ndtl != 0) { 9597 spa_split_dtl_arg_t ssda; 9598 9599 ssda.ssda_spa = newspa; 9600 ssda.ssda_objs = dtl_objs; 9601 ssda.ssda_count = ndtl; 9602 VERIFY0(dsl_sync_task(spa_name(newspa), NULL, 9603 spa_split_dtl_free_sync, &ssda, 0, ZFS_SPACE_CHECK_NONE)); 9604 } 9605 9606 /* set the props */ 9607 if (props != NULL) { 9608 spa_configfile_set(newspa, props, B_FALSE); 9609 error = spa_prop_set(newspa, props); 9610 if (error) 9611 goto out; 9612 } 9613 9614 /* flush everything */ 9615 txg = spa_vdev_config_enter(newspa); 9616 vdev_config_dirty(newspa->spa_root_vdev); 9617 (void) spa_vdev_config_exit(newspa, NULL, txg, 0, FTAG); 9618 9619 if (zio_injection_enabled) 9620 zio_handle_panic_injection(spa, FTAG, 2); 9621 9622 spa_async_resume(newspa); 9623 9624 /* finally, update the original pool's config */ 9625 txg = spa_vdev_config_enter(spa); 9626 tx = dmu_tx_create_dd(spa_get_dsl(spa)->dp_mos_dir); 9627 error = dmu_tx_assign(tx, DMU_TX_WAIT); 9628 if (error != 0) 9629 dmu_tx_abort(tx); 9630 for (c = 0; c < children; c++) { 9631 if (vml[c] != NULL && vml[c]->vdev_ops != &vdev_indirect_ops) { 9632 vdev_t *tvd = vml[c]->vdev_top; 9633 9634 /* 9635 * Need to be sure the detachable VDEV is not 9636 * on any *other* txg's DTL list to prevent it 9637 * from being accessed after it's freed. 9638 */ 9639 for (int t = 0; t < TXG_SIZE; t++) { 9640 (void) txg_list_remove_this( 9641 &tvd->vdev_dtl_list, vml[c], t); 9642 } 9643 9644 vdev_split(vml[c]); 9645 9646 /* 9647 * As in spa_vdev_detach(), mark the vdev detached 9648 * and dirty its DTL, so that vdev_dtl_sync() frees 9649 * the leaf's DTL space map object. 9650 */ 9651 vml[c]->vdev_detached = B_TRUE; 9652 9653 /* 9654 * The leaf ZAP was transferred to the new pool 9655 * and this pool's copy is destroyed by the AVZ 9656 * rebuild below, so clear it to keep 9657 * vdev_dtl_sync() from destroying it again. 9658 */ 9659 vml[c]->vdev_leaf_zap = 0; 9660 9661 /* 9662 * vml[c]->vdev_top may be stale; the 9663 * surviving top-level vdev is rvd->vdev_child[c]. 9664 */ 9665 if (vml[c]->vdev_dtl_sm != NULL) 9666 vdev_dirty(rvd->vdev_child[c], VDD_DTL, 9667 vml[c], txg); 9668 9669 if (error == 0) 9670 spa_history_log_internal(spa, "detach", tx, 9671 "vdev=%s", vml[c]->vdev_path); 9672 } 9673 } 9674 spa->spa_avz_action = AVZ_ACTION_REBUILD; 9675 vdev_config_dirty(spa->spa_root_vdev); 9676 spa->spa_config_splitting = NULL; 9677 nvlist_free(nvl); 9678 if (error == 0) 9679 dmu_tx_commit(tx); 9680 (void) spa_vdev_exit(spa, NULL, txg, 0); 9681 9682 /* 9683 * txg is synced, free vdevs. 9684 */ 9685 spa_config_enter(spa, SCL_STATE_ALL, spa, RW_WRITER); 9686 for (c = 0; c < children; c++) { 9687 if (vml[c] != NULL && vml[c]->vdev_ops != &vdev_indirect_ops) { 9688 ASSERT0P(vml[c]->vdev_dtl_sm); 9689 vdev_free(vml[c]); 9690 } 9691 } 9692 spa_config_exit(spa, SCL_STATE_ALL, spa); 9693 9694 if (zio_injection_enabled) 9695 zio_handle_panic_injection(spa, FTAG, 3); 9696 9697 /* split is complete; log a history record */ 9698 spa_history_log_internal(newspa, "split", NULL, 9699 "from pool %s", spa_name(spa)); 9700 9701 newspa->spa_is_splitting = B_FALSE; 9702 kmem_free(vml, children * sizeof (vdev_t *)); 9703 if (dtl_objs != NULL) 9704 kmem_free(dtl_objs, nleaves * sizeof (uint64_t)); 9705 9706 /* if we're not going to mount the filesystems in userland, export */ 9707 if (exp) 9708 error = spa_export_common(newname, POOL_STATE_EXPORTED, NULL, 9709 B_FALSE, B_FALSE); 9710 9711 return (error); 9712 9713 out: 9714 spa_unload(newspa); 9715 spa_deactivate(newspa); 9716 spa_remove(newspa); 9717 9718 txg = spa_vdev_config_enter(spa); 9719 9720 /* re-online all offlined disks */ 9721 for (c = 0; c < children; c++) { 9722 if (vml[c] != NULL) 9723 vml[c]->vdev_offline = B_FALSE; 9724 } 9725 9726 /* restart initializing or trimming disks as necessary */ 9727 spa_async_request(spa, SPA_ASYNC_INITIALIZE_RESTART); 9728 spa_async_request(spa, SPA_ASYNC_TRIM_RESTART); 9729 spa_async_request(spa, SPA_ASYNC_AUTOTRIM_RESTART); 9730 9731 vdev_reopen(spa->spa_root_vdev); 9732 9733 nvlist_free(spa->spa_config_splitting); 9734 spa->spa_config_splitting = NULL; 9735 (void) spa_vdev_exit(spa, NULL, txg, error); 9736 9737 kmem_free(vml, children * sizeof (vdev_t *)); 9738 if (dtl_objs != NULL) 9739 kmem_free(dtl_objs, nleaves * sizeof (uint64_t)); 9740 9741 return (error); 9742 } 9743 9744 /* 9745 * Find any device that's done replacing, or a vdev marked 'unspare' that's 9746 * currently spared, so we can detach it. 9747 */ 9748 static vdev_t * 9749 spa_vdev_resilver_done_hunt(vdev_t *vd) 9750 { 9751 vdev_t *newvd, *oldvd; 9752 9753 for (int c = 0; c < vd->vdev_children; c++) { 9754 oldvd = spa_vdev_resilver_done_hunt(vd->vdev_child[c]); 9755 if (oldvd != NULL) 9756 return (oldvd); 9757 } 9758 9759 /* 9760 * Check for a completed replacement. We always consider the first 9761 * vdev in the list to be the oldest vdev, and the last one to be 9762 * the newest (see spa_vdev_attach() for how that works). In 9763 * the case where the newest vdev is faulted, we will not automatically 9764 * remove it after a resilver completes. This is OK as it will require 9765 * user intervention to determine which disk the admin wishes to keep. 9766 */ 9767 if (vd->vdev_ops == &vdev_replacing_ops) { 9768 ASSERT(vd->vdev_children > 1); 9769 9770 newvd = vd->vdev_child[vd->vdev_children - 1]; 9771 oldvd = vd->vdev_child[0]; 9772 9773 if (vdev_dtl_empty(newvd, DTL_MISSING) && 9774 vdev_dtl_empty(newvd, DTL_OUTAGE) && 9775 !vdev_dtl_required(oldvd)) 9776 return (oldvd); 9777 } 9778 9779 /* 9780 * Check for a completed resilver with the 'unspare' flag set. 9781 * Also potentially update faulted state. 9782 */ 9783 if (vd->vdev_ops == &vdev_spare_ops) { 9784 vdev_t *first = vd->vdev_child[0]; 9785 vdev_t *last = vd->vdev_child[vd->vdev_children - 1]; 9786 9787 if (last->vdev_unspare) { 9788 oldvd = first; 9789 newvd = last; 9790 } else if (first->vdev_unspare) { 9791 oldvd = last; 9792 newvd = first; 9793 } else { 9794 oldvd = NULL; 9795 } 9796 9797 if (oldvd != NULL && 9798 vdev_dtl_empty(newvd, DTL_MISSING) && 9799 vdev_dtl_empty(newvd, DTL_OUTAGE) && 9800 !vdev_dtl_required(oldvd)) 9801 return (oldvd); 9802 9803 vdev_propagate_state(vd); 9804 9805 /* 9806 * If there are more than two spares attached to a disk, 9807 * and those spares are not required, then we want to 9808 * attempt to free them up now so that they can be used 9809 * by other pools. Once we're back down to a single 9810 * disk+spare, we stop removing them. 9811 */ 9812 if (vd->vdev_children > 2) { 9813 newvd = vd->vdev_child[1]; 9814 9815 if (newvd->vdev_isspare && last->vdev_isspare && 9816 vdev_dtl_empty(last, DTL_MISSING) && 9817 vdev_dtl_empty(last, DTL_OUTAGE) && 9818 !vdev_dtl_required(newvd)) 9819 return (newvd); 9820 } 9821 } 9822 9823 return (NULL); 9824 } 9825 9826 static void 9827 spa_vdev_resilver_done(spa_t *spa) 9828 { 9829 vdev_t *vd, *pvd, *ppvd; 9830 uint64_t guid, sguid, pguid, ppguid; 9831 9832 spa_config_enter(spa, SCL_ALL, FTAG, RW_WRITER); 9833 9834 while ((vd = spa_vdev_resilver_done_hunt(spa->spa_root_vdev)) != NULL) { 9835 pvd = vd->vdev_parent; 9836 ppvd = pvd->vdev_parent; 9837 guid = vd->vdev_guid; 9838 pguid = pvd->vdev_guid; 9839 ppguid = ppvd->vdev_guid; 9840 sguid = 0; 9841 /* 9842 * If we have just finished replacing a hot spared device, then 9843 * we need to detach the parent's first child (the original hot 9844 * spare) as well. 9845 */ 9846 if (ppvd->vdev_ops == &vdev_spare_ops && pvd->vdev_id == 0 && 9847 ppvd->vdev_children == 2) { 9848 ASSERT(pvd->vdev_ops == &vdev_replacing_ops); 9849 sguid = ppvd->vdev_child[1]->vdev_guid; 9850 } 9851 ASSERT(vd->vdev_resilver_txg == 0 || !vdev_dtl_required(vd)); 9852 9853 spa_config_exit(spa, SCL_ALL, FTAG); 9854 if (spa_vdev_detach(spa, guid, pguid, B_TRUE) != 0) 9855 return; 9856 if (sguid && spa_vdev_detach(spa, sguid, ppguid, B_TRUE) != 0) 9857 return; 9858 spa_config_enter(spa, SCL_ALL, FTAG, RW_WRITER); 9859 } 9860 9861 spa_config_exit(spa, SCL_ALL, FTAG); 9862 9863 /* 9864 * If a detach was not performed above replace waiters will not have 9865 * been notified. In which case we must do so now. 9866 */ 9867 spa_notify_waiters(spa); 9868 } 9869 9870 /* 9871 * Update the stored path or FRU for this vdev. 9872 */ 9873 static int 9874 spa_vdev_set_common(spa_t *spa, uint64_t guid, const char *value, 9875 boolean_t ispath) 9876 { 9877 vdev_t *vd; 9878 boolean_t sync = B_FALSE; 9879 9880 ASSERT(spa_writeable(spa)); 9881 9882 spa_vdev_state_enter(spa, SCL_ALL); 9883 9884 if ((vd = spa_lookup_by_guid(spa, guid, B_TRUE)) == NULL) 9885 return (spa_vdev_state_exit(spa, NULL, ENOENT)); 9886 9887 if (!vd->vdev_ops->vdev_op_leaf) 9888 return (spa_vdev_state_exit(spa, NULL, ENOTSUP)); 9889 9890 if (ispath) { 9891 if (strcmp(value, vd->vdev_path) != 0) { 9892 spa_strfree(vd->vdev_path); 9893 vd->vdev_path = spa_strdup(value); 9894 sync = B_TRUE; 9895 } 9896 } else { 9897 if (vd->vdev_fru == NULL) { 9898 vd->vdev_fru = spa_strdup(value); 9899 sync = B_TRUE; 9900 } else if (strcmp(value, vd->vdev_fru) != 0) { 9901 spa_strfree(vd->vdev_fru); 9902 vd->vdev_fru = spa_strdup(value); 9903 sync = B_TRUE; 9904 } 9905 } 9906 9907 return (spa_vdev_state_exit(spa, sync ? vd : NULL, 0)); 9908 } 9909 9910 int 9911 spa_vdev_setpath(spa_t *spa, uint64_t guid, const char *newpath) 9912 { 9913 return (spa_vdev_set_common(spa, guid, newpath, B_TRUE)); 9914 } 9915 9916 int 9917 spa_vdev_setfru(spa_t *spa, uint64_t guid, const char *newfru) 9918 { 9919 return (spa_vdev_set_common(spa, guid, newfru, B_FALSE)); 9920 } 9921 9922 /* 9923 * ========================================================================== 9924 * SPA Scanning 9925 * ========================================================================== 9926 */ 9927 int 9928 spa_scrub_pause_resume(spa_t *spa, pool_scrub_cmd_t cmd) 9929 { 9930 ASSERT0(spa_config_held(spa, SCL_ALL, RW_WRITER)); 9931 9932 if (dsl_scan_resilvering(spa->spa_dsl_pool)) 9933 return (SET_ERROR(EBUSY)); 9934 9935 return (dsl_scrub_set_pause_resume(spa->spa_dsl_pool, cmd)); 9936 } 9937 9938 int 9939 spa_scan_stop(spa_t *spa) 9940 { 9941 ASSERT0(spa_config_held(spa, SCL_ALL, RW_WRITER)); 9942 if (dsl_scan_resilvering(spa->spa_dsl_pool)) 9943 return (SET_ERROR(EBUSY)); 9944 9945 return (dsl_scan_cancel(spa->spa_dsl_pool)); 9946 } 9947 9948 int 9949 spa_scan(spa_t *spa, pool_scan_func_t func, pool_scrub_flags_t flags) 9950 { 9951 return (spa_scan_range(spa, func, 0, 0, flags)); 9952 } 9953 9954 int 9955 spa_scan_range(spa_t *spa, pool_scan_func_t func, uint64_t txgstart, 9956 uint64_t txgend, pool_scrub_flags_t flags) 9957 { 9958 dsl_scan_flags_t dsl_flags = 0; 9959 9960 ASSERT0(spa_config_held(spa, SCL_ALL, RW_WRITER)); 9961 9962 if (flags & POOL_SCRUB_THOROUGH) 9963 dsl_flags |= DSF_SCRUB_THOROUGH; 9964 9965 if (func >= POOL_SCAN_FUNCS || func == POOL_SCAN_NONE) 9966 return (SET_ERROR(ENOTSUP)); 9967 9968 if (func == POOL_SCAN_RESILVER && 9969 !spa_feature_is_enabled(spa, SPA_FEATURE_RESILVER_DEFER)) 9970 return (SET_ERROR(ENOTSUP)); 9971 9972 if (func != POOL_SCAN_SCRUB && (txgstart != 0 || txgend != 0)) 9973 return (SET_ERROR(ENOTSUP)); 9974 9975 /* 9976 * If a resilver was requested, but there is no DTL on a 9977 * writeable leaf device, we have nothing to do. 9978 */ 9979 if (func == POOL_SCAN_RESILVER && 9980 !vdev_resilver_needed(spa->spa_root_vdev, NULL, NULL)) { 9981 spa_async_request(spa, SPA_ASYNC_RESILVER_DONE); 9982 return (0); 9983 } 9984 9985 if (func == POOL_SCAN_ERRORSCRUB && 9986 !spa_feature_is_enabled(spa, SPA_FEATURE_HEAD_ERRLOG)) 9987 return (SET_ERROR(ENOTSUP)); 9988 9989 return (dsl_scan(spa->spa_dsl_pool, func, txgstart, txgend, dsl_flags)); 9990 } 9991 9992 /* 9993 * ========================================================================== 9994 * SPA async task processing 9995 * ========================================================================== 9996 */ 9997 9998 static void 9999 spa_async_remove(spa_t *spa, vdev_t *vd, boolean_t by_kernel) 10000 { 10001 if (vd->vdev_remove_wanted) { 10002 vd->vdev_remove_wanted = B_FALSE; 10003 vd->vdev_delayed_close = B_FALSE; 10004 vdev_set_state(vd, B_FALSE, VDEV_STATE_REMOVED, VDEV_AUX_NONE); 10005 10006 /* 10007 * We want to clear the stats, but we don't want to do a full 10008 * vdev_clear() as that will cause us to throw away 10009 * degraded/faulted state as well as attempt to reopen the 10010 * device, all of which is a waste. 10011 */ 10012 vd->vdev_stat.vs_read_errors = 0; 10013 vd->vdev_stat.vs_write_errors = 0; 10014 vd->vdev_stat.vs_checksum_errors = 0; 10015 10016 vdev_state_dirty(vd->vdev_top); 10017 10018 /* Tell userspace that the vdev is gone. */ 10019 zfs_post_remove(spa, vd, by_kernel); 10020 } 10021 10022 for (int c = 0; c < vd->vdev_children; c++) 10023 spa_async_remove(spa, vd->vdev_child[c], by_kernel); 10024 } 10025 10026 static void 10027 spa_async_fault_vdev(vdev_t *vd, boolean_t *suspend) 10028 { 10029 if (vd->vdev_fault_wanted) { 10030 vdev_state_t newstate = VDEV_STATE_FAULTED; 10031 vd->vdev_fault_wanted = B_FALSE; 10032 10033 /* 10034 * If this device has the only valid copy of the data, then 10035 * back off and simply mark the vdev as degraded instead. 10036 */ 10037 if (!vd->vdev_top->vdev_islog && vd->vdev_aux == NULL && 10038 vdev_dtl_required(vd)) { 10039 newstate = VDEV_STATE_DEGRADED; 10040 /* A required disk is missing so suspend the pool */ 10041 *suspend = B_TRUE; 10042 } 10043 vdev_set_state(vd, B_FALSE, newstate, VDEV_AUX_ERR_EXCEEDED); 10044 } 10045 for (int c = 0; c < vd->vdev_children; c++) 10046 spa_async_fault_vdev(vd->vdev_child[c], suspend); 10047 } 10048 10049 static void 10050 spa_async_autoexpand(spa_t *spa, vdev_t *vd) 10051 { 10052 if (!spa->spa_autoexpand) 10053 return; 10054 10055 for (int c = 0; c < vd->vdev_children; c++) { 10056 vdev_t *cvd = vd->vdev_child[c]; 10057 spa_async_autoexpand(spa, cvd); 10058 } 10059 10060 if (!vd->vdev_ops->vdev_op_leaf || vd->vdev_physpath == NULL) 10061 return; 10062 10063 spa_event_notify(vd->vdev_spa, vd, NULL, ESC_ZFS_VDEV_AUTOEXPAND); 10064 } 10065 10066 static __attribute__((noreturn)) void 10067 spa_async_thread(void *arg) 10068 { 10069 spa_t *spa = (spa_t *)arg; 10070 dsl_pool_t *dp = spa->spa_dsl_pool; 10071 uint32_t tasks; 10072 10073 ASSERT(spa->spa_sync_on); 10074 10075 mutex_enter(&spa->spa_async_lock); 10076 tasks = spa->spa_async_tasks; 10077 spa->spa_async_tasks = 0; 10078 mutex_exit(&spa->spa_async_lock); 10079 10080 /* 10081 * See if the config needs to be updated. 10082 */ 10083 if (tasks & SPA_ASYNC_CONFIG_UPDATE) { 10084 uint64_t old_space, new_space; 10085 10086 spa_namespace_enter(FTAG); 10087 old_space = metaslab_class_get_space(spa_normal_class(spa)); 10088 old_space += metaslab_class_get_space(spa_special_class(spa)); 10089 old_space += metaslab_class_get_space(spa_dedup_class(spa)); 10090 old_space += metaslab_class_get_space( 10091 spa_embedded_log_class(spa)); 10092 old_space += metaslab_class_get_space( 10093 spa_special_embedded_log_class(spa)); 10094 10095 spa_config_update(spa, SPA_CONFIG_UPDATE_POOL); 10096 10097 new_space = metaslab_class_get_space(spa_normal_class(spa)); 10098 new_space += metaslab_class_get_space(spa_special_class(spa)); 10099 new_space += metaslab_class_get_space(spa_dedup_class(spa)); 10100 new_space += metaslab_class_get_space( 10101 spa_embedded_log_class(spa)); 10102 new_space += metaslab_class_get_space( 10103 spa_special_embedded_log_class(spa)); 10104 spa_namespace_exit(FTAG); 10105 10106 /* 10107 * If the pool grew as a result of the config update, 10108 * then log an internal history event. 10109 */ 10110 if (new_space != old_space) { 10111 spa_history_log_internal(spa, "vdev online", NULL, 10112 "pool '%s' size: %llu(+%llu)", 10113 spa_name(spa), (u_longlong_t)new_space, 10114 (u_longlong_t)(new_space - old_space)); 10115 } 10116 } 10117 10118 /* 10119 * See if any devices need to be marked REMOVED. 10120 */ 10121 if (tasks & (SPA_ASYNC_REMOVE | SPA_ASYNC_REMOVE_BY_USER)) { 10122 boolean_t by_kernel = B_TRUE; 10123 if (tasks & SPA_ASYNC_REMOVE_BY_USER) 10124 by_kernel = B_FALSE; 10125 spa_vdev_state_enter(spa, SCL_NONE); 10126 spa_async_remove(spa, spa->spa_root_vdev, by_kernel); 10127 for (int i = 0; i < spa->spa_l2cache.sav_count; i++) 10128 spa_async_remove(spa, spa->spa_l2cache.sav_vdevs[i], 10129 by_kernel); 10130 for (int i = 0; i < spa->spa_spares.sav_count; i++) 10131 spa_async_remove(spa, spa->spa_spares.sav_vdevs[i], 10132 by_kernel); 10133 (void) spa_vdev_state_exit(spa, NULL, 0); 10134 } 10135 10136 if ((tasks & SPA_ASYNC_AUTOEXPAND) && !spa_suspended(spa)) { 10137 spa_config_enter(spa, SCL_CONFIG, FTAG, RW_READER); 10138 spa_async_autoexpand(spa, spa->spa_root_vdev); 10139 spa_config_exit(spa, SCL_CONFIG, FTAG); 10140 } 10141 10142 /* 10143 * See if any devices need to be marked faulted. 10144 */ 10145 if (tasks & SPA_ASYNC_FAULT_VDEV) { 10146 spa_vdev_state_enter(spa, SCL_NONE); 10147 boolean_t suspend = B_FALSE; 10148 spa_async_fault_vdev(spa->spa_root_vdev, &suspend); 10149 (void) spa_vdev_state_exit(spa, NULL, 0); 10150 if (suspend) 10151 zio_suspend(spa, NULL, ZIO_SUSPEND_IOERR); 10152 } 10153 10154 /* 10155 * If any devices are done replacing, detach them. 10156 */ 10157 if (tasks & SPA_ASYNC_RESILVER_DONE || 10158 tasks & SPA_ASYNC_REBUILD_DONE || 10159 tasks & SPA_ASYNC_DETACH_SPARE) { 10160 spa_vdev_resilver_done(spa); 10161 } 10162 10163 /* 10164 * Kick off a resilver. 10165 */ 10166 if (tasks & SPA_ASYNC_RESILVER && 10167 !vdev_rebuild_active(spa->spa_root_vdev) && 10168 (!dsl_scan_resilvering(dp) || 10169 !spa_feature_is_enabled(dp->dp_spa, SPA_FEATURE_RESILVER_DEFER))) 10170 dsl_scan_restart_resilver(dp, 0); 10171 10172 if (tasks & SPA_ASYNC_INITIALIZE_RESTART) { 10173 spa_namespace_enter(FTAG); 10174 spa_config_enter(spa, SCL_CONFIG, FTAG, RW_READER); 10175 vdev_initialize_restart(spa->spa_root_vdev); 10176 spa_config_exit(spa, SCL_CONFIG, FTAG); 10177 spa_namespace_exit(FTAG); 10178 } 10179 10180 if (tasks & SPA_ASYNC_TRIM_RESTART) { 10181 spa_namespace_enter(FTAG); 10182 spa_config_enter(spa, SCL_CONFIG, FTAG, RW_READER); 10183 vdev_trim_restart(spa->spa_root_vdev); 10184 spa_config_exit(spa, SCL_CONFIG, FTAG); 10185 spa_namespace_exit(FTAG); 10186 } 10187 10188 if (tasks & SPA_ASYNC_AUTOTRIM_RESTART) { 10189 spa_namespace_enter(FTAG); 10190 spa_config_enter(spa, SCL_CONFIG, FTAG, RW_READER); 10191 vdev_autotrim_restart(spa); 10192 spa_config_exit(spa, SCL_CONFIG, FTAG); 10193 spa_namespace_exit(FTAG); 10194 } 10195 10196 /* 10197 * Kick off L2 cache whole device TRIM. 10198 */ 10199 if (tasks & SPA_ASYNC_L2CACHE_TRIM) { 10200 spa_namespace_enter(FTAG); 10201 spa_config_enter(spa, SCL_CONFIG, FTAG, RW_READER); 10202 vdev_trim_l2arc(spa); 10203 spa_config_exit(spa, SCL_CONFIG, FTAG); 10204 spa_namespace_exit(FTAG); 10205 } 10206 10207 /* 10208 * Kick off L2 cache rebuilding. 10209 */ 10210 if (tasks & SPA_ASYNC_L2CACHE_REBUILD) { 10211 spa_namespace_enter(FTAG); 10212 spa_config_enter(spa, SCL_L2ARC, FTAG, RW_READER); 10213 l2arc_spa_rebuild_start(spa); 10214 spa_config_exit(spa, SCL_L2ARC, FTAG); 10215 spa_namespace_exit(FTAG); 10216 } 10217 10218 /* 10219 * Finish off snapshots whose deferred destruction was waiting on 10220 * something that has since let go of them. The destroy is a sync 10221 * task, which a suspended pool would never come back from, and the 10222 * export path waits on this thread, so leave the mark where it is 10223 * and pick it up on the next request or at the next import. 10224 */ 10225 if ((tasks & SPA_ASYNC_DEFER_DESTROY) && spa_writeable(spa) && 10226 !spa_suspended(spa)) 10227 dsl_destroy_snapshot_deferred(spa_name(spa)); 10228 10229 /* 10230 * Let the world know that we're done. 10231 */ 10232 mutex_enter(&spa->spa_async_lock); 10233 spa->spa_async_thread = NULL; 10234 cv_broadcast(&spa->spa_async_cv); 10235 mutex_exit(&spa->spa_async_lock); 10236 thread_exit(); 10237 } 10238 10239 void 10240 spa_async_suspend(spa_t *spa) 10241 { 10242 mutex_enter(&spa->spa_async_lock); 10243 spa->spa_async_suspended++; 10244 while (spa->spa_async_thread != NULL) 10245 cv_wait(&spa->spa_async_cv, &spa->spa_async_lock); 10246 mutex_exit(&spa->spa_async_lock); 10247 10248 spa_vdev_remove_suspend(spa); 10249 10250 zthr_t *condense_thread = spa->spa_condense_zthr; 10251 if (condense_thread != NULL) 10252 zthr_cancel(condense_thread); 10253 10254 zthr_t *raidz_expand_thread = spa->spa_raidz_expand_zthr; 10255 if (raidz_expand_thread != NULL) 10256 zthr_cancel(raidz_expand_thread); 10257 10258 zthr_t *discard_thread = spa->spa_checkpoint_discard_zthr; 10259 if (discard_thread != NULL) 10260 zthr_cancel(discard_thread); 10261 10262 zthr_t *ll_delete_thread = spa->spa_livelist_delete_zthr; 10263 if (ll_delete_thread != NULL) 10264 zthr_cancel(ll_delete_thread); 10265 10266 zthr_t *ll_condense_thread = spa->spa_livelist_condense_zthr; 10267 if (ll_condense_thread != NULL) 10268 zthr_cancel(ll_condense_thread); 10269 } 10270 10271 void 10272 spa_async_resume(spa_t *spa) 10273 { 10274 mutex_enter(&spa->spa_async_lock); 10275 ASSERT(spa->spa_async_suspended != 0); 10276 spa->spa_async_suspended--; 10277 mutex_exit(&spa->spa_async_lock); 10278 spa_restart_removal(spa); 10279 10280 zthr_t *condense_thread = spa->spa_condense_zthr; 10281 if (condense_thread != NULL) 10282 zthr_resume(condense_thread); 10283 10284 zthr_t *raidz_expand_thread = spa->spa_raidz_expand_zthr; 10285 if (raidz_expand_thread != NULL) 10286 zthr_resume(raidz_expand_thread); 10287 10288 zthr_t *discard_thread = spa->spa_checkpoint_discard_zthr; 10289 if (discard_thread != NULL) 10290 zthr_resume(discard_thread); 10291 10292 zthr_t *ll_delete_thread = spa->spa_livelist_delete_zthr; 10293 if (ll_delete_thread != NULL) 10294 zthr_resume(ll_delete_thread); 10295 10296 zthr_t *ll_condense_thread = spa->spa_livelist_condense_zthr; 10297 if (ll_condense_thread != NULL) 10298 zthr_resume(ll_condense_thread); 10299 } 10300 10301 static boolean_t 10302 spa_async_tasks_pending(spa_t *spa) 10303 { 10304 uint_t non_config_tasks; 10305 uint_t config_task; 10306 boolean_t config_task_suspended; 10307 10308 non_config_tasks = spa->spa_async_tasks & ~SPA_ASYNC_CONFIG_UPDATE; 10309 config_task = spa->spa_async_tasks & SPA_ASYNC_CONFIG_UPDATE; 10310 if (spa->spa_ccw_fail_time == 0) { 10311 config_task_suspended = B_FALSE; 10312 } else { 10313 config_task_suspended = 10314 (gethrtime() - spa->spa_ccw_fail_time) < 10315 ((hrtime_t)zfs_ccw_retry_interval * NANOSEC); 10316 } 10317 10318 return (non_config_tasks || (config_task && !config_task_suspended)); 10319 } 10320 10321 static void 10322 spa_async_dispatch(spa_t *spa) 10323 { 10324 mutex_enter(&spa->spa_async_lock); 10325 if (spa_async_tasks_pending(spa) && 10326 !spa->spa_async_suspended && 10327 spa->spa_async_thread == NULL) 10328 spa->spa_async_thread = thread_create(NULL, 0, 10329 spa_async_thread, spa, 0, &p0, TS_RUN, maxclsyspri); 10330 mutex_exit(&spa->spa_async_lock); 10331 } 10332 10333 void 10334 spa_async_request(spa_t *spa, int task) 10335 { 10336 zfs_dbgmsg("spa=%s async request task=%u", spa_load_name(spa), task); 10337 mutex_enter(&spa->spa_async_lock); 10338 spa->spa_async_tasks |= task; 10339 mutex_exit(&spa->spa_async_lock); 10340 } 10341 10342 int 10343 spa_async_tasks(spa_t *spa) 10344 { 10345 return (spa->spa_async_tasks); 10346 } 10347 10348 /* 10349 * ========================================================================== 10350 * SPA syncing routines 10351 * ========================================================================== 10352 */ 10353 10354 10355 static int 10356 bpobj_enqueue_cb(void *arg, const blkptr_t *bp, boolean_t bp_freed, 10357 dmu_tx_t *tx) 10358 { 10359 bpobj_t *bpo = arg; 10360 bpobj_enqueue(bpo, bp, bp_freed, tx); 10361 return (0); 10362 } 10363 10364 int 10365 bpobj_enqueue_alloc_cb(void *arg, const blkptr_t *bp, dmu_tx_t *tx) 10366 { 10367 return (bpobj_enqueue_cb(arg, bp, B_FALSE, tx)); 10368 } 10369 10370 int 10371 bpobj_enqueue_free_cb(void *arg, const blkptr_t *bp, dmu_tx_t *tx) 10372 { 10373 return (bpobj_enqueue_cb(arg, bp, B_TRUE, tx)); 10374 } 10375 10376 static int 10377 spa_free_sync_cb(void *arg, const blkptr_t *bp, dmu_tx_t *tx) 10378 { 10379 zio_t *pio = arg; 10380 10381 zio_nowait(zio_free_sync(pio, pio->io_spa, dmu_tx_get_txg(tx), bp, 10382 pio->io_flags)); 10383 return (0); 10384 } 10385 10386 static int 10387 bpobj_spa_free_sync_cb(void *arg, const blkptr_t *bp, boolean_t bp_freed, 10388 dmu_tx_t *tx) 10389 { 10390 ASSERT(!bp_freed); 10391 return (spa_free_sync_cb(arg, bp, tx)); 10392 } 10393 10394 /* 10395 * Note: this simple function is not inlined to make it easier to dtrace the 10396 * amount of time spent syncing frees. 10397 */ 10398 static void 10399 spa_sync_frees(spa_t *spa, bplist_t *bpl, dmu_tx_t *tx) 10400 { 10401 zio_t *zio = zio_root(spa, NULL, NULL, 0); 10402 bplist_iterate(bpl, spa_free_sync_cb, zio, tx); 10403 VERIFY0(zio_wait(zio)); 10404 } 10405 10406 /* 10407 * Note: this simple function is not inlined to make it easier to dtrace the 10408 * amount of time spent syncing deferred frees. 10409 */ 10410 static void 10411 spa_sync_deferred_frees(spa_t *spa, dmu_tx_t *tx) 10412 { 10413 if (spa_sync_pass(spa) != 1) 10414 return; 10415 10416 /* 10417 * Note: 10418 * If the log space map feature is active, we stop deferring 10419 * frees to the next TXG and therefore running this function 10420 * would be considered a no-op as spa_deferred_bpobj should 10421 * not have any entries. 10422 * 10423 * That said we run this function anyway (instead of returning 10424 * immediately) for the edge-case scenario where we just 10425 * activated the log space map feature in this TXG but we have 10426 * deferred frees from the previous TXG. 10427 */ 10428 zio_t *zio = zio_root(spa, NULL, NULL, 0); 10429 VERIFY3U(bpobj_iterate(&spa->spa_deferred_bpobj, 10430 bpobj_spa_free_sync_cb, zio, tx), ==, 0); 10431 VERIFY0(zio_wait(zio)); 10432 } 10433 10434 static void 10435 spa_sync_nvlist(spa_t *spa, uint64_t obj, nvlist_t *nv, dmu_tx_t *tx) 10436 { 10437 char *packed = NULL; 10438 size_t bufsize; 10439 size_t nvsize = 0; 10440 dmu_buf_t *db; 10441 10442 VERIFY0(nvlist_size(nv, &nvsize, NV_ENCODE_XDR)); 10443 10444 /* 10445 * Write full (SPA_CONFIG_BLOCKSIZE) blocks of configuration 10446 * information. This avoids the dmu_buf_will_dirty() path and 10447 * saves us a pre-read to get data we don't actually care about. 10448 */ 10449 bufsize = P2ROUNDUP((uint64_t)nvsize, SPA_CONFIG_BLOCKSIZE); 10450 packed = vmem_alloc(bufsize, KM_SLEEP); 10451 10452 VERIFY0(nvlist_pack(nv, &packed, &nvsize, NV_ENCODE_XDR, 10453 KM_SLEEP)); 10454 memset(packed + nvsize, 0, bufsize - nvsize); 10455 10456 dmu_write(spa->spa_meta_objset, obj, 0, bufsize, packed, tx, 10457 DMU_READ_NO_PREFETCH); 10458 10459 vmem_free(packed, bufsize); 10460 10461 VERIFY0(dmu_bonus_hold(spa->spa_meta_objset, obj, FTAG, &db)); 10462 dmu_buf_will_dirty(db, tx); 10463 *(uint64_t *)db->db_data = nvsize; 10464 dmu_buf_rele(db, FTAG); 10465 } 10466 10467 static void 10468 spa_sync_aux_dev(spa_t *spa, spa_aux_vdev_t *sav, dmu_tx_t *tx, 10469 const char *config, const char *entry) 10470 { 10471 nvlist_t *nvroot; 10472 nvlist_t **list; 10473 int i; 10474 10475 if (!sav->sav_sync) 10476 return; 10477 10478 /* 10479 * Update the MOS nvlist describing the list of available devices. 10480 * spa_validate_aux() will have already made sure this nvlist is 10481 * valid and the vdevs are labeled appropriately. 10482 */ 10483 if (sav->sav_object == 0) { 10484 sav->sav_object = dmu_object_alloc(spa->spa_meta_objset, 10485 DMU_OT_PACKED_NVLIST, 1 << 14, DMU_OT_PACKED_NVLIST_SIZE, 10486 sizeof (uint64_t), tx); 10487 VERIFY(zap_update(spa->spa_meta_objset, 10488 DMU_POOL_DIRECTORY_OBJECT, entry, sizeof (uint64_t), 1, 10489 &sav->sav_object, tx) == 0); 10490 } 10491 10492 nvroot = fnvlist_alloc(); 10493 if (sav->sav_count == 0) { 10494 fnvlist_add_nvlist_array(nvroot, config, 10495 (const nvlist_t * const *)NULL, 0); 10496 } else { 10497 list = kmem_alloc(sav->sav_count*sizeof (void *), KM_SLEEP); 10498 for (i = 0; i < sav->sav_count; i++) 10499 list[i] = vdev_config_generate(spa, sav->sav_vdevs[i], 10500 B_FALSE, VDEV_CONFIG_L2CACHE); 10501 fnvlist_add_nvlist_array(nvroot, config, 10502 (const nvlist_t * const *)list, sav->sav_count); 10503 for (i = 0; i < sav->sav_count; i++) 10504 nvlist_free(list[i]); 10505 kmem_free(list, sav->sav_count * sizeof (void *)); 10506 } 10507 10508 spa_sync_nvlist(spa, sav->sav_object, nvroot, tx); 10509 nvlist_free(nvroot); 10510 10511 sav->sav_sync = B_FALSE; 10512 } 10513 10514 /* 10515 * Rebuild spa's all-vdev ZAP from the vdev ZAPs indicated in each vdev_t. 10516 * The all-vdev ZAP must be empty. 10517 */ 10518 static void 10519 spa_avz_build(vdev_t *vd, uint64_t avz, dmu_tx_t *tx) 10520 { 10521 spa_t *spa = vd->vdev_spa; 10522 10523 if (vd->vdev_root_zap != 0 && 10524 spa_feature_is_active(spa, SPA_FEATURE_AVZ_V2)) { 10525 VERIFY0(zap_add_int(spa->spa_meta_objset, avz, 10526 vd->vdev_root_zap, tx)); 10527 } 10528 if (vd->vdev_top_zap != 0) { 10529 VERIFY0(zap_add_int(spa->spa_meta_objset, avz, 10530 vd->vdev_top_zap, tx)); 10531 } 10532 if (vd->vdev_leaf_zap != 0) { 10533 VERIFY0(zap_add_int(spa->spa_meta_objset, avz, 10534 vd->vdev_leaf_zap, tx)); 10535 } 10536 for (uint64_t i = 0; i < vd->vdev_children; i++) { 10537 spa_avz_build(vd->vdev_child[i], avz, tx); 10538 } 10539 } 10540 10541 static void 10542 spa_sync_config_object(spa_t *spa, dmu_tx_t *tx) 10543 { 10544 nvlist_t *config; 10545 10546 /* 10547 * If the pool is being imported from a pre-per-vdev-ZAP version of ZFS, 10548 * its config may not be dirty but we still need to build per-vdev ZAPs. 10549 * Similarly, if the pool is being assembled (e.g. after a split), we 10550 * need to rebuild the AVZ although the config may not be dirty. 10551 */ 10552 if (list_is_empty(&spa->spa_config_dirty_list) && 10553 spa->spa_avz_action == AVZ_ACTION_NONE) 10554 return; 10555 10556 spa_config_enter(spa, SCL_STATE, FTAG, RW_READER); 10557 10558 ASSERT(spa->spa_avz_action == AVZ_ACTION_NONE || 10559 spa->spa_avz_action == AVZ_ACTION_INITIALIZE || 10560 spa->spa_all_vdev_zaps != 0); 10561 10562 if (spa->spa_avz_action == AVZ_ACTION_REBUILD) { 10563 /* Make and build the new AVZ */ 10564 uint64_t new_avz = zap_create(spa->spa_meta_objset, 10565 DMU_OTN_ZAP_METADATA, DMU_OT_NONE, 0, tx); 10566 spa_avz_build(spa->spa_root_vdev, new_avz, tx); 10567 10568 /* Diff old AVZ with new one */ 10569 zap_cursor_t zc; 10570 zap_attribute_t *za = zap_attribute_alloc(); 10571 10572 for (zap_cursor_init(&zc, spa->spa_meta_objset, 10573 spa->spa_all_vdev_zaps); 10574 zap_cursor_retrieve(&zc, za) == 0; 10575 zap_cursor_advance(&zc)) { 10576 uint64_t vdzap = za->za_first_integer; 10577 if (zap_lookup_int(spa->spa_meta_objset, new_avz, 10578 vdzap) == ENOENT) { 10579 /* 10580 * ZAP is listed in old AVZ but not in new one; 10581 * destroy it 10582 */ 10583 VERIFY0(zap_destroy(spa->spa_meta_objset, vdzap, 10584 tx)); 10585 } 10586 } 10587 10588 zap_cursor_fini(&zc); 10589 zap_attribute_free(za); 10590 10591 /* Destroy the old AVZ */ 10592 VERIFY0(zap_destroy(spa->spa_meta_objset, 10593 spa->spa_all_vdev_zaps, tx)); 10594 10595 /* Replace the old AVZ in the dir obj with the new one */ 10596 VERIFY0(zap_update(spa->spa_meta_objset, 10597 DMU_POOL_DIRECTORY_OBJECT, DMU_POOL_VDEV_ZAP_MAP, 10598 sizeof (new_avz), 1, &new_avz, tx)); 10599 10600 spa->spa_all_vdev_zaps = new_avz; 10601 } else if (spa->spa_avz_action == AVZ_ACTION_DESTROY) { 10602 zap_cursor_t zc; 10603 zap_attribute_t *za = zap_attribute_alloc(); 10604 10605 /* Walk through the AVZ and destroy all listed ZAPs */ 10606 for (zap_cursor_init(&zc, spa->spa_meta_objset, 10607 spa->spa_all_vdev_zaps); 10608 zap_cursor_retrieve(&zc, za) == 0; 10609 zap_cursor_advance(&zc)) { 10610 uint64_t zap = za->za_first_integer; 10611 VERIFY0(zap_destroy(spa->spa_meta_objset, zap, tx)); 10612 } 10613 10614 zap_cursor_fini(&zc); 10615 zap_attribute_free(za); 10616 10617 /* Destroy and unlink the AVZ itself */ 10618 VERIFY0(zap_destroy(spa->spa_meta_objset, 10619 spa->spa_all_vdev_zaps, tx)); 10620 VERIFY0(zap_remove(spa->spa_meta_objset, 10621 DMU_POOL_DIRECTORY_OBJECT, DMU_POOL_VDEV_ZAP_MAP, tx)); 10622 spa->spa_all_vdev_zaps = 0; 10623 } 10624 10625 if (spa->spa_all_vdev_zaps == 0) { 10626 spa->spa_all_vdev_zaps = zap_create_link(spa->spa_meta_objset, 10627 DMU_OTN_ZAP_METADATA, DMU_POOL_DIRECTORY_OBJECT, 10628 DMU_POOL_VDEV_ZAP_MAP, tx); 10629 } 10630 spa->spa_avz_action = AVZ_ACTION_NONE; 10631 10632 /* Create ZAPs for vdevs that don't have them. */ 10633 vdev_construct_zaps(spa->spa_root_vdev, tx); 10634 10635 config = spa_config_generate(spa, spa->spa_root_vdev, 10636 dmu_tx_get_txg(tx), B_FALSE); 10637 10638 /* 10639 * If we're upgrading the spa version then make sure that 10640 * the config object gets updated with the correct version. 10641 */ 10642 if (spa->spa_ubsync.ub_version < spa->spa_uberblock.ub_version) 10643 fnvlist_add_uint64(config, ZPOOL_CONFIG_VERSION, 10644 spa->spa_uberblock.ub_version); 10645 10646 spa_config_exit(spa, SCL_STATE, FTAG); 10647 10648 nvlist_free(spa->spa_config_syncing); 10649 spa->spa_config_syncing = config; 10650 10651 spa_sync_nvlist(spa, spa->spa_config_object, config, tx); 10652 } 10653 10654 static void 10655 spa_sync_version(void *arg, dmu_tx_t *tx) 10656 { 10657 uint64_t *versionp = arg; 10658 uint64_t version = *versionp; 10659 spa_t *spa = dmu_tx_pool(tx)->dp_spa; 10660 10661 /* 10662 * Setting the version is special cased when first creating the pool. 10663 */ 10664 ASSERT(tx->tx_txg != TXG_INITIAL); 10665 10666 ASSERT(SPA_VERSION_IS_SUPPORTED(version)); 10667 ASSERT(version >= spa_version(spa)); 10668 10669 spa->spa_uberblock.ub_version = version; 10670 vdev_config_dirty(spa->spa_root_vdev); 10671 spa_history_log_internal(spa, "set", tx, "version=%lld", 10672 (longlong_t)version); 10673 } 10674 10675 /* 10676 * Set zpool properties. 10677 */ 10678 static void 10679 spa_sync_props(void *arg, dmu_tx_t *tx) 10680 { 10681 nvlist_t *nvp = arg; 10682 spa_t *spa = dmu_tx_pool(tx)->dp_spa; 10683 objset_t *mos = spa->spa_meta_objset; 10684 nvpair_t *elem = NULL; 10685 10686 mutex_enter(&spa->spa_props_lock); 10687 10688 while ((elem = nvlist_next_nvpair(nvp, elem))) { 10689 uint64_t intval; 10690 const char *strval, *fname; 10691 zpool_prop_t prop; 10692 const char *propname; 10693 const char *elemname = nvpair_name(elem); 10694 zprop_type_t proptype; 10695 spa_feature_t fid; 10696 10697 switch (prop = zpool_name_to_prop(elemname)) { 10698 case ZPOOL_PROP_VERSION: 10699 intval = fnvpair_value_uint64(elem); 10700 /* 10701 * The version is synced separately before other 10702 * properties and should be correct by now. 10703 */ 10704 ASSERT3U(spa_version(spa), >=, intval); 10705 break; 10706 10707 case ZPOOL_PROP_ALTROOT: 10708 /* 10709 * 'altroot' is a non-persistent property. It should 10710 * have been set temporarily at creation or import time. 10711 */ 10712 ASSERT(spa->spa_root != NULL); 10713 break; 10714 10715 case ZPOOL_PROP_READONLY: 10716 case ZPOOL_PROP_CACHEFILE: 10717 /* 10718 * 'readonly' and 'cachefile' are also non-persistent 10719 * properties. 10720 */ 10721 break; 10722 case ZPOOL_PROP_COMMENT: 10723 strval = fnvpair_value_string(elem); 10724 if (spa->spa_comment != NULL) 10725 spa_strfree(spa->spa_comment); 10726 spa->spa_comment = spa_strdup(strval); 10727 /* 10728 * We need to dirty the configuration on all the vdevs 10729 * so that their labels get updated. We also need to 10730 * update the cache file to keep it in sync with the 10731 * MOS version. It's unnecessary to do this for pool 10732 * creation since the vdev's configuration has already 10733 * been dirtied. 10734 */ 10735 if (tx->tx_txg != TXG_INITIAL) { 10736 vdev_config_dirty(spa->spa_root_vdev); 10737 spa_async_request(spa, SPA_ASYNC_CONFIG_UPDATE); 10738 } 10739 spa_history_log_internal(spa, "set", tx, 10740 "%s=%s", elemname, strval); 10741 break; 10742 case ZPOOL_PROP_COMPATIBILITY: 10743 strval = fnvpair_value_string(elem); 10744 if (spa->spa_compatibility != NULL) 10745 spa_strfree(spa->spa_compatibility); 10746 spa->spa_compatibility = spa_strdup(strval); 10747 /* 10748 * Dirty the configuration on vdevs as above. 10749 */ 10750 if (tx->tx_txg != TXG_INITIAL) { 10751 vdev_config_dirty(spa->spa_root_vdev); 10752 spa_async_request(spa, SPA_ASYNC_CONFIG_UPDATE); 10753 } 10754 10755 spa_history_log_internal(spa, "set", tx, 10756 "%s=%s", nvpair_name(elem), strval); 10757 break; 10758 10759 case ZPOOL_PROP_INVAL: 10760 if (zpool_prop_feature(elemname)) { 10761 fname = strchr(elemname, '@') + 1; 10762 VERIFY0(zfeature_lookup_name(fname, &fid)); 10763 10764 spa_feature_enable(spa, fid, tx); 10765 spa_history_log_internal(spa, "set", tx, 10766 "%s=enabled", elemname); 10767 break; 10768 } else if (!zfs_prop_user(elemname)) { 10769 ASSERT(zpool_prop_feature(elemname)); 10770 break; 10771 } 10772 zfs_fallthrough; 10773 default: 10774 /* 10775 * Set pool property values in the poolprops mos object. 10776 */ 10777 if (spa->spa_pool_props_object == 0) { 10778 spa->spa_pool_props_object = 10779 zap_create_link(mos, DMU_OT_POOL_PROPS, 10780 DMU_POOL_DIRECTORY_OBJECT, DMU_POOL_PROPS, 10781 tx); 10782 } 10783 10784 /* normalize the property name */ 10785 if (prop == ZPOOL_PROP_INVAL) { 10786 propname = elemname; 10787 proptype = PROP_TYPE_STRING; 10788 } else { 10789 propname = zpool_prop_to_name(prop); 10790 proptype = zpool_prop_get_type(prop); 10791 } 10792 10793 if (nvpair_type(elem) == DATA_TYPE_STRING) { 10794 ASSERT(proptype == PROP_TYPE_STRING); 10795 strval = fnvpair_value_string(elem); 10796 if (strlen(strval) == 0) { 10797 /* remove the property if value == "" */ 10798 (void) zap_remove(mos, 10799 spa->spa_pool_props_object, 10800 propname, tx); 10801 } else { 10802 VERIFY0(zap_update(mos, 10803 spa->spa_pool_props_object, 10804 propname, 1, strlen(strval) + 1, 10805 strval, tx)); 10806 } 10807 spa_history_log_internal(spa, "set", tx, 10808 "%s=%s", elemname, strval); 10809 } else if (nvpair_type(elem) == DATA_TYPE_UINT64) { 10810 intval = fnvpair_value_uint64(elem); 10811 10812 if (proptype == PROP_TYPE_INDEX) { 10813 const char *unused; 10814 VERIFY0(zpool_prop_index_to_string( 10815 prop, intval, &unused)); 10816 } 10817 VERIFY0(zap_update(mos, 10818 spa->spa_pool_props_object, propname, 10819 8, 1, &intval, tx)); 10820 spa_history_log_internal(spa, "set", tx, 10821 "%s=%lld", elemname, 10822 (longlong_t)intval); 10823 10824 switch (prop) { 10825 case ZPOOL_PROP_DELEGATION: 10826 spa->spa_delegation = intval; 10827 break; 10828 case ZPOOL_PROP_BOOTFS: 10829 spa->spa_bootfs = intval; 10830 break; 10831 case ZPOOL_PROP_FAILUREMODE: 10832 spa->spa_failmode = intval; 10833 break; 10834 case ZPOOL_PROP_AUTOTRIM: 10835 spa->spa_autotrim = intval; 10836 spa_async_request(spa, 10837 SPA_ASYNC_AUTOTRIM_RESTART); 10838 break; 10839 case ZPOOL_PROP_AUTOEXPAND: 10840 spa->spa_autoexpand = intval; 10841 if (tx->tx_txg != TXG_INITIAL) 10842 spa_async_request(spa, 10843 SPA_ASYNC_AUTOEXPAND); 10844 break; 10845 case ZPOOL_PROP_MULTIHOST: 10846 spa->spa_multihost = intval; 10847 break; 10848 case ZPOOL_PROP_DEDUP_TABLE_QUOTA: 10849 spa->spa_dedup_table_quota = intval; 10850 break; 10851 default: 10852 break; 10853 } 10854 } else { 10855 ASSERT(0); /* not allowed */ 10856 } 10857 } 10858 10859 } 10860 10861 mutex_exit(&spa->spa_props_lock); 10862 } 10863 10864 /* 10865 * Perform one-time upgrade on-disk changes. spa_version() does not 10866 * reflect the new version this txg, so there must be no changes this 10867 * txg to anything that the upgrade code depends on after it executes. 10868 * Therefore this must be called after dsl_pool_sync() does the sync 10869 * tasks. 10870 */ 10871 static void 10872 spa_sync_upgrades(spa_t *spa, dmu_tx_t *tx) 10873 { 10874 if (spa_sync_pass(spa) != 1) 10875 return; 10876 10877 uint64_t oldver = spa->spa_ubsync.ub_version; 10878 uint64_t newver = spa->spa_uberblock.ub_version; 10879 10880 /* 10881 * These upgrades change DSL namespace, so they need the 10882 * writer lock. 10883 */ 10884 boolean_t need_origin = oldver < SPA_VERSION_ORIGIN && 10885 newver >= SPA_VERSION_ORIGIN; 10886 boolean_t need_clones = oldver < SPA_VERSION_NEXT_CLONES && 10887 newver >= SPA_VERSION_NEXT_CLONES; 10888 boolean_t need_dir_clones = oldver < SPA_VERSION_DIR_CLONES && 10889 newver >= SPA_VERSION_DIR_CLONES; 10890 10891 if (need_origin || need_clones || need_dir_clones) { 10892 dsl_pool_t *dp = spa->spa_dsl_pool; 10893 10894 rrw_enter(&dp->dp_config_rwlock, RW_WRITER, FTAG); 10895 10896 if (need_origin) { 10897 dsl_pool_create_origin(dp, tx); 10898 10899 /* Keeping the origin open increases spa_minref */ 10900 spa->spa_minref += 3; 10901 } 10902 10903 if (need_clones) { 10904 dsl_pool_upgrade_clones(dp, tx); 10905 } 10906 10907 if (need_dir_clones) { 10908 dsl_pool_upgrade_dir_clones(dp, tx); 10909 10910 /* Keeping the freedir open increases spa_minref */ 10911 spa->spa_minref += 3; 10912 } 10913 10914 rrw_exit(&dp->dp_config_rwlock, FTAG); 10915 } 10916 10917 /* Remaining upgrades do not need dp_config_rwlock */ 10918 10919 if (oldver < SPA_VERSION_FEATURES && newver >= SPA_VERSION_FEATURES) { 10920 spa_feature_create_zap_objects(spa, tx); 10921 } 10922 10923 /* 10924 * LZ4_COMPRESS feature's behaviour was changed to activate_on_enable 10925 * when possibility to use lz4 compression for metadata was added 10926 * Old pools that have this feature enabled must be upgraded to have 10927 * this feature active 10928 */ 10929 if (newver >= SPA_VERSION_FEATURES) { 10930 boolean_t lz4_en = spa_feature_is_enabled(spa, 10931 SPA_FEATURE_LZ4_COMPRESS); 10932 boolean_t lz4_ac = spa_feature_is_active(spa, 10933 SPA_FEATURE_LZ4_COMPRESS); 10934 10935 if (lz4_en && !lz4_ac) 10936 spa_feature_incr(spa, SPA_FEATURE_LZ4_COMPRESS, tx); 10937 } 10938 10939 /* 10940 * If we haven't written the salt, do so now. Note that the 10941 * feature may not be activated yet, but that's fine since 10942 * the presence of this ZAP entry is backwards compatible. 10943 */ 10944 if (zap_contains(spa->spa_meta_objset, DMU_POOL_DIRECTORY_OBJECT, 10945 DMU_POOL_CHECKSUM_SALT) == ENOENT) { 10946 VERIFY0(zap_add(spa->spa_meta_objset, 10947 DMU_POOL_DIRECTORY_OBJECT, DMU_POOL_CHECKSUM_SALT, 1, 10948 sizeof (spa->spa_cksum_salt.zcs_bytes), 10949 spa->spa_cksum_salt.zcs_bytes, tx)); 10950 } 10951 } 10952 10953 static void 10954 vdev_indirect_state_sync_verify(vdev_t *vd) 10955 { 10956 vdev_indirect_mapping_t *vim __maybe_unused = vd->vdev_indirect_mapping; 10957 vdev_indirect_births_t *vib __maybe_unused = vd->vdev_indirect_births; 10958 10959 if (vd->vdev_ops == &vdev_indirect_ops) { 10960 ASSERT(vim != NULL); 10961 ASSERT(vib != NULL); 10962 } 10963 10964 uint64_t obsolete_sm_object = 0; 10965 ASSERT0(vdev_obsolete_sm_object(vd, &obsolete_sm_object)); 10966 if (obsolete_sm_object != 0) { 10967 ASSERT(vd->vdev_obsolete_sm != NULL); 10968 ASSERT(vd->vdev_removing || 10969 vd->vdev_ops == &vdev_indirect_ops); 10970 ASSERT(vdev_indirect_mapping_num_entries(vim) > 0); 10971 ASSERT(vdev_indirect_mapping_bytes_mapped(vim) > 0); 10972 ASSERT3U(obsolete_sm_object, ==, 10973 space_map_object(vd->vdev_obsolete_sm)); 10974 ASSERT3U(vdev_indirect_mapping_bytes_mapped(vim), >=, 10975 space_map_allocated(vd->vdev_obsolete_sm)); 10976 } 10977 ASSERT(vd->vdev_obsolete_segments != NULL); 10978 10979 /* 10980 * Since frees / remaps to an indirect vdev can only 10981 * happen in syncing context, the obsolete segments 10982 * tree must be empty when we start syncing. 10983 */ 10984 ASSERT0(zfs_range_tree_space(vd->vdev_obsolete_segments)); 10985 } 10986 10987 /* 10988 * Set the top-level vdev's max queue depth. Evaluate each top-level's 10989 * async write queue depth in case it changed. The max queue depth will 10990 * not change in the middle of syncing out this txg. 10991 */ 10992 static void 10993 spa_sync_adjust_vdev_max_queue_depth(spa_t *spa) 10994 { 10995 ASSERT(spa_writeable(spa)); 10996 10997 metaslab_class_balance(spa_normal_class(spa), B_TRUE); 10998 metaslab_class_balance(spa_special_class(spa), B_TRUE); 10999 metaslab_class_balance(spa_dedup_class(spa), B_TRUE); 11000 } 11001 11002 static void 11003 spa_sync_condense_indirect(spa_t *spa, dmu_tx_t *tx) 11004 { 11005 ASSERT(spa_writeable(spa)); 11006 11007 vdev_t *rvd = spa->spa_root_vdev; 11008 for (int c = 0; c < rvd->vdev_children; c++) { 11009 vdev_t *vd = rvd->vdev_child[c]; 11010 vdev_indirect_state_sync_verify(vd); 11011 11012 if (vdev_indirect_should_condense(vd)) { 11013 spa_condense_indirect_start_sync(vd, tx); 11014 break; 11015 } 11016 } 11017 } 11018 11019 static void 11020 spa_sync_iterate_to_convergence(spa_t *spa, dmu_tx_t *tx) 11021 { 11022 objset_t *mos = spa->spa_meta_objset; 11023 dsl_pool_t *dp = spa->spa_dsl_pool; 11024 uint64_t txg = tx->tx_txg; 11025 bplist_t *free_bpl = &spa->spa_free_bplist[txg & TXG_MASK]; 11026 11027 do { 11028 int pass = ++spa->spa_sync_pass; 11029 11030 spa_sync_config_object(spa, tx); 11031 spa_sync_aux_dev(spa, &spa->spa_spares, tx, 11032 ZPOOL_CONFIG_SPARES, DMU_POOL_SPARES); 11033 spa_sync_aux_dev(spa, &spa->spa_l2cache, tx, 11034 ZPOOL_CONFIG_L2CACHE, DMU_POOL_L2CACHE); 11035 spa_errlog_sync(spa, txg); 11036 dsl_pool_sync(dp, txg); 11037 11038 if (pass < zfs_sync_pass_deferred_free || 11039 spa_feature_is_active(spa, SPA_FEATURE_LOG_SPACEMAP)) { 11040 /* 11041 * If the log space map feature is active we don't 11042 * care about deferred frees and the deferred bpobj 11043 * as the log space map should effectively have the 11044 * same results (i.e. appending only to one object). 11045 */ 11046 spa_sync_frees(spa, free_bpl, tx); 11047 } else { 11048 /* 11049 * We can not defer frees in pass 1, because 11050 * we sync the deferred frees later in pass 1. 11051 */ 11052 ASSERT3U(pass, >, 1); 11053 bplist_iterate(free_bpl, bpobj_enqueue_alloc_cb, 11054 &spa->spa_deferred_bpobj, tx); 11055 } 11056 11057 brt_sync(spa, txg); 11058 ddt_sync(spa, txg); 11059 dsl_scan_sync(dp, tx); 11060 dsl_errorscrub_sync(dp, tx); 11061 svr_sync(spa, tx); 11062 spa_sync_upgrades(spa, tx); 11063 11064 spa_flush_metaslabs(spa, tx); 11065 11066 vdev_t *vd = NULL; 11067 while ((vd = txg_list_remove(&spa->spa_vdev_txg_list, txg)) 11068 != NULL) 11069 vdev_sync(vd, txg); 11070 11071 if (pass == 1) { 11072 /* 11073 * dsl_pool_sync() -> dp_sync_tasks may have dirtied 11074 * the config. If that happens, this txg should not 11075 * be a no-op. So we must sync the config to the MOS 11076 * before checking for no-op. 11077 * 11078 * Note that when the config is dirty, it will 11079 * be written to the MOS (i.e. the MOS will be 11080 * dirtied) every time we call spa_sync_config_object() 11081 * in this txg. Therefore we can't call this after 11082 * dsl_pool_sync() every pass, because it would 11083 * prevent us from converging, since we'd dirty 11084 * the MOS every pass. 11085 * 11086 * Sync tasks can only be processed in pass 1, so 11087 * there's no need to do this in later passes. 11088 */ 11089 spa_sync_config_object(spa, tx); 11090 } 11091 11092 /* 11093 * Note: We need to check if the MOS is dirty because we could 11094 * have marked the MOS dirty without updating the uberblock 11095 * (e.g. if we have sync tasks but no dirty user data). We need 11096 * to check the uberblock's rootbp because it is updated if we 11097 * have synced out dirty data (though in this case the MOS will 11098 * most likely also be dirty due to second order effects, we 11099 * don't want to rely on that here). 11100 */ 11101 if (pass == 1 && 11102 BP_GET_LOGICAL_BIRTH(&spa->spa_uberblock.ub_rootbp) < txg && 11103 !dmu_objset_is_dirty(mos, txg)) { 11104 /* 11105 * Nothing changed on the first pass, therefore this 11106 * TXG is a no-op. Avoid syncing deferred frees, so 11107 * that we can keep this TXG as a no-op. 11108 */ 11109 ASSERT(txg_list_empty(&dp->dp_dirty_datasets, txg)); 11110 ASSERT(txg_list_empty(&dp->dp_dirty_dirs, txg)); 11111 ASSERT(txg_list_empty(&dp->dp_sync_tasks, txg)); 11112 ASSERT(txg_list_empty(&dp->dp_early_sync_tasks, txg)); 11113 break; 11114 } 11115 11116 spa_sync_deferred_frees(spa, tx); 11117 } while (dmu_objset_is_dirty(mos, txg)); 11118 } 11119 11120 /* 11121 * Select up to SPA_SYNC_MIN_VDEVS top-level vdevs to write the uberblock to. 11122 * First take the ones written during this txg, so that the idle ones may stay 11123 * asleep. If there are not enough, top up from special and dedup vdevs, which 11124 * are expected to have no seek penalty. Pools having none of those keep the 11125 * old behavior of topping up from any vdev. 11126 */ 11127 static int 11128 spa_select_uberblock_vdevs(spa_t *spa, vdev_t **svd, uint64_t txg) 11129 { 11130 vdev_t *rvd = spa->spa_root_vdev; 11131 uint64_t children = rvd->vdev_children; 11132 uint64_t c0 = random_in_range(children); 11133 boolean_t tiered = spa_has_special(spa) || spa_has_dedup(spa); 11134 int svdcount = 0; 11135 11136 for (int pass = 0; pass < 3; pass++) { 11137 if (pass == 2 && svdcount > 0 && tiered) 11138 break; 11139 11140 for (uint64_t c = 0; c < children && 11141 svdcount < SPA_SYNC_MIN_VDEVS; c++) { 11142 vdev_t *vd = rvd->vdev_child[(c0 + c) % children]; 11143 boolean_t dup = B_FALSE; 11144 11145 if (vd->vdev_ms_array == 0 || vd->vdev_islog || 11146 !vdev_is_concrete(vd)) 11147 continue; 11148 11149 if (pass == 0 && !txg_list_member( 11150 &spa->spa_vdev_txg_list, vd, TXG_CLEAN(txg))) 11151 continue; 11152 11153 if (pass == 1) { 11154 metaslab_class_t *mc = vd->vdev_mg != NULL ? 11155 vd->vdev_mg->mg_class : NULL; 11156 if (mc != spa_special_class(spa) && 11157 mc != spa_dedup_class(spa)) 11158 continue; 11159 } 11160 11161 for (int i = 0; i < svdcount; i++) 11162 dup |= (svd[i] == vd); 11163 if (dup) 11164 continue; 11165 11166 svd[svdcount++] = vd; 11167 } 11168 } 11169 11170 return (svdcount); 11171 } 11172 11173 /* 11174 * Rewrite the vdev configuration (which includes the uberblock) to 11175 * commit the transaction group. 11176 * 11177 * If there are no dirty vdevs, we sync the uberblock to a few random 11178 * top-level vdevs that are known to be visible in the config cache 11179 * (see spa_vdev_add() for a complete description). If there *are* dirty 11180 * vdevs, sync the uberblock to all vdevs. 11181 */ 11182 static void 11183 spa_sync_rewrite_vdev_config(spa_t *spa, dmu_tx_t *tx) 11184 { 11185 vdev_t *rvd = spa->spa_root_vdev; 11186 uint64_t txg = tx->tx_txg; 11187 11188 for (;;) { 11189 int error = 0; 11190 11191 /* 11192 * We hold SCL_STATE to prevent vdev open/close/etc. 11193 * while we're attempting to write the vdev labels. 11194 */ 11195 spa_config_enter(spa, SCL_STATE, FTAG, RW_READER); 11196 11197 if (list_is_empty(&spa->spa_config_dirty_list)) { 11198 vdev_t *svd[SPA_SYNC_MIN_VDEVS] = { NULL }; 11199 int svdcount = spa_select_uberblock_vdevs(spa, svd, 11200 txg); 11201 11202 error = vdev_config_sync(spa, svd, svdcount, txg); 11203 } else { 11204 error = vdev_config_sync(spa, rvd->vdev_child, 11205 rvd->vdev_children, txg); 11206 } 11207 11208 if (error == 0) 11209 spa->spa_last_synced_guid = rvd->vdev_guid; 11210 11211 spa_config_exit(spa, SCL_STATE, FTAG); 11212 11213 if (error == 0) 11214 break; 11215 zio_suspend(spa, NULL, ZIO_SUSPEND_IOERR); 11216 zio_resume_wait(spa); 11217 } 11218 } 11219 11220 /* 11221 * Sync the specified transaction group. New blocks may be dirtied as 11222 * part of the process, so we iterate until it converges. 11223 */ 11224 void 11225 spa_sync(spa_t *spa, uint64_t txg) 11226 { 11227 vdev_t *vd = NULL; 11228 11229 VERIFY(spa_writeable(spa)); 11230 11231 /* 11232 * Wait for i/os issued in open context that need to complete 11233 * before this txg syncs. 11234 */ 11235 (void) zio_wait(spa->spa_txg_zio[txg & TXG_MASK]); 11236 spa->spa_txg_zio[txg & TXG_MASK] = zio_root(spa, NULL, NULL, 11237 ZIO_FLAG_CANFAIL); 11238 11239 /* 11240 * Now that there can be no more cloning in this transaction group, 11241 * but we are still before issuing frees, we can process pending BRT 11242 * updates. 11243 */ 11244 brt_pending_apply(spa, txg); 11245 11246 spa_sync_time_logger(spa, txg, B_FALSE); 11247 11248 /* 11249 * Lock out configuration changes. 11250 */ 11251 spa_config_enter(spa, SCL_CONFIG, FTAG, RW_READER); 11252 11253 spa->spa_syncing_txg = txg; 11254 spa->spa_sync_pass = 0; 11255 11256 /* 11257 * If there are any pending vdev state changes, convert them 11258 * into config changes that go out with this transaction group. 11259 */ 11260 spa_config_enter(spa, SCL_STATE, FTAG, RW_READER); 11261 while ((vd = list_head(&spa->spa_state_dirty_list)) != NULL) { 11262 /* Avoid holding the write lock unless actually necessary */ 11263 if (vd->vdev_aux == NULL) { 11264 vdev_state_clean(vd); 11265 vdev_config_dirty(vd); 11266 continue; 11267 } 11268 /* 11269 * We need the write lock here because, for aux vdevs, 11270 * calling vdev_config_dirty() modifies sav_config. 11271 * This is ugly and will become unnecessary when we 11272 * eliminate the aux vdev wart by integrating all vdevs 11273 * into the root vdev tree. 11274 */ 11275 spa_config_exit(spa, SCL_CONFIG | SCL_STATE, FTAG); 11276 spa_config_enter(spa, SCL_CONFIG | SCL_STATE, FTAG, RW_WRITER); 11277 while ((vd = list_head(&spa->spa_state_dirty_list)) != NULL) { 11278 vdev_state_clean(vd); 11279 vdev_config_dirty(vd); 11280 } 11281 spa_config_exit(spa, SCL_CONFIG | SCL_STATE, FTAG); 11282 spa_config_enter(spa, SCL_CONFIG | SCL_STATE, FTAG, RW_READER); 11283 } 11284 spa_config_exit(spa, SCL_STATE, FTAG); 11285 11286 dsl_pool_t *dp = spa->spa_dsl_pool; 11287 dmu_tx_t *tx = dmu_tx_create_assigned(dp, txg); 11288 11289 spa->spa_sync_starttime = getlrtime(); 11290 11291 taskq_cancel_id(system_delay_taskq, spa->spa_deadman_tqid, B_TRUE); 11292 spa->spa_deadman_tqid = taskq_dispatch_delay(system_delay_taskq, 11293 spa_deadman, spa, TQ_SLEEP, ddi_get_lbolt() + 11294 NSEC_TO_TICK(spa->spa_deadman_synctime)); 11295 11296 /* 11297 * If we are upgrading to SPA_VERSION_RAIDZ_DEFLATE this txg, 11298 * set spa_deflate if we have no raid-z vdevs. 11299 */ 11300 if (spa->spa_ubsync.ub_version < SPA_VERSION_RAIDZ_DEFLATE && 11301 spa->spa_uberblock.ub_version >= SPA_VERSION_RAIDZ_DEFLATE) { 11302 vdev_t *rvd = spa->spa_root_vdev; 11303 11304 int i; 11305 for (i = 0; i < rvd->vdev_children; i++) { 11306 vd = rvd->vdev_child[i]; 11307 if (vd->vdev_deflate_ratio != SPA_MINBLOCKSIZE) 11308 break; 11309 } 11310 if (i == rvd->vdev_children) { 11311 spa->spa_deflate = TRUE; 11312 VERIFY0(zap_add(spa->spa_meta_objset, 11313 DMU_POOL_DIRECTORY_OBJECT, DMU_POOL_DEFLATE, 11314 sizeof (uint64_t), 1, &spa->spa_deflate, tx)); 11315 } 11316 } 11317 11318 spa_sync_adjust_vdev_max_queue_depth(spa); 11319 11320 spa_sync_condense_indirect(spa, tx); 11321 11322 spa_sync_iterate_to_convergence(spa, tx); 11323 11324 #ifdef ZFS_DEBUG 11325 if (!list_is_empty(&spa->spa_config_dirty_list)) { 11326 /* 11327 * Make sure that the number of ZAPs for all the vdevs matches 11328 * the number of ZAPs in the per-vdev ZAP list. This only gets 11329 * called if the config is dirty; otherwise there may be 11330 * outstanding AVZ operations that weren't completed in 11331 * spa_sync_config_object. 11332 */ 11333 uint64_t all_vdev_zap_entry_count; 11334 ASSERT0(zap_count(spa->spa_meta_objset, 11335 spa->spa_all_vdev_zaps, &all_vdev_zap_entry_count)); 11336 ASSERT3U(vdev_count_verify_zaps(spa->spa_root_vdev), ==, 11337 all_vdev_zap_entry_count); 11338 } 11339 #endif 11340 11341 if (spa->spa_vdev_removal != NULL) { 11342 ASSERT0(spa->spa_vdev_removal->svr_bytes_done[txg & TXG_MASK]); 11343 } 11344 11345 for (vd = txg_list_head(&spa->spa_vdev_txg_list, TXG_CLEAN(txg)); vd; 11346 vd = txg_list_next(&spa->spa_vdev_txg_list, vd, TXG_CLEAN(txg))) 11347 vdev_sync_dispatch(vd, txg); 11348 11349 spa_sync_rewrite_vdev_config(spa, tx); 11350 dmu_tx_commit(tx); 11351 11352 taskq_cancel_id(system_delay_taskq, spa->spa_deadman_tqid, B_TRUE); 11353 spa->spa_deadman_tqid = 0; 11354 11355 /* 11356 * Clear the dirty config list. 11357 */ 11358 while ((vd = list_head(&spa->spa_config_dirty_list)) != NULL) 11359 vdev_config_clean(vd); 11360 11361 /* 11362 * Now that the new config has synced transactionally, 11363 * let it become visible to the config cache. 11364 */ 11365 if (spa->spa_config_syncing != NULL) { 11366 spa_config_set(spa, spa->spa_config_syncing); 11367 spa->spa_config_txg = txg; 11368 spa->spa_config_syncing = NULL; 11369 } 11370 11371 dsl_pool_sync_done(dp, txg); 11372 11373 while ((vd = txg_list_remove(&spa->spa_vdev_txg_list, TXG_CLEAN(txg))) 11374 != NULL) 11375 vdev_sync_done(vd, txg); 11376 11377 metaslab_class_evict_old(spa->spa_normal_class, txg); 11378 metaslab_class_evict_old(spa->spa_log_class, txg); 11379 /* Embedded log classes have only one metaslab per vdev. */ 11380 metaslab_class_evict_old(spa->spa_special_class, txg); 11381 metaslab_class_evict_old(spa->spa_dedup_class, txg); 11382 11383 spa_sync_close_syncing_log_sm(spa); 11384 11385 spa_update_dspace(spa); 11386 spa_log_sm_stats_update(spa); 11387 11388 if (spa_get_autotrim(spa) == SPA_AUTOTRIM_ON) 11389 vdev_autotrim_kick(spa); 11390 11391 /* 11392 * It had better be the case that we didn't dirty anything 11393 * since vdev_config_sync(). 11394 */ 11395 ASSERT(txg_list_empty(&dp->dp_dirty_datasets, txg)); 11396 ASSERT(txg_list_empty(&dp->dp_dirty_dirs, txg)); 11397 ASSERT(txg_list_empty(&spa->spa_vdev_txg_list, txg)); 11398 11399 while (zfs_pause_spa_sync) 11400 delay(1); 11401 11402 spa->spa_sync_pass = 0; 11403 11404 /* 11405 * Update the last synced uberblock here. We want to do this at 11406 * the end of spa_sync() so that consumers of spa_last_synced_txg() 11407 * will be guaranteed that all the processing associated with 11408 * that txg has been completed. 11409 */ 11410 spa->spa_ubsync = spa->spa_uberblock; 11411 spa_config_exit(spa, SCL_CONFIG, FTAG); 11412 11413 /* 11414 * An activity that ended in this txg is only over for a reader of 11415 * the pool now that the txg is on disk, so let the waiters look 11416 * again (see spa_activity_in_progress()). 11417 */ 11418 spa_notify_waiters(spa); 11419 11420 spa_handle_ignored_writes(spa); 11421 11422 /* 11423 * If any async tasks have been requested, kick them off. 11424 */ 11425 spa_async_dispatch(spa); 11426 } 11427 11428 /* 11429 * Sync all pools. We don't want to hold the namespace lock across these 11430 * operations, so we take a reference on the spa_t and drop the lock during the 11431 * sync. 11432 */ 11433 void 11434 spa_sync_allpools(void) 11435 { 11436 spa_t *spa = NULL; 11437 spa_namespace_enter(FTAG); 11438 while ((spa = spa_next(spa)) != NULL) { 11439 if (spa_state(spa) != POOL_STATE_ACTIVE || 11440 !spa_writeable(spa) || spa_suspended(spa)) 11441 continue; 11442 spa_open_ref(spa, FTAG); 11443 spa_namespace_exit(FTAG); 11444 txg_wait_synced(spa_get_dsl(spa), 0); 11445 spa_namespace_enter(FTAG); 11446 spa_close(spa, FTAG); 11447 } 11448 spa_namespace_exit(FTAG); 11449 } 11450 11451 taskq_t * 11452 spa_sync_tq_create(spa_t *spa, const char *name) 11453 { 11454 kthread_t **kthreads; 11455 11456 ASSERT0P(spa->spa_sync_tq); 11457 ASSERT3S(spa->spa_alloc_count, <=, boot_ncpus); 11458 11459 /* 11460 * - do not allow more allocators than cpus. 11461 * - there may be more cpus than allocators. 11462 * - do not allow more sync taskq threads than allocators or cpus. 11463 */ 11464 int nthreads = spa->spa_alloc_count; 11465 spa->spa_syncthreads = kmem_zalloc(sizeof (spa_syncthread_info_t) * 11466 nthreads, KM_SLEEP); 11467 11468 spa->spa_sync_tq = taskq_create_synced(name, nthreads, minclsyspri, 11469 nthreads, INT_MAX, TASKQ_PREPOPULATE, &kthreads); 11470 VERIFY(spa->spa_sync_tq != NULL); 11471 VERIFY(kthreads != NULL); 11472 11473 spa_syncthread_info_t *ti = spa->spa_syncthreads; 11474 for (int i = 0; i < nthreads; i++, ti++) { 11475 ti->sti_thread = kthreads[i]; 11476 ti->sti_allocator = i; 11477 } 11478 11479 kmem_free(kthreads, sizeof (*kthreads) * nthreads); 11480 return (spa->spa_sync_tq); 11481 } 11482 11483 void 11484 spa_sync_tq_destroy(spa_t *spa) 11485 { 11486 ASSERT(spa->spa_sync_tq != NULL); 11487 11488 taskq_wait(spa->spa_sync_tq); 11489 taskq_destroy(spa->spa_sync_tq); 11490 kmem_free(spa->spa_syncthreads, 11491 sizeof (spa_syncthread_info_t) * spa->spa_alloc_count); 11492 spa->spa_sync_tq = NULL; 11493 } 11494 11495 uint_t 11496 spa_acq_allocator(spa_t *spa) 11497 { 11498 int i; 11499 11500 if (spa->spa_alloc_count == 1) 11501 return (0); 11502 11503 mutex_enter(&spa->spa_allocs_use->sau_lock); 11504 uint_t r = spa->spa_allocs_use->sau_rotor; 11505 do { 11506 if (++r == spa->spa_alloc_count) 11507 r = 0; 11508 } while (spa->spa_allocs_use->sau_inuse[r]); 11509 spa->spa_allocs_use->sau_inuse[r] = B_TRUE; 11510 spa->spa_allocs_use->sau_rotor = r; 11511 mutex_exit(&spa->spa_allocs_use->sau_lock); 11512 11513 spa_syncthread_info_t *ti = spa->spa_syncthreads; 11514 for (i = 0; i < spa->spa_alloc_count; i++, ti++) { 11515 if (ti->sti_thread == curthread) { 11516 ti->sti_allocator = r; 11517 break; 11518 } 11519 } 11520 ASSERT3S(i, <, spa->spa_alloc_count); 11521 return (r); 11522 } 11523 11524 void 11525 spa_rel_allocator(spa_t *spa, uint_t allocator) 11526 { 11527 if (spa->spa_alloc_count > 1) 11528 spa->spa_allocs_use->sau_inuse[allocator] = B_FALSE; 11529 } 11530 11531 void 11532 spa_select_allocator(zio_t *zio) 11533 { 11534 zbookmark_phys_t *bm = &zio->io_bookmark; 11535 spa_t *spa = zio->io_spa; 11536 11537 ASSERT(zio->io_type == ZIO_TYPE_WRITE); 11538 11539 /* 11540 * A gang block (for example) may have inherited its parent's 11541 * allocator, in which case there is nothing further to do here. 11542 */ 11543 if (ZIO_HAS_ALLOCATOR(zio)) 11544 return; 11545 11546 ASSERT(spa != NULL); 11547 ASSERT(bm != NULL); 11548 11549 /* 11550 * First try to use an allocator assigned to the syncthread, and set 11551 * the corresponding write issue taskq for the allocator. 11552 * Note, we must have an open pool to do this. 11553 */ 11554 if (spa->spa_sync_tq != NULL) { 11555 spa_syncthread_info_t *ti = spa->spa_syncthreads; 11556 for (int i = 0; i < spa->spa_alloc_count; i++, ti++) { 11557 if (ti->sti_thread == curthread) { 11558 zio->io_allocator = ti->sti_allocator; 11559 return; 11560 } 11561 } 11562 } 11563 11564 /* 11565 * We want to try to use as many allocators as possible to help improve 11566 * performance, but we also want logically adjacent IOs to be physically 11567 * adjacent to improve sequential read performance. We chunk each object 11568 * into 2^20 block regions, and then hash based on the objset, object, 11569 * level, and region to accomplish both of these goals. 11570 */ 11571 uint64_t hv = cityhash4(bm->zb_objset, bm->zb_object, bm->zb_level, 11572 bm->zb_blkid >> 20); 11573 11574 zio->io_allocator = (uint_t)hv % spa->spa_alloc_count; 11575 } 11576 11577 /* 11578 * ========================================================================== 11579 * Miscellaneous routines 11580 * ========================================================================== 11581 */ 11582 11583 /* 11584 * Remove all pools in the system. 11585 */ 11586 void 11587 spa_evict_all(void) 11588 { 11589 spa_t *spa; 11590 11591 /* 11592 * Remove all cached state. All pools should be closed now, 11593 * so every spa in the AVL tree should be unreferenced. 11594 */ 11595 spa_namespace_enter(FTAG); 11596 while ((spa = spa_next(NULL)) != NULL) { 11597 /* 11598 * Stop async tasks. The async thread may need to detach 11599 * a device that's been replaced, which requires grabbing 11600 * spa_namespace_lock, so we must drop it here. 11601 */ 11602 spa_open_ref(spa, FTAG); 11603 spa_namespace_exit(FTAG); 11604 spa_async_suspend(spa); 11605 spa_namespace_enter(FTAG); 11606 spa_close(spa, FTAG); 11607 11608 if (spa->spa_state != POOL_STATE_UNINITIALIZED) { 11609 spa_unload(spa); 11610 spa_deactivate(spa); 11611 } 11612 spa_remove(spa); 11613 } 11614 spa_namespace_exit(FTAG); 11615 } 11616 11617 vdev_t * 11618 spa_lookup_by_guid(spa_t *spa, uint64_t guid, boolean_t aux) 11619 { 11620 vdev_t *vd; 11621 int i; 11622 11623 if ((vd = vdev_lookup_by_guid(spa->spa_root_vdev, guid)) != NULL) 11624 return (vd); 11625 11626 if (aux) { 11627 for (i = 0; i < spa->spa_l2cache.sav_count; i++) { 11628 vd = spa->spa_l2cache.sav_vdevs[i]; 11629 if (vd->vdev_guid == guid) 11630 return (vd); 11631 } 11632 11633 for (i = 0; i < spa->spa_spares.sav_count; i++) { 11634 vd = spa->spa_spares.sav_vdevs[i]; 11635 if (vd->vdev_guid == guid) 11636 return (vd); 11637 } 11638 } 11639 11640 return (NULL); 11641 } 11642 11643 void 11644 spa_upgrade(spa_t *spa, uint64_t version) 11645 { 11646 ASSERT(spa_writeable(spa)); 11647 11648 spa_config_enter(spa, SCL_ALL, FTAG, RW_WRITER); 11649 11650 /* 11651 * This should only be called for a non-faulted pool, and since a 11652 * future version would result in an unopenable pool, this shouldn't be 11653 * possible. 11654 */ 11655 ASSERT(SPA_VERSION_IS_SUPPORTED(spa->spa_uberblock.ub_version)); 11656 ASSERT3U(version, >=, spa->spa_uberblock.ub_version); 11657 11658 spa->spa_uberblock.ub_version = version; 11659 vdev_config_dirty(spa->spa_root_vdev); 11660 11661 spa_config_exit(spa, SCL_ALL, FTAG); 11662 11663 txg_wait_synced(spa_get_dsl(spa), 0); 11664 } 11665 11666 static boolean_t 11667 spa_has_aux_vdev(spa_t *spa, uint64_t guid, spa_aux_vdev_t *sav) 11668 { 11669 (void) spa; 11670 int i; 11671 uint64_t vdev_guid; 11672 11673 for (i = 0; i < sav->sav_count; i++) 11674 if (sav->sav_vdevs[i]->vdev_guid == guid) 11675 return (B_TRUE); 11676 11677 for (i = 0; i < sav->sav_npending; i++) { 11678 if (nvlist_lookup_uint64(sav->sav_pending[i], ZPOOL_CONFIG_GUID, 11679 &vdev_guid) == 0 && vdev_guid == guid) 11680 return (B_TRUE); 11681 } 11682 11683 return (B_FALSE); 11684 } 11685 11686 boolean_t 11687 spa_has_l2cache(spa_t *spa, uint64_t guid) 11688 { 11689 return (spa_has_aux_vdev(spa, guid, &spa->spa_l2cache)); 11690 } 11691 11692 boolean_t 11693 spa_has_spare(spa_t *spa, uint64_t guid) 11694 { 11695 return (spa_has_aux_vdev(spa, guid, &spa->spa_spares)); 11696 } 11697 11698 /* 11699 * Check if a pool has an active shared spare device. 11700 * Note: reference count of an active spare is 2, as a spare and as a replace 11701 */ 11702 static boolean_t 11703 spa_has_active_shared_spare(spa_t *spa) 11704 { 11705 int i, refcnt; 11706 uint64_t pool; 11707 spa_aux_vdev_t *sav = &spa->spa_spares; 11708 11709 for (i = 0; i < sav->sav_count; i++) { 11710 if (spa_spare_exists(sav->sav_vdevs[i]->vdev_guid, &pool, 11711 &refcnt) && pool != 0ULL && pool == spa_guid(spa) && 11712 refcnt > 2) 11713 return (B_TRUE); 11714 } 11715 11716 return (B_FALSE); 11717 } 11718 11719 uint64_t 11720 spa_total_metaslabs(spa_t *spa) 11721 { 11722 vdev_t *rvd = spa->spa_root_vdev; 11723 11724 uint64_t m = 0; 11725 for (uint64_t c = 0; c < rvd->vdev_children; c++) { 11726 vdev_t *vd = rvd->vdev_child[c]; 11727 if (!vdev_is_concrete(vd)) 11728 continue; 11729 m += vd->vdev_ms_count; 11730 } 11731 return (m); 11732 } 11733 11734 /* 11735 * Notify any waiting threads that some activity has switched from being in- 11736 * progress to not-in-progress so that the thread can wake up and determine 11737 * whether it is finished waiting. 11738 */ 11739 void 11740 spa_notify_waiters(spa_t *spa) 11741 { 11742 /* 11743 * Acquiring spa_activities_lock here prevents the cv_broadcast from 11744 * happening between the waiting thread's check and cv_wait. 11745 */ 11746 mutex_enter(&spa->spa_activities_lock); 11747 cv_broadcast(&spa->spa_activities_cv); 11748 mutex_exit(&spa->spa_activities_lock); 11749 } 11750 11751 /* 11752 * Notify any waiting threads that the pool is exporting, and then block until 11753 * they are finished using the spa_t. 11754 */ 11755 void 11756 spa_wake_waiters(spa_t *spa) 11757 { 11758 mutex_enter(&spa->spa_activities_lock); 11759 spa->spa_waiters_cancel = B_TRUE; 11760 cv_broadcast(&spa->spa_activities_cv); 11761 while (spa->spa_waiters != 0) 11762 cv_wait(&spa->spa_waiters_cv, &spa->spa_activities_lock); 11763 spa->spa_waiters_cancel = B_FALSE; 11764 mutex_exit(&spa->spa_activities_lock); 11765 } 11766 11767 /* Whether the vdev or any of its descendants are being initialized/trimmed. */ 11768 static boolean_t 11769 spa_vdev_activity_in_progress_impl(vdev_t *vd, zpool_wait_activity_t activity) 11770 { 11771 spa_t *spa = vd->vdev_spa; 11772 11773 ASSERT(spa_config_held(spa, SCL_CONFIG | SCL_STATE, RW_READER)); 11774 ASSERT(MUTEX_HELD(&spa->spa_activities_lock)); 11775 ASSERT(activity == ZPOOL_WAIT_INITIALIZE || 11776 activity == ZPOOL_WAIT_TRIM); 11777 11778 kmutex_t *lock = activity == ZPOOL_WAIT_INITIALIZE ? 11779 &vd->vdev_initialize_lock : &vd->vdev_trim_lock; 11780 11781 mutex_exit(&spa->spa_activities_lock); 11782 mutex_enter(lock); 11783 mutex_enter(&spa->spa_activities_lock); 11784 11785 /* 11786 * A thread that has finished still has to sync out the new state 11787 * before it exits, and until it does the vdev cannot be initialized 11788 * or trimmed again. Wait for the thread itself, not just the state, 11789 * so that a command issued after the wait returns does not fail with 11790 * EBUSY. 11791 */ 11792 boolean_t in_progress = (activity == ZPOOL_WAIT_INITIALIZE) ? 11793 (vd->vdev_initialize_state == VDEV_INITIALIZE_ACTIVE || 11794 vd->vdev_initialize_thread != NULL) : 11795 (vd->vdev_trim_state == VDEV_TRIM_ACTIVE || 11796 vd->vdev_trim_thread != NULL); 11797 mutex_exit(lock); 11798 11799 if (in_progress) 11800 return (B_TRUE); 11801 11802 for (int i = 0; i < vd->vdev_children; i++) { 11803 if (spa_vdev_activity_in_progress_impl(vd->vdev_child[i], 11804 activity)) 11805 return (B_TRUE); 11806 } 11807 11808 return (B_FALSE); 11809 } 11810 11811 /* 11812 * If use_guid is true, this checks whether the vdev specified by guid is 11813 * being initialized/trimmed. Otherwise, it checks whether any vdev in the pool 11814 * is being initialized/trimmed. The caller must hold the config lock and 11815 * spa_activities_lock. 11816 */ 11817 static int 11818 spa_vdev_activity_in_progress(spa_t *spa, boolean_t use_guid, uint64_t guid, 11819 zpool_wait_activity_t activity, boolean_t *in_progress) 11820 { 11821 mutex_exit(&spa->spa_activities_lock); 11822 spa_config_enter(spa, SCL_CONFIG | SCL_STATE, FTAG, RW_READER); 11823 mutex_enter(&spa->spa_activities_lock); 11824 11825 vdev_t *vd; 11826 if (use_guid) { 11827 vd = spa_lookup_by_guid(spa, guid, B_FALSE); 11828 if (vd == NULL || !vd->vdev_ops->vdev_op_leaf) { 11829 spa_config_exit(spa, SCL_CONFIG | SCL_STATE, FTAG); 11830 return (EINVAL); 11831 } 11832 } else { 11833 vd = spa->spa_root_vdev; 11834 } 11835 11836 *in_progress = spa_vdev_activity_in_progress_impl(vd, activity); 11837 11838 spa_config_exit(spa, SCL_CONFIG | SCL_STATE, FTAG); 11839 return (0); 11840 } 11841 11842 /* 11843 * Locking for waiting threads 11844 * --------------------------- 11845 * 11846 * Waiting threads need a way to check whether a given activity is in progress, 11847 * and then, if it is, wait for it to complete. Each activity will have some 11848 * in-memory representation of the relevant on-disk state which can be used to 11849 * determine whether or not the activity is in progress. The in-memory state and 11850 * the locking used to protect it will be different for each activity, and may 11851 * not be suitable for use with a cvar (e.g., some state is protected by the 11852 * config lock). To allow waiting threads to wait without any races, another 11853 * lock, spa_activities_lock, is used. 11854 * 11855 * When the state is checked, both the activity-specific lock (if there is one) 11856 * and spa_activities_lock are held. In some cases, the activity-specific lock 11857 * is acquired explicitly (e.g. the config lock). In others, the locking is 11858 * internal to some check (e.g. bpobj_is_empty). After checking, the waiting 11859 * thread releases the activity-specific lock and, if the activity is in 11860 * progress, then cv_waits using spa_activities_lock. 11861 * 11862 * The waiting thread is woken when another thread, one completing some 11863 * activity, updates the state of the activity and then calls 11864 * spa_notify_waiters, which will cv_broadcast. This 'completing' thread only 11865 * needs to hold its activity-specific lock when updating the state, and this 11866 * lock can (but doesn't have to) be dropped before calling spa_notify_waiters. 11867 * 11868 * Because spa_notify_waiters acquires spa_activities_lock before broadcasting, 11869 * and because it is held when the waiting thread checks the state of the 11870 * activity, it can never be the case that the completing thread both updates 11871 * the activity state and cv_broadcasts in between the waiting thread's check 11872 * and cv_wait. Thus, a waiting thread can never miss a wakeup. 11873 * 11874 * In order to prevent deadlock, when the waiting thread does its check, in some 11875 * cases it will temporarily drop spa_activities_lock in order to acquire the 11876 * activity-specific lock. The order in which spa_activities_lock and the 11877 * activity specific lock are acquired in the waiting thread is determined by 11878 * the order in which they are acquired in the completing thread; if the 11879 * completing thread calls spa_notify_waiters with the activity-specific lock 11880 * held, then the waiting thread must also acquire the activity-specific lock 11881 * first. 11882 */ 11883 11884 static int 11885 spa_activity_in_progress(spa_t *spa, zpool_wait_activity_t activity, 11886 boolean_t use_tag, uint64_t tag, boolean_t *in_progress) 11887 { 11888 int error = 0; 11889 11890 ASSERT(MUTEX_HELD(&spa->spa_activities_lock)); 11891 11892 switch (activity) { 11893 case ZPOOL_WAIT_CKPT_DISCARD: 11894 *in_progress = 11895 (spa_feature_is_active(spa, SPA_FEATURE_POOL_CHECKPOINT) && 11896 zap_contains(spa_meta_objset(spa), 11897 DMU_POOL_DIRECTORY_OBJECT, DMU_POOL_ZPOOL_CHECKPOINT) == 11898 ENOENT); 11899 break; 11900 case ZPOOL_WAIT_FREE: 11901 *in_progress = ((spa_version(spa) >= SPA_VERSION_DEADLISTS && 11902 !bpobj_is_empty(&spa->spa_dsl_pool->dp_free_bpobj)) || 11903 spa_feature_is_active(spa, SPA_FEATURE_ASYNC_DESTROY) || 11904 spa_livelist_delete_check(spa)); 11905 break; 11906 case ZPOOL_WAIT_INITIALIZE: 11907 case ZPOOL_WAIT_TRIM: 11908 error = spa_vdev_activity_in_progress(spa, use_tag, tag, 11909 activity, in_progress); 11910 break; 11911 case ZPOOL_WAIT_REPLACE: 11912 mutex_exit(&spa->spa_activities_lock); 11913 spa_config_enter(spa, SCL_CONFIG | SCL_STATE, FTAG, RW_READER); 11914 mutex_enter(&spa->spa_activities_lock); 11915 11916 *in_progress = vdev_replace_in_progress(spa->spa_root_vdev); 11917 spa_config_exit(spa, SCL_CONFIG | SCL_STATE, FTAG); 11918 break; 11919 case ZPOOL_WAIT_REMOVE: 11920 *in_progress = (spa->spa_removing_phys.sr_state == 11921 DSS_SCANNING); 11922 break; 11923 case ZPOOL_WAIT_RESILVER: 11924 *in_progress = vdev_rebuild_active(spa->spa_root_vdev); 11925 if (*in_progress) 11926 break; 11927 zfs_fallthrough; 11928 case ZPOOL_WAIT_SCRUB: 11929 { 11930 boolean_t scanning, paused, is_scrub, finishing; 11931 dsl_scan_t *scn = spa->spa_dsl_pool->dp_scan; 11932 11933 is_scrub = (scn->scn_phys.scn_func == POOL_SCAN_SCRUB); 11934 scanning = (scn->scn_phys.scn_state == DSS_SCANNING); 11935 paused = dsl_scan_is_paused_scrub(scn); 11936 11937 /* 11938 * dsl_scan_done() marks the scan finished in syncing 11939 * context, ahead of the config and label writes that the 11940 * same txg carries, so the scan is not over for anyone 11941 * reading the pool until that txg has synced. Keep 11942 * reporting it as in progress until then, the way the 11943 * initialize and trim waits cover the whole operation. 11944 */ 11945 finishing = (scn->scn_finished_txg != 0 && 11946 spa_last_synced_txg(spa) < scn->scn_finished_txg); 11947 11948 *in_progress = ((scanning || finishing) && !paused && 11949 is_scrub == (activity == ZPOOL_WAIT_SCRUB)); 11950 break; 11951 } 11952 case ZPOOL_WAIT_RAIDZ_EXPAND: 11953 { 11954 vdev_raidz_expand_t *vre = spa->spa_raidz_expand; 11955 *in_progress = (vre != NULL && vre->vre_state == DSS_SCANNING); 11956 break; 11957 } 11958 case ZPOOL_WAIT_CONDENSE: { 11959 *in_progress = B_FALSE; 11960 spa_condense_stat_t *scns; 11961 11962 for (spa_condense_type_t type = 0; 11963 type < SPA_CONDENSE_TYPES; type++) { 11964 scns = &spa->spa_condense_stats[type]; 11965 if (scns->scns_start_time > 0 && 11966 scns->scns_end_time == 0) { 11967 *in_progress = B_TRUE; 11968 break; 11969 } 11970 } 11971 break; 11972 } 11973 default: 11974 panic("unrecognized value for activity %d", activity); 11975 } 11976 11977 return (error); 11978 } 11979 11980 static int 11981 spa_wait_common(const char *pool, zpool_wait_activity_t activity, 11982 boolean_t use_tag, uint64_t tag, boolean_t *waited) 11983 { 11984 /* 11985 * The tag is used to distinguish between instances of an activity. 11986 * 'initialize' and 'trim' are the only activities that we use this for. 11987 * The other activities can only have a single instance in progress in a 11988 * pool at one time, making the tag unnecessary. 11989 * 11990 * There can be multiple devices being replaced at once, but since they 11991 * all finish once resilvering finishes, we don't bother keeping track 11992 * of them individually, we just wait for them all to finish. 11993 */ 11994 if (use_tag && activity != ZPOOL_WAIT_INITIALIZE && 11995 activity != ZPOOL_WAIT_TRIM) 11996 return (EINVAL); 11997 11998 if (activity < 0 || activity >= ZPOOL_WAIT_NUM_ACTIVITIES) 11999 return (EINVAL); 12000 12001 spa_t *spa; 12002 int error = spa_open(pool, &spa, FTAG); 12003 if (error != 0) 12004 return (error); 12005 12006 /* 12007 * Increment the spa's waiter count so that we can call spa_close and 12008 * still ensure that the spa_t doesn't get freed before this thread is 12009 * finished with it when the pool is exported. We want to call spa_close 12010 * before we start waiting because otherwise the additional ref would 12011 * prevent the pool from being exported or destroyed throughout the 12012 * potentially long wait. 12013 */ 12014 mutex_enter(&spa->spa_activities_lock); 12015 spa->spa_waiters++; 12016 spa_close(spa, FTAG); 12017 12018 *waited = B_FALSE; 12019 for (;;) { 12020 boolean_t in_progress; 12021 error = spa_activity_in_progress(spa, activity, use_tag, tag, 12022 &in_progress); 12023 12024 if (error || !in_progress || spa->spa_waiters_cancel) 12025 break; 12026 12027 *waited = B_TRUE; 12028 12029 if (cv_wait_sig(&spa->spa_activities_cv, 12030 &spa->spa_activities_lock) == 0) { 12031 error = EINTR; 12032 break; 12033 } 12034 } 12035 12036 spa->spa_waiters--; 12037 cv_signal(&spa->spa_waiters_cv); 12038 mutex_exit(&spa->spa_activities_lock); 12039 12040 return (error); 12041 } 12042 12043 /* 12044 * Wait for a particular instance of the specified activity to complete, where 12045 * the instance is identified by 'tag' 12046 */ 12047 int 12048 spa_wait_tag(const char *pool, zpool_wait_activity_t activity, uint64_t tag, 12049 boolean_t *waited) 12050 { 12051 return (spa_wait_common(pool, activity, B_TRUE, tag, waited)); 12052 } 12053 12054 /* 12055 * Wait for all instances of the specified activity complete 12056 */ 12057 int 12058 spa_wait(const char *pool, zpool_wait_activity_t activity, boolean_t *waited) 12059 { 12060 12061 return (spa_wait_common(pool, activity, B_FALSE, 0, waited)); 12062 } 12063 12064 sysevent_t * 12065 spa_event_create(spa_t *spa, vdev_t *vd, nvlist_t *hist_nvl, const char *name) 12066 { 12067 sysevent_t *ev = NULL; 12068 #ifdef _KERNEL 12069 nvlist_t *resource; 12070 12071 resource = zfs_event_create(spa, vd, FM_SYSEVENT_CLASS, name, hist_nvl); 12072 if (resource) { 12073 ev = kmem_alloc(sizeof (sysevent_t), KM_SLEEP); 12074 ev->resource = resource; 12075 } 12076 #else 12077 (void) spa, (void) vd, (void) hist_nvl, (void) name; 12078 #endif 12079 return (ev); 12080 } 12081 12082 void 12083 spa_event_post(sysevent_t *ev) 12084 { 12085 #ifdef _KERNEL 12086 if (ev) { 12087 zfs_zevent_post(ev->resource, NULL, zfs_zevent_post_cb); 12088 kmem_free(ev, sizeof (*ev)); 12089 } 12090 #else 12091 (void) ev; 12092 #endif 12093 } 12094 12095 /* 12096 * Post a zevent corresponding to the given sysevent. The 'name' must be one 12097 * of the event definitions in sys/sysevent/eventdefs.h. The payload will be 12098 * filled in from the spa and (optionally) the vdev. This doesn't do anything 12099 * in the userland libzpool, as we don't want consumers to misinterpret ztest 12100 * or zdb as real changes. 12101 */ 12102 void 12103 spa_event_notify(spa_t *spa, vdev_t *vd, nvlist_t *hist_nvl, const char *name) 12104 { 12105 spa_event_post(spa_event_create(spa, vd, hist_nvl, name)); 12106 } 12107 12108 #ifdef ZFS_DEBUG 12109 /* 12110 * This runs the "debug" condense type, which does nothing, just updates the 12111 * condense counters every second for ten seconds. This exists entirely for 12112 * testing and debugging the condense system itself, which is why it is 12113 * compiled out of production builds. 12114 */ 12115 #define SPA_CONDENSE_DEBUG_STEP (10) 12116 12117 static void 12118 spa_condense_debug_task(void *arg) 12119 { 12120 spa_t *spa = arg; 12121 spa_condense_stat_t *scns = 12122 &spa->spa_condense_stats[SPA_CONDENSE_DEBUG]; 12123 12124 mutex_enter(&spa->spa_condense_stats_lock); 12125 12126 if (spa->spa_condense_debug_tqid == TASKQID_INVALID) { 12127 /* 12128 * Task no longer required, probably cancelled by 12129 * spa_condense_debug_cancel(). Just exit. 12130 */ 12131 mutex_exit(&spa->spa_condense_stats_lock); 12132 return; 12133 } 12134 12135 spa->spa_condense_debug_tqid = TASKQID_INVALID; 12136 12137 /* Move the condense progress along a bit. */ 12138 scns->scns_processed = MIN(scns->scns_total, scns->scns_processed + 12139 (scns->scns_total / SPA_CONDENSE_DEBUG_STEP)); 12140 if (scns->scns_processed == scns->scns_total) { 12141 /* 12142 * Reached the end. Set the end time to "complete" the 12143 * condense, signal waiters, release resources and we're done. 12144 */ 12145 scns->scns_end_time = gethrestime_sec(); 12146 mutex_exit(&spa->spa_condense_stats_lock); 12147 spa_notify_waiters(spa); 12148 spa_close(spa, scns); 12149 return; 12150 } 12151 12152 /* More to do, re-arm the timer for another round. */ 12153 spa->spa_condense_debug_tqid = taskq_dispatch_delay(system_delay_taskq, 12154 spa_condense_debug_task, spa, TQ_SLEEP, 12155 ddi_get_lbolt() + SEC_TO_TICK(1)); 12156 mutex_exit(&spa->spa_condense_stats_lock); 12157 } 12158 12159 void 12160 spa_condense_debug_start(spa_t *spa) 12161 { 12162 uint32_t nitems = 10 + random_in_range(90) * SPA_CONDENSE_DEBUG_STEP; 12163 12164 spa_condense_stat_t *scns = 12165 &spa->spa_condense_stats[SPA_CONDENSE_DEBUG]; 12166 12167 mutex_enter(&spa->spa_condense_stats_lock); 12168 12169 if (scns->scns_start_time == 0 || scns->scns_end_time > 0) { 12170 /* Previous run finished, or no previous run. Start fresh. */ 12171 scns->scns_start_time = gethrestime_sec(); 12172 scns->scns_end_time = 0; 12173 scns->scns_processed = 0; 12174 scns->scns_total = nitems; 12175 } else { 12176 /* In progress, just add some more work. */ 12177 scns->scns_total += nitems; 12178 } 12179 12180 if (spa->spa_condense_debug_tqid == TASKQID_INVALID) { 12181 spa_open_ref(spa, scns); 12182 spa->spa_condense_debug_tqid = taskq_dispatch_delay( 12183 system_delay_taskq, spa_condense_debug_task, spa, TQ_SLEEP, 12184 ddi_get_lbolt() + SEC_TO_TICK(1)); 12185 } 12186 12187 mutex_exit(&spa->spa_condense_stats_lock); 12188 } 12189 12190 void 12191 spa_condense_debug_cancel(spa_t *spa) 12192 { 12193 spa_condense_stat_t *scns = 12194 &spa->spa_condense_stats[SPA_CONDENSE_DEBUG]; 12195 12196 mutex_enter(&spa->spa_condense_stats_lock); 12197 12198 /* "Cancel" by just setting the end time. */ 12199 if (scns->scns_end_time == 0) 12200 scns->scns_end_time = gethrestime_sec(); 12201 12202 if (spa->spa_condense_debug_tqid == TASKQID_INVALID) { 12203 /* No task, so nothing else to do. */ 12204 mutex_exit(&spa->spa_condense_stats_lock); 12205 spa_notify_waiters(spa); 12206 return; 12207 } 12208 12209 /* 12210 * Task is either waiting to run, or running and waiting to take 12211 * spa_condense_stats_lock. Clear the tqid, so if it does run after we 12212 * drop the lock, it will immediately exit. 12213 */ 12214 taskqid_t tqid = spa->spa_condense_debug_tqid; 12215 spa->spa_condense_debug_tqid = TASKQID_INVALID; 12216 12217 mutex_exit(&spa->spa_condense_stats_lock); 12218 12219 /* 12220 * Cancel the task. If its running, wait for it to complete (ie do 12221 * nothing, per above). 12222 */ 12223 taskq_cancel_id(system_delay_taskq, tqid, B_TRUE); 12224 12225 /* 12226 * Task didn't run or aborted, so it never cleaned up. We do it on its 12227 * behalf. 12228 */ 12229 spa_notify_waiters(spa); 12230 spa_close(spa, scns); 12231 } 12232 #endif 12233 12234 /* state manipulation functions */ 12235 EXPORT_SYMBOL(spa_open); 12236 EXPORT_SYMBOL(spa_open_rewind); 12237 EXPORT_SYMBOL(spa_get_stats); 12238 EXPORT_SYMBOL(spa_create); 12239 EXPORT_SYMBOL(spa_import); 12240 EXPORT_SYMBOL(spa_tryimport); 12241 EXPORT_SYMBOL(spa_destroy); 12242 EXPORT_SYMBOL(spa_export); 12243 EXPORT_SYMBOL(spa_reset); 12244 EXPORT_SYMBOL(spa_async_request); 12245 EXPORT_SYMBOL(spa_async_suspend); 12246 EXPORT_SYMBOL(spa_async_resume); 12247 EXPORT_SYMBOL(spa_inject_addref); 12248 EXPORT_SYMBOL(spa_inject_delref); 12249 EXPORT_SYMBOL(spa_scan_stat_init); 12250 EXPORT_SYMBOL(spa_scan_get_stats); 12251 12252 /* device manipulation */ 12253 EXPORT_SYMBOL(spa_vdev_add); 12254 EXPORT_SYMBOL(spa_vdev_attach); 12255 EXPORT_SYMBOL(spa_vdev_detach); 12256 EXPORT_SYMBOL(spa_vdev_setpath); 12257 EXPORT_SYMBOL(spa_vdev_setfru); 12258 EXPORT_SYMBOL(spa_vdev_split_mirror); 12259 12260 /* spare statech is global across all pools) */ 12261 EXPORT_SYMBOL(spa_spare_add); 12262 EXPORT_SYMBOL(spa_spare_remove); 12263 EXPORT_SYMBOL(spa_spare_exists); 12264 EXPORT_SYMBOL(spa_spare_activate); 12265 12266 /* L2ARC statech is global across all pools) */ 12267 EXPORT_SYMBOL(spa_l2cache_add); 12268 EXPORT_SYMBOL(spa_l2cache_remove); 12269 EXPORT_SYMBOL(spa_l2cache_exists); 12270 EXPORT_SYMBOL(spa_l2cache_activate); 12271 EXPORT_SYMBOL(spa_l2cache_drop); 12272 12273 /* scanning */ 12274 EXPORT_SYMBOL(spa_scan); 12275 EXPORT_SYMBOL(spa_scan_range); 12276 EXPORT_SYMBOL(spa_scan_stop); 12277 12278 /* spa syncing */ 12279 EXPORT_SYMBOL(spa_sync); /* only for DMU use */ 12280 EXPORT_SYMBOL(spa_sync_allpools); 12281 12282 /* properties */ 12283 EXPORT_SYMBOL(spa_prop_set); 12284 EXPORT_SYMBOL(spa_prop_get); 12285 EXPORT_SYMBOL(spa_prop_clear_bootfs); 12286 12287 /* asynchronous event notification */ 12288 EXPORT_SYMBOL(spa_event_notify); 12289 12290 ZFS_MODULE_PARAM(zfs_metaslab, metaslab_, preload_pct, UINT, ZMOD_RW, 12291 "Percentage of CPUs to run a metaslab preload taskq"); 12292 12293 ZFS_MODULE_PARAM(zfs_spa, spa_, load_verify_shift, UINT, ZMOD_RW, 12294 "log2 fraction of arc that can be used by inflight I/Os when " 12295 "verifying pool during import"); 12296 12297 ZFS_MODULE_PARAM(zfs_spa, spa_, load_verify_metadata, INT, ZMOD_RW, 12298 "Set to traverse metadata on pool import"); 12299 12300 ZFS_MODULE_PARAM(zfs_spa, spa_, load_verify_data, INT, ZMOD_RW, 12301 "Set to traverse data on pool import"); 12302 12303 ZFS_MODULE_PARAM(zfs_spa, spa_, load_print_vdev_tree, INT, ZMOD_RW, 12304 "Print vdev tree to zfs_dbgmsg during pool import"); 12305 12306 ZFS_MODULE_PARAM(zfs_zio, zio_, taskq_batch_pct, UINT, ZMOD_RW, 12307 "Percentage of CPUs to run an IO worker thread"); 12308 12309 ZFS_MODULE_PARAM(zfs_zio, zio_, taskq_batch_tpq, UINT, ZMOD_RW, 12310 "Number of threads per IO worker taskqueue"); 12311 12312 ZFS_MODULE_PARAM(zfs, zfs_, max_missing_tvds, U64, ZMOD_RW, 12313 "Allow importing pool with up to this number of missing top-level " 12314 "vdevs (in read-only mode)"); 12315 12316 ZFS_MODULE_PARAM(zfs, zfs_, max_missing_tvds_cachefile, U64, ZMOD_RW, 12317 "Allow importing pools with missing top-level vdevs in cache file"); 12318 12319 ZFS_MODULE_PARAM(zfs, zfs_, max_missing_tvds_scan, U64, ZMOD_RW, 12320 "Allow importing pools with missing top-level vdevs during scan"); 12321 12322 ZFS_MODULE_PARAM(zfs_livelist_condense, zfs_livelist_condense_, zthr_pause, INT, 12323 ZMOD_RW, "Set the livelist condense zthr to pause"); 12324 12325 ZFS_MODULE_PARAM(zfs_livelist_condense, zfs_livelist_condense_, sync_pause, INT, 12326 ZMOD_RW, "Set the livelist condense synctask to pause"); 12327 12328 ZFS_MODULE_PARAM(zfs_livelist_condense, zfs_livelist_condense_, sync_cancel, 12329 INT, ZMOD_RW, 12330 "Whether livelist condensing was canceled in the synctask"); 12331 12332 ZFS_MODULE_PARAM(zfs_livelist_condense, zfs_livelist_condense_, zthr_cancel, 12333 INT, ZMOD_RW, 12334 "Whether livelist condensing was canceled in the zthr function"); 12335 12336 ZFS_MODULE_PARAM(zfs_livelist_condense, zfs_livelist_condense_, new_alloc, INT, 12337 ZMOD_RW, 12338 "Whether extra ALLOC blkptrs were added to a livelist entry while it " 12339 "was being condensed"); 12340 12341 ZFS_MODULE_PARAM(zfs_spa, spa_, note_txg_time, UINT, ZMOD_RW, 12342 "How frequently TXG timestamps are stored internally (in seconds)"); 12343 12344 ZFS_MODULE_PARAM(zfs_spa, spa_, flush_txg_time, UINT, ZMOD_RW, 12345 "How frequently the TXG timestamps database should be flushed " 12346 "to disk (in seconds)"); 12347 12348 #ifdef _KERNEL 12349 ZFS_MODULE_VIRTUAL_PARAM_CALL(zfs_zio, zio_, taskq_read, 12350 spa_taskq_read_param_set, spa_taskq_read_param_get, ZMOD_RW, 12351 "Configure IO queues for read IO"); 12352 ZFS_MODULE_VIRTUAL_PARAM_CALL(zfs_zio, zio_, taskq_write, 12353 spa_taskq_write_param_set, spa_taskq_write_param_get, ZMOD_RW, 12354 "Configure IO queues for write IO"); 12355 ZFS_MODULE_VIRTUAL_PARAM_CALL(zfs_zio, zio_, taskq_free, 12356 spa_taskq_free_param_set, spa_taskq_free_param_get, ZMOD_RW, 12357 "Configure IO queues for free IO"); 12358 #endif 12359 12360 ZFS_MODULE_PARAM(zfs_zio, zio_, taskq_write_tpq, UINT, ZMOD_RW, 12361 "Number of CPUs per write issue taskq"); 12362 12363 ZFS_MODULE_PARAM(zfs, zfs_, ccw_retry_interval, INT, ZMOD_RW, 12364 "Configuration cache file write, retry after failure, interval " 12365 "(seconds)"); 12366