xref: /freebsd/sys/contrib/openzfs/module/zfs/zio.c (revision fc6ed8627222a625a700e99cdfcda19654a0c651)
1 // SPDX-License-Identifier: CDDL-1.0
2 /*
3  * This file and its contents are supplied under the terms of the
4  * Common Development and Distribution License ("CDDL"), version 1.0.
5  * You may only use this file in accordance with the terms of version
6  * 1.0 of the CDDL.
7  *
8  * A full copy of the text of the CDDL should have accompanied this
9  * source.  A copy of the CDDL is also available via the Internet at
10  * https://opensource.org/license/CDDL-1.0.
11  */
12 /*
13  * Copyright (c) 2005, 2010, Oracle and/or its affiliates. All rights reserved.
14  * Copyright (c) 2011, 2022 by Delphix. All rights reserved.
15  * Copyright (c) 2011 Nexenta Systems, Inc. All rights reserved.
16  * Copyright (c) 2017, Intel Corporation.
17  * Copyright (c) 2019, 2023, 2024, 2025, Klara, Inc.
18  * Copyright (c) 2019, Allan Jude
19  * Copyright (c) 2021, Datto, Inc.
20  * Copyright (c) 2021, 2024 by George Melikov. All rights reserved.
21  */
22 
23 #include <sys/sysmacros.h>
24 #include <sys/zfs_context.h>
25 #include <sys/fm/fs/zfs.h>
26 #include <sys/spa.h>
27 #include <sys/txg.h>
28 #include <sys/spa_impl.h>
29 #include <sys/vdev_impl.h>
30 #include <sys/vdev_trim.h>
31 #include <sys/zio_impl.h>
32 #include <sys/zio_compress.h>
33 #include <sys/zio_checksum.h>
34 #include <sys/dmu_objset.h>
35 #include <sys/arc.h>
36 #include <sys/brt.h>
37 #include <sys/ddt.h>
38 #include <sys/blkptr.h>
39 #include <sys/zfeature.h>
40 #include <sys/dsl_scan.h>
41 #include <sys/metaslab_impl.h>
42 #include <sys/time.h>
43 #include <sys/trace_zfs.h>
44 #include <sys/abd.h>
45 #include <sys/dsl_crypt.h>
46 #include <cityhash.h>
47 
48 /*
49  * ==========================================================================
50  * I/O type descriptions
51  * ==========================================================================
52  */
53 const char *const zio_type_name[ZIO_TYPES] = {
54 	/*
55 	 * Note: Linux kernel thread name length is limited
56 	 * so these names will differ from upstream open zfs.
57 	 */
58 	"z_null", "z_rd", "z_wr", "z_fr", "z_cl", "z_flush", "z_trim"
59 };
60 
61 int zio_dva_throttle_enabled = B_TRUE;
62 static int zio_deadman_log_all = B_FALSE;
63 
64 /*
65  * ==========================================================================
66  * I/O kmem caches
67  * ==========================================================================
68  */
69 static kmem_cache_t *zio_cache;
70 static kmem_cache_t *zio_link_cache;
71 kmem_cache_t *zio_buf_cache[SPA_MAXBLOCKSIZE >> SPA_MINBLOCKSHIFT];
72 kmem_cache_t *zio_data_buf_cache[SPA_MAXBLOCKSIZE >> SPA_MINBLOCKSHIFT];
73 #if defined(ZFS_DEBUG) && !defined(_KERNEL)
74 static uint64_t zio_buf_cache_allocs[SPA_MAXBLOCKSIZE >> SPA_MINBLOCKSHIFT];
75 static uint64_t zio_buf_cache_frees[SPA_MAXBLOCKSIZE >> SPA_MINBLOCKSHIFT];
76 #endif
77 
78 /* Mark IOs as "slow" if they take longer than 30 seconds */
79 static uint_t zio_slow_io_ms = (30 * MILLISEC);
80 
81 #define	COMPARE_META_LEVEL	0x80000000ul
82 /*
83  * The following actions directly effect the spa's sync-to-convergence logic.
84  * The values below define the sync pass when we start performing the action.
85  * Care should be taken when changing these values as they directly impact
86  * spa_sync() performance. Tuning these values may introduce subtle performance
87  * pathologies and should only be done in the context of performance analysis.
88  * These tunables will eventually be removed and replaced with #defines once
89  * enough analysis has been done to determine optimal values.
90  *
91  * The 'zfs_sync_pass_deferred_free' pass must be greater than 1 to ensure that
92  * regular blocks are not deferred.
93  *
94  * Starting in sync pass 8 (zfs_sync_pass_dont_compress), we disable
95  * compression (including of metadata).  In practice, we don't have this
96  * many sync passes, so this has no effect.
97  *
98  * The original intent was that disabling compression would help the sync
99  * passes to converge. However, in practice disabling compression increases
100  * the average number of sync passes, because when we turn compression off, a
101  * lot of block's size will change and thus we have to re-allocate (not
102  * overwrite) them. It also increases the number of 128KB allocations (e.g.
103  * for indirect blocks and spacemaps) because these will not be compressed.
104  * The 128K allocations are especially detrimental to performance on highly
105  * fragmented systems, which may have very few free segments of this size,
106  * and may need to load new metaslabs to satisfy 128K allocations.
107  */
108 
109 /* defer frees starting in this pass */
110 uint_t zfs_sync_pass_deferred_free = 2;
111 
112 /* don't compress starting in this pass */
113 static uint_t zfs_sync_pass_dont_compress = 8;
114 
115 /* rewrite new bps starting in this pass */
116 static uint_t zfs_sync_pass_rewrite = 2;
117 
118 /*
119  * An allocating zio is one that either currently has the DVA allocate
120  * stage set or will have it later in its lifetime.
121  */
122 #define	IO_IS_ALLOCATING(zio) ((zio)->io_orig_pipeline & ZIO_STAGE_DVA_ALLOCATE)
123 
124 /*
125  * Enable smaller cores by excluding metadata
126  * allocations as well.
127  */
128 int zio_exclude_metadata = 0;
129 static int zio_requeue_io_start_cut_in_line = 1;
130 
131 #ifdef ZFS_DEBUG
132 static const int zio_buf_debug_limit = 16384;
133 #else
134 static const int zio_buf_debug_limit = 0;
135 #endif
136 
137 typedef struct zio_stats {
138 	kstat_named_t ziostat_total_allocations;
139 	kstat_named_t ziostat_alloc_class_fallbacks;
140 	kstat_named_t ziostat_gang_writes;
141 	kstat_named_t ziostat_gang_multilevel;
142 } zio_stats_t;
143 
144 static zio_stats_t zio_stats = {
145 	{ "total_allocations",	KSTAT_DATA_UINT64 },
146 	{ "alloc_class_fallbacks",	KSTAT_DATA_UINT64 },
147 	{ "gang_writes",	KSTAT_DATA_UINT64 },
148 	{ "gang_multilevel",	KSTAT_DATA_UINT64 },
149 };
150 
151 struct {
152 	wmsum_t ziostat_total_allocations;
153 	wmsum_t ziostat_alloc_class_fallbacks;
154 	wmsum_t ziostat_gang_writes;
155 	wmsum_t ziostat_gang_multilevel;
156 } ziostat_sums;
157 
158 #define	ZIOSTAT_BUMP(stat)	wmsum_add(&ziostat_sums.stat, 1);
159 
160 static kstat_t *zio_ksp;
161 
162 static inline void __zio_execute(zio_t *zio);
163 
164 static void zio_taskq_dispatch(zio_t *, zio_taskq_type_t, boolean_t);
165 static void zio_batch_join(zio_batch_t *, zio_t *);
166 
167 static int
168 zio_kstats_update(kstat_t *ksp, int rw)
169 {
170 	zio_stats_t *zs = ksp->ks_data;
171 	if (rw == KSTAT_WRITE)
172 		return (EACCES);
173 
174 	zs->ziostat_total_allocations.value.ui64 =
175 	    wmsum_value(&ziostat_sums.ziostat_total_allocations);
176 	zs->ziostat_alloc_class_fallbacks.value.ui64 =
177 	    wmsum_value(&ziostat_sums.ziostat_alloc_class_fallbacks);
178 	zs->ziostat_gang_writes.value.ui64 =
179 	    wmsum_value(&ziostat_sums.ziostat_gang_writes);
180 	zs->ziostat_gang_multilevel.value.ui64 =
181 	    wmsum_value(&ziostat_sums.ziostat_gang_multilevel);
182 	return (0);
183 }
184 
185 void
186 zio_init(void)
187 {
188 	size_t c;
189 
190 	zio_cache = kmem_cache_create("zio_cache",
191 	    sizeof (zio_t), 0, NULL, NULL, NULL, NULL, NULL, 0);
192 	zio_link_cache = kmem_cache_create("zio_link_cache",
193 	    sizeof (zio_link_t), 0, NULL, NULL, NULL, NULL, NULL, 0);
194 
195 	wmsum_init(&ziostat_sums.ziostat_total_allocations, 0);
196 	wmsum_init(&ziostat_sums.ziostat_alloc_class_fallbacks, 0);
197 	wmsum_init(&ziostat_sums.ziostat_gang_writes, 0);
198 	wmsum_init(&ziostat_sums.ziostat_gang_multilevel, 0);
199 	zio_ksp = kstat_create("zfs", 0, "zio_stats",
200 	    "misc", KSTAT_TYPE_NAMED, sizeof (zio_stats) /
201 	    sizeof (kstat_named_t), KSTAT_FLAG_VIRTUAL);
202 	if (zio_ksp != NULL) {
203 		zio_ksp->ks_data = &zio_stats;
204 		zio_ksp->ks_update = zio_kstats_update;
205 		kstat_install(zio_ksp);
206 	}
207 
208 	for (c = 0; c < SPA_MAXBLOCKSIZE >> SPA_MINBLOCKSHIFT; c++) {
209 		size_t size = (c + 1) << SPA_MINBLOCKSHIFT;
210 		size_t align, cflags, data_cflags;
211 		char name[32];
212 
213 		/*
214 		 * Create cache for each half-power of 2 size, starting from
215 		 * SPA_MINBLOCKSIZE.  It should give us memory space efficiency
216 		 * of ~7/8, sufficient for transient allocations mostly using
217 		 * these caches.
218 		 */
219 		size_t p2 = size;
220 		while (!ISP2(p2))
221 			p2 &= p2 - 1;
222 		if (!IS_P2ALIGNED(size, p2 / 2))
223 			continue;
224 
225 #ifndef _KERNEL
226 		/*
227 		 * If we are using watchpoints, put each buffer on its own page,
228 		 * to eliminate the performance overhead of trapping to the
229 		 * kernel when modifying a non-watched buffer that shares the
230 		 * page with a watched buffer.
231 		 */
232 		if (arc_watch && !IS_P2ALIGNED(size, PAGESIZE))
233 			continue;
234 #endif
235 
236 		if (IS_P2ALIGNED(size, PAGESIZE))
237 			align = PAGESIZE;
238 		else
239 			align = 1 << (highbit64(size ^ (size - 1)) - 1);
240 
241 		cflags = (zio_exclude_metadata || size > zio_buf_debug_limit) ?
242 		    KMC_NODEBUG : 0;
243 		data_cflags = KMC_NODEBUG;
244 		if (abd_size_alloc_linear(size)) {
245 			cflags |= KMC_RECLAIMABLE;
246 			data_cflags |= KMC_RECLAIMABLE;
247 		}
248 		if (cflags == data_cflags) {
249 			/*
250 			 * Resulting kmem caches would be identical.
251 			 * Save memory by creating only one.
252 			 */
253 			(void) snprintf(name, sizeof (name),
254 			    "zio_buf_comb_%lu", (ulong_t)size);
255 			zio_buf_cache[c] = kmem_cache_create(name, size, align,
256 			    NULL, NULL, NULL, NULL, NULL, cflags);
257 			zio_data_buf_cache[c] = zio_buf_cache[c];
258 			continue;
259 		}
260 		(void) snprintf(name, sizeof (name), "zio_buf_%lu",
261 		    (ulong_t)size);
262 		zio_buf_cache[c] = kmem_cache_create(name, size, align,
263 		    NULL, NULL, NULL, NULL, NULL, cflags);
264 
265 		(void) snprintf(name, sizeof (name), "zio_data_buf_%lu",
266 		    (ulong_t)size);
267 		zio_data_buf_cache[c] = kmem_cache_create(name, size, align,
268 		    NULL, NULL, NULL, NULL, NULL, data_cflags);
269 	}
270 
271 	while (--c != 0) {
272 		ASSERT(zio_buf_cache[c] != NULL);
273 		if (zio_buf_cache[c - 1] == NULL)
274 			zio_buf_cache[c - 1] = zio_buf_cache[c];
275 
276 		ASSERT(zio_data_buf_cache[c] != NULL);
277 		if (zio_data_buf_cache[c - 1] == NULL)
278 			zio_data_buf_cache[c - 1] = zio_data_buf_cache[c];
279 	}
280 
281 	zio_inject_init();
282 
283 	lz4_init();
284 }
285 
286 void
287 zio_fini(void)
288 {
289 	size_t n = SPA_MAXBLOCKSIZE >> SPA_MINBLOCKSHIFT;
290 
291 #if defined(ZFS_DEBUG) && !defined(_KERNEL)
292 	for (size_t i = 0; i < n; i++) {
293 		if (zio_buf_cache_allocs[i] != zio_buf_cache_frees[i])
294 			(void) printf("zio_fini: [%d] %llu != %llu\n",
295 			    (int)((i + 1) << SPA_MINBLOCKSHIFT),
296 			    (long long unsigned)zio_buf_cache_allocs[i],
297 			    (long long unsigned)zio_buf_cache_frees[i]);
298 	}
299 #endif
300 
301 	/*
302 	 * The same kmem cache can show up multiple times in both zio_buf_cache
303 	 * and zio_data_buf_cache. Do a wasteful but trivially correct scan to
304 	 * sort it out.
305 	 */
306 	for (size_t i = 0; i < n; i++) {
307 		kmem_cache_t *cache = zio_buf_cache[i];
308 		if (cache == NULL)
309 			continue;
310 		for (size_t j = i; j < n; j++) {
311 			if (cache == zio_buf_cache[j])
312 				zio_buf_cache[j] = NULL;
313 			if (cache == zio_data_buf_cache[j])
314 				zio_data_buf_cache[j] = NULL;
315 		}
316 		kmem_cache_destroy(cache);
317 	}
318 
319 	for (size_t i = 0; i < n; i++) {
320 		kmem_cache_t *cache = zio_data_buf_cache[i];
321 		if (cache == NULL)
322 			continue;
323 		for (size_t j = i; j < n; j++) {
324 			if (cache == zio_data_buf_cache[j])
325 				zio_data_buf_cache[j] = NULL;
326 		}
327 		kmem_cache_destroy(cache);
328 	}
329 
330 	for (size_t i = 0; i < n; i++) {
331 		VERIFY0P(zio_buf_cache[i]);
332 		VERIFY0P(zio_data_buf_cache[i]);
333 	}
334 
335 	if (zio_ksp != NULL) {
336 		kstat_delete(zio_ksp);
337 		zio_ksp = NULL;
338 	}
339 
340 	wmsum_fini(&ziostat_sums.ziostat_total_allocations);
341 	wmsum_fini(&ziostat_sums.ziostat_alloc_class_fallbacks);
342 	wmsum_fini(&ziostat_sums.ziostat_gang_writes);
343 	wmsum_fini(&ziostat_sums.ziostat_gang_multilevel);
344 
345 	kmem_cache_destroy(zio_link_cache);
346 	kmem_cache_destroy(zio_cache);
347 
348 	zio_inject_fini();
349 
350 	lz4_fini();
351 }
352 
353 /*
354  * ==========================================================================
355  * Allocate and free I/O buffers
356  * ==========================================================================
357  */
358 
359 #if defined(ZFS_DEBUG) && defined(_KERNEL)
360 #define	ZFS_ZIO_BUF_CANARY	1
361 #endif
362 
363 #ifdef ZFS_ZIO_BUF_CANARY
364 static const ulong_t zio_buf_canary = (ulong_t)0xdeadc0dedead210b;
365 
366 /*
367  * Use empty space after the buffer to detect overflows.
368  *
369  * Since zio_init() creates kmem caches only for certain set of buffer sizes,
370  * allocations of different sizes may have some unused space after the data.
371  * Filling part of that space with a known pattern on allocation and checking
372  * it on free should allow us to detect some buffer overflows.
373  */
374 static void
375 zio_buf_put_canary(ulong_t *p, size_t size, kmem_cache_t **cache, size_t c)
376 {
377 	size_t off = P2ROUNDUP(size, sizeof (ulong_t));
378 	ulong_t *canary = p + off / sizeof (ulong_t);
379 	size_t asize = (c + 1) << SPA_MINBLOCKSHIFT;
380 	if (c + 1 < SPA_MAXBLOCKSIZE >> SPA_MINBLOCKSHIFT &&
381 	    cache[c] == cache[c + 1])
382 		asize = (c + 2) << SPA_MINBLOCKSHIFT;
383 	for (; off < asize; canary++, off += sizeof (ulong_t))
384 		*canary = zio_buf_canary;
385 }
386 
387 static void
388 zio_buf_check_canary(ulong_t *p, size_t size, kmem_cache_t **cache, size_t c)
389 {
390 	size_t off = P2ROUNDUP(size, sizeof (ulong_t));
391 	ulong_t *canary = p + off / sizeof (ulong_t);
392 	size_t asize = (c + 1) << SPA_MINBLOCKSHIFT;
393 	if (c + 1 < SPA_MAXBLOCKSIZE >> SPA_MINBLOCKSHIFT &&
394 	    cache[c] == cache[c + 1])
395 		asize = (c + 2) << SPA_MINBLOCKSHIFT;
396 	for (; off < asize; canary++, off += sizeof (ulong_t)) {
397 		if (unlikely(*canary != zio_buf_canary)) {
398 			PANIC("ZIO buffer overflow %p (%zu) + %zu %#lx != %#lx",
399 			    p, size, (canary - p) * sizeof (ulong_t),
400 			    *canary, zio_buf_canary);
401 		}
402 	}
403 }
404 #endif
405 
406 /*
407  * Use zio_buf_alloc to allocate ZFS metadata.  This data will appear in a
408  * crashdump if the kernel panics, so use it judiciously.  Obviously, it's
409  * useful to inspect ZFS metadata, but if possible, we should avoid keeping
410  * excess / transient data in-core during a crashdump.
411  */
412 void *
413 zio_buf_alloc(size_t size)
414 {
415 	size_t c = (size - 1) >> SPA_MINBLOCKSHIFT;
416 
417 	VERIFY3U(c, <, SPA_MAXBLOCKSIZE >> SPA_MINBLOCKSHIFT);
418 #if defined(ZFS_DEBUG) && !defined(_KERNEL)
419 	atomic_add_64(&zio_buf_cache_allocs[c], 1);
420 #endif
421 
422 	void *p = kmem_cache_alloc(zio_buf_cache[c], KM_PUSHPAGE);
423 #ifdef ZFS_ZIO_BUF_CANARY
424 	zio_buf_put_canary(p, size, zio_buf_cache, c);
425 #endif
426 	return (p);
427 }
428 
429 /*
430  * Use zio_data_buf_alloc to allocate data.  The data will not appear in a
431  * crashdump if the kernel panics.  This exists so that we will limit the amount
432  * of ZFS data that shows up in a kernel crashdump.  (Thus reducing the amount
433  * of kernel heap dumped to disk when the kernel panics)
434  */
435 void *
436 zio_data_buf_alloc(size_t size)
437 {
438 	size_t c = (size - 1) >> SPA_MINBLOCKSHIFT;
439 
440 	VERIFY3U(c, <, SPA_MAXBLOCKSIZE >> SPA_MINBLOCKSHIFT);
441 
442 	void *p = kmem_cache_alloc(zio_data_buf_cache[c], KM_PUSHPAGE);
443 #ifdef ZFS_ZIO_BUF_CANARY
444 	zio_buf_put_canary(p, size, zio_data_buf_cache, c);
445 #endif
446 	return (p);
447 }
448 
449 void
450 zio_buf_free(void *buf, size_t size)
451 {
452 	size_t c = (size - 1) >> SPA_MINBLOCKSHIFT;
453 
454 	VERIFY3U(c, <, SPA_MAXBLOCKSIZE >> SPA_MINBLOCKSHIFT);
455 #if defined(ZFS_DEBUG) && !defined(_KERNEL)
456 	atomic_add_64(&zio_buf_cache_frees[c], 1);
457 #endif
458 
459 #ifdef ZFS_ZIO_BUF_CANARY
460 	zio_buf_check_canary(buf, size, zio_buf_cache, c);
461 #endif
462 	kmem_cache_free(zio_buf_cache[c], buf);
463 }
464 
465 void
466 zio_data_buf_free(void *buf, size_t size)
467 {
468 	size_t c = (size - 1) >> SPA_MINBLOCKSHIFT;
469 
470 	VERIFY3U(c, <, SPA_MAXBLOCKSIZE >> SPA_MINBLOCKSHIFT);
471 
472 #ifdef ZFS_ZIO_BUF_CANARY
473 	zio_buf_check_canary(buf, size, zio_data_buf_cache, c);
474 #endif
475 	kmem_cache_free(zio_data_buf_cache[c], buf);
476 }
477 
478 static void
479 zio_abd_free(void *abd, size_t size)
480 {
481 	(void) size;
482 	abd_free((abd_t *)abd);
483 }
484 
485 /*
486  * ==========================================================================
487  * Push and pop I/O transform buffers
488  * ==========================================================================
489  */
490 void
491 zio_push_transform(zio_t *zio, abd_t *data, uint64_t size, uint64_t bufsize,
492     zio_transform_func_t *transform)
493 {
494 	zio_transform_t *zt = kmem_alloc(sizeof (zio_transform_t), KM_SLEEP);
495 
496 	zt->zt_orig_abd = zio->io_abd;
497 	zt->zt_orig_size = zio->io_size;
498 	zt->zt_bufsize = bufsize;
499 	zt->zt_transform = transform;
500 
501 	zt->zt_next = zio->io_transform_stack;
502 	zio->io_transform_stack = zt;
503 
504 	zio->io_abd = data;
505 	zio->io_size = size;
506 }
507 
508 void
509 zio_pop_transforms(zio_t *zio)
510 {
511 	zio_transform_t *zt;
512 
513 	while ((zt = zio->io_transform_stack) != NULL) {
514 		if (zt->zt_transform != NULL)
515 			zt->zt_transform(zio,
516 			    zt->zt_orig_abd, zt->zt_orig_size);
517 
518 		if (zt->zt_bufsize != 0)
519 			abd_free(zio->io_abd);
520 
521 		zio->io_abd = zt->zt_orig_abd;
522 		zio->io_size = zt->zt_orig_size;
523 		zio->io_transform_stack = zt->zt_next;
524 
525 		kmem_free(zt, sizeof (zio_transform_t));
526 	}
527 }
528 
529 /*
530  * ==========================================================================
531  * I/O transform callbacks for subblocks, decompression, and decryption
532  * ==========================================================================
533  */
534 static void
535 zio_subblock(zio_t *zio, abd_t *data, uint64_t size)
536 {
537 	ASSERT(zio->io_size > size);
538 
539 	if (zio->io_type == ZIO_TYPE_READ)
540 		abd_copy(data, zio->io_abd, size);
541 }
542 
543 static void
544 zio_decompress(zio_t *zio, abd_t *data, uint64_t size)
545 {
546 	if (zio->io_error == 0) {
547 		int ret = zio_decompress_data(BP_GET_COMPRESS(zio->io_bp),
548 		    zio->io_abd, data, zio->io_size, size,
549 		    &zio->io_prop.zp_complevel);
550 
551 		if (zio_injection_enabled && ret == 0)
552 			ret = zio_handle_fault_injection(zio, EINVAL);
553 
554 		if (ret != 0)
555 			zio->io_error = SET_ERROR(EIO);
556 	}
557 }
558 
559 static void
560 zio_decrypt(zio_t *zio, abd_t *data, uint64_t size)
561 {
562 	int ret;
563 	void *tmp;
564 	blkptr_t *bp = zio->io_bp;
565 	spa_t *spa = zio->io_spa;
566 	uint64_t dsobj = zio->io_bookmark.zb_objset;
567 	uint64_t lsize = BP_GET_LSIZE(bp);
568 	dmu_object_type_t ot = BP_GET_TYPE(bp);
569 	uint8_t salt[ZIO_DATA_SALT_LEN];
570 	uint8_t iv[ZIO_DATA_IV_LEN];
571 	uint8_t mac[ZIO_DATA_MAC_LEN];
572 	boolean_t no_crypt = B_FALSE;
573 
574 	ASSERT(BP_USES_CRYPT(bp));
575 	ASSERT3U(size, !=, 0);
576 
577 	if (zio->io_error != 0)
578 		return;
579 
580 	/*
581 	 * Verify the cksum of MACs stored in an indirect bp. It will always
582 	 * be possible to verify this since it does not require an encryption
583 	 * key.
584 	 */
585 	if (BP_HAS_INDIRECT_MAC_CKSUM(bp)) {
586 		zio_crypt_decode_mac_bp(bp, mac);
587 
588 		if (BP_GET_COMPRESS(bp) != ZIO_COMPRESS_OFF) {
589 			/*
590 			 * We haven't decompressed the data yet, but
591 			 * zio_crypt_do_indirect_mac_checksum() requires
592 			 * decompressed data to be able to parse out the MACs
593 			 * from the indirect block. We decompress it now and
594 			 * throw away the result after we are finished.
595 			 */
596 			abd_t *abd = abd_alloc_linear(lsize, B_TRUE);
597 			ret = zio_decompress_data(BP_GET_COMPRESS(bp),
598 			    zio->io_abd, abd, zio->io_size, lsize,
599 			    &zio->io_prop.zp_complevel);
600 			if (ret != 0) {
601 				abd_free(abd);
602 				ret = SET_ERROR(EIO);
603 				goto error;
604 			}
605 			ret = zio_crypt_do_indirect_mac_checksum_abd(B_FALSE,
606 			    abd, lsize, BP_SHOULD_BYTESWAP(bp), mac);
607 			abd_free(abd);
608 		} else {
609 			ret = zio_crypt_do_indirect_mac_checksum_abd(B_FALSE,
610 			    zio->io_abd, size, BP_SHOULD_BYTESWAP(bp), mac);
611 		}
612 		abd_copy(data, zio->io_abd, size);
613 
614 		if (zio_injection_enabled && ot != DMU_OT_DNODE && ret == 0) {
615 			ret = zio_handle_decrypt_injection(spa,
616 			    &zio->io_bookmark, ot, ECKSUM);
617 		}
618 		if (ret != 0)
619 			goto error;
620 
621 		return;
622 	}
623 
624 	/*
625 	 * If this is an authenticated block, just check the MAC. It would be
626 	 * nice to separate this out into its own flag, but when this was done,
627 	 * we had run out of bits in what is now zio_flag_t. Future cleanup
628 	 * could make this a flag bit.
629 	 */
630 	if (BP_IS_AUTHENTICATED(bp)) {
631 		if (ot == DMU_OT_OBJSET) {
632 			ret = spa_do_crypt_objset_mac_abd(B_FALSE, spa,
633 			    dsobj, zio->io_abd, size, BP_SHOULD_BYTESWAP(bp));
634 		} else {
635 			zio_crypt_decode_mac_bp(bp, mac);
636 			ret = spa_do_crypt_mac_abd(B_FALSE, spa, dsobj,
637 			    zio->io_abd, size, mac);
638 			if (zio_injection_enabled && ret == 0) {
639 				ret = zio_handle_decrypt_injection(spa,
640 				    &zio->io_bookmark, ot, ECKSUM);
641 			}
642 		}
643 		abd_copy(data, zio->io_abd, size);
644 
645 		if (ret != 0)
646 			goto error;
647 
648 		return;
649 	}
650 
651 	zio_crypt_decode_params_bp(bp, salt, iv);
652 
653 	if (ot == DMU_OT_INTENT_LOG) {
654 		tmp = abd_borrow_buf_copy(zio->io_abd, sizeof (zil_chain_t));
655 		zio_crypt_decode_mac_zil(tmp, mac);
656 		abd_return_buf(zio->io_abd, tmp, sizeof (zil_chain_t));
657 	} else {
658 		zio_crypt_decode_mac_bp(bp, mac);
659 	}
660 
661 	ret = spa_do_crypt_abd(B_FALSE, spa, &zio->io_bookmark, BP_GET_TYPE(bp),
662 	    BP_GET_DEDUP(bp), BP_SHOULD_BYTESWAP(bp), salt, iv, mac, size, data,
663 	    zio->io_abd, &no_crypt);
664 	if (no_crypt)
665 		abd_copy(data, zio->io_abd, size);
666 
667 	if (ret != 0)
668 		goto error;
669 
670 	return;
671 
672 error:
673 	/* the key was found unless this was speculative or a thorough scrub */
674 	ASSERT(ret != EACCES || (zio->io_flags & ZIO_FLAG_SPECULATIVE) ||
675 	    ((zio->io_flags & ZIO_FLAG_SCRUB) &&
676 	    !(zio->io_flags & ZIO_FLAG_RAW)));
677 
678 	/*
679 	 * If there was a decryption / authentication error return EIO as
680 	 * the io_error. If this was not a speculative zio, create an ereport.
681 	 */
682 	if (ret == ECKSUM) {
683 		zio->io_error = SET_ERROR(EIO);
684 		if ((zio->io_flags & ZIO_FLAG_SPECULATIVE) == 0) {
685 			spa_log_error(spa, &zio->io_bookmark,
686 			    BP_GET_PHYSICAL_BIRTH(zio->io_bp));
687 			(void) zfs_ereport_post(FM_EREPORT_ZFS_AUTHENTICATION,
688 			    spa, NULL, &zio->io_bookmark, zio, 0);
689 		}
690 	} else {
691 		zio->io_error = ret;
692 	}
693 }
694 
695 /*
696  * ==========================================================================
697  * I/O parent/child relationships and pipeline interlocks
698  * ==========================================================================
699  */
700 zio_t *
701 zio_walk_parents(zio_t *cio, zio_link_t **zl)
702 {
703 	list_t *pl = &cio->io_parent_list;
704 
705 	*zl = (*zl == NULL) ? list_head(pl) : list_next(pl, *zl);
706 	if (*zl == NULL)
707 		return (NULL);
708 
709 	ASSERT((*zl)->zl_child == cio);
710 	return ((*zl)->zl_parent);
711 }
712 
713 zio_t *
714 zio_walk_children(zio_t *pio, zio_link_t **zl)
715 {
716 	list_t *cl = &pio->io_child_list;
717 
718 	ASSERT(MUTEX_HELD(&pio->io_lock));
719 
720 	*zl = (*zl == NULL) ? list_head(cl) : list_next(cl, *zl);
721 	if (*zl == NULL)
722 		return (NULL);
723 
724 	ASSERT((*zl)->zl_parent == pio);
725 	return ((*zl)->zl_child);
726 }
727 
728 zio_t *
729 zio_unique_parent(zio_t *cio)
730 {
731 	zio_link_t *zl = NULL;
732 	zio_t *pio = zio_walk_parents(cio, &zl);
733 
734 	VERIFY3P(zio_walk_parents(cio, &zl), ==, NULL);
735 	return (pio);
736 }
737 
738 static void
739 zio_add_child_impl(zio_t *pio, zio_t *cio, boolean_t first)
740 {
741 	/*
742 	 * Logical I/Os can have logical, gang, or vdev children.
743 	 * Gang I/Os can have gang or vdev children.
744 	 * Vdev I/Os can only have vdev children.
745 	 * The following ASSERT captures all of these constraints.
746 	 */
747 	ASSERT3S(cio->io_child_type, <=, pio->io_child_type);
748 
749 	/* Parent should not have READY stage if child doesn't have it. */
750 	IMPLY((cio->io_pipeline & ZIO_STAGE_READY) == 0 &&
751 	    (cio->io_child_type != ZIO_CHILD_VDEV),
752 	    (pio->io_pipeline & ZIO_STAGE_READY) == 0);
753 
754 	zio_link_t *zl = kmem_cache_alloc(zio_link_cache, KM_SLEEP);
755 	zl->zl_parent = pio;
756 	zl->zl_child = cio;
757 
758 	mutex_enter(&pio->io_lock);
759 
760 	if (first)
761 		ASSERT(list_is_empty(&cio->io_parent_list));
762 	else
763 		mutex_enter(&cio->io_lock);
764 
765 	ASSERT0(pio->io_state[ZIO_WAIT_DONE]);
766 
767 	uint64_t *countp = pio->io_children[cio->io_child_type];
768 	for (int w = 0; w < ZIO_WAIT_TYPES; w++)
769 		countp[w] += !cio->io_state[w];
770 
771 	list_insert_head(&pio->io_child_list, zl);
772 	list_insert_head(&cio->io_parent_list, zl);
773 
774 	if (!first)
775 		mutex_exit(&cio->io_lock);
776 
777 	mutex_exit(&pio->io_lock);
778 }
779 
780 void
781 zio_add_child(zio_t *pio, zio_t *cio)
782 {
783 	zio_add_child_impl(pio, cio, B_FALSE);
784 }
785 
786 static void
787 zio_add_child_first(zio_t *pio, zio_t *cio)
788 {
789 	zio_add_child_impl(pio, cio, B_TRUE);
790 }
791 
792 static void
793 zio_remove_child(zio_t *pio, zio_t *cio, zio_link_t *zl)
794 {
795 	ASSERT(zl->zl_parent == pio);
796 	ASSERT(zl->zl_child == cio);
797 
798 	mutex_enter(&pio->io_lock);
799 	mutex_enter(&cio->io_lock);
800 
801 	list_remove(&pio->io_child_list, zl);
802 	list_remove(&cio->io_parent_list, zl);
803 
804 	mutex_exit(&cio->io_lock);
805 	mutex_exit(&pio->io_lock);
806 	kmem_cache_free(zio_link_cache, zl);
807 }
808 
809 static boolean_t
810 zio_wait_for_children(zio_t *zio, uint8_t childbits, enum zio_wait_type wait)
811 {
812 	boolean_t waiting = B_FALSE;
813 
814 	mutex_enter(&zio->io_lock);
815 	ASSERT0P(zio->io_stall);
816 	for (int c = 0; c < ZIO_CHILD_TYPES; c++) {
817 		if (!(ZIO_CHILD_BIT_IS_SET(childbits, c)))
818 			continue;
819 
820 		uint64_t *countp = &zio->io_children[c][wait];
821 		if (*countp != 0) {
822 			zio->io_stage >>= 1;
823 			ASSERT3U(zio->io_stage, !=, ZIO_STAGE_OPEN);
824 			zio->io_stall = countp;
825 			waiting = B_TRUE;
826 			break;
827 		}
828 	}
829 	mutex_exit(&zio->io_lock);
830 	return (waiting);
831 }
832 
833 /*
834  * The zios a pipeline stage hands back to zio_execute() to run once the
835  * current one stops, chained through io_exec_next in the order they were
836  * added.
837  */
838 typedef struct zio_next {
839 	zio_t		*zn_list;
840 	zio_t		**zn_tailp;	/* where the next one is appended */
841 } zio_next_t;
842 
843 static inline void
844 zio_next_init(zio_next_t *next)
845 {
846 	next->zn_list = NULL;
847 	next->zn_tailp = &next->zn_list;
848 }
849 
850 __attribute__((always_inline))
851 static inline void
852 zio_notify_parent(zio_t *pio, zio_t *zio, enum zio_wait_type wait,
853     zio_next_t *nextp)
854 {
855 	uint64_t *countp = &pio->io_children[zio->io_child_type][wait];
856 	int *errorp = &pio->io_child_error[zio->io_child_type];
857 
858 	mutex_enter(&pio->io_lock);
859 	if (zio->io_error && !(zio->io_flags & ZIO_FLAG_DONT_PROPAGATE))
860 		*errorp = zio_worst_error(*errorp, zio->io_error);
861 	pio->io_post |= zio->io_post;
862 	ASSERT3U(*countp, >, 0);
863 
864 	(*countp)--;
865 
866 	if (*countp == 0 && pio->io_stall == countp) {
867 		zio_taskq_type_t type =
868 		    pio->io_stage < ZIO_STAGE_VDEV_IO_START ? ZIO_TASKQ_ISSUE :
869 		    ZIO_TASKQ_INTERRUPT;
870 		pio->io_stall = NULL;
871 		mutex_exit(&pio->io_lock);
872 
873 		/*
874 		 * If we can tell the caller to execute this parent next, do
875 		 * so. We do this if the parent's zio type matches the child's
876 		 * type, or if it's a zio_null() with no done callback, and so
877 		 * has no actual work to do. Otherwise dispatch the parent zio
878 		 * in its own taskq.
879 		 *
880 		 * Having the caller execute the parent when possible reduces
881 		 * locking on the zio taskq's, reduces context switch
882 		 * overhead, and has no recursion penalty.  Note that one
883 		 * read from disk typically causes at least 3 zio's: a
884 		 * zio_null(), the logical zio_read(), and then a physical
885 		 * zio.  When the physical ZIO completes, we are able to call
886 		 * zio_done() on all 3 of these zio's from one invocation of
887 		 * zio_execute() by returning the parent back to
888 		 * zio_execute().  Since the parent isn't executed until this
889 		 * thread returns back to zio_execute(), the caller should do
890 		 * so promptly.
891 		 *
892 		 * In other cases, dispatching the parent prevents
893 		 * overflowing the stack when we have deeply nested
894 		 * parent-child relationships, as we do with the "mega zio"
895 		 * of writes for spa_sync(), and the chain of ZIL blocks.
896 		 *
897 		 * More than one parent may become executable at once, and all
898 		 * of them go back to the caller.  It is the caller that keeps
899 		 * one and dispatches the rest, since only it knows what else
900 		 * is already waiting for its thread.
901 		 */
902 		if (nextp != NULL &&
903 		    (pio->io_type == zio->io_type ||
904 		    (pio->io_type == ZIO_TYPE_NULL && !pio->io_done))) {
905 			ASSERT3P(pio->io_exec_next, ==, NULL);
906 			*nextp->zn_tailp = pio;
907 			nextp->zn_tailp = &pio->io_exec_next;
908 		} else {
909 			zio_taskq_dispatch(pio, type, B_FALSE);
910 		}
911 	} else {
912 		mutex_exit(&pio->io_lock);
913 	}
914 }
915 
916 static void
917 zio_inherit_child_errors(zio_t *zio, enum zio_child c)
918 {
919 	if (zio->io_child_error[c] != 0 && zio->io_error == 0)
920 		zio->io_error = zio->io_child_error[c];
921 }
922 
923 int
924 zio_bookmark_compare(const void *x1, const void *x2)
925 {
926 	const zio_t *z1 = x1;
927 	const zio_t *z2 = x2;
928 	const zbookmark_phys_t *zb1 = &z1->io_bookmark;
929 	const zbookmark_phys_t *zb2 = &z2->io_bookmark;
930 
931 	int cmp = TREE_CMP(zb1->zb_objset, zb2->zb_objset);
932 	if (cmp != 0)
933 		return (cmp);
934 
935 	cmp = TREE_CMP(zb1->zb_object, zb2->zb_object);
936 	if (cmp != 0)
937 		return (cmp);
938 
939 	cmp = TREE_CMP(zb1->zb_level, zb2->zb_level);
940 	if (cmp != 0)
941 		return (cmp);
942 
943 	cmp = TREE_CMP(zb1->zb_blkid, zb2->zb_blkid);
944 	if (cmp != 0)
945 		return (cmp);
946 
947 	return (TREE_PCMP(z1, z2));
948 }
949 
950 /*
951  * ==========================================================================
952  * Create the various types of I/O (read, write, free, etc)
953  * ==========================================================================
954  */
955 static zio_t *
956 zio_create(zio_t *pio, spa_t *spa, uint64_t txg, const blkptr_t *bp,
957     abd_t *data, uint64_t lsize, uint64_t psize, zio_done_func_t *done,
958     void *private, zio_type_t type, zio_priority_t priority,
959     zio_flag_t flags, vdev_t *vd, uint64_t offset,
960     const zbookmark_phys_t *zb, enum zio_stage stage,
961     enum zio_stage pipeline)
962 {
963 	zio_t *zio;
964 
965 	IMPLY(type != ZIO_TYPE_TRIM, psize <= SPA_MAXBLOCKSIZE);
966 	ASSERT0(P2PHASE(psize, SPA_MINBLOCKSIZE));
967 	ASSERT0(P2PHASE(offset, SPA_MINBLOCKSIZE));
968 
969 	ASSERT(!vd || spa_config_held(spa, SCL_STATE_ALL, RW_READER));
970 	ASSERT(!bp || !(flags & ZIO_FLAG_CONFIG_WRITER));
971 	ASSERT(vd || stage == ZIO_STAGE_OPEN);
972 
973 	IMPLY(lsize != psize, (flags & ZIO_FLAG_RAW_COMPRESS) != 0);
974 
975 	zio = kmem_cache_alloc(zio_cache, KM_SLEEP);
976 	memset(zio, 0, sizeof (zio_t));
977 
978 	mutex_init(&zio->io_lock, NULL, MUTEX_NOLOCKDEP, NULL);
979 	cv_init(&zio->io_cv, NULL, CV_DEFAULT, NULL);
980 
981 	list_create(&zio->io_parent_list, sizeof (zio_link_t),
982 	    offsetof(zio_link_t, zl_parent_node));
983 	list_create(&zio->io_child_list, sizeof (zio_link_t),
984 	    offsetof(zio_link_t, zl_child_node));
985 	metaslab_trace_init(ZIO_ALLOC_LIST(zio));
986 
987 	if (vd != NULL)
988 		zio->io_child_type = ZIO_CHILD_VDEV;
989 	else if (flags & ZIO_FLAG_GANG_CHILD)
990 		zio->io_child_type = ZIO_CHILD_GANG;
991 	else if (flags & ZIO_FLAG_DDT_CHILD)
992 		zio->io_child_type = ZIO_CHILD_DDT;
993 	else
994 		zio->io_child_type = ZIO_CHILD_LOGICAL;
995 
996 	if (bp != NULL) {
997 		if (type != ZIO_TYPE_WRITE ||
998 		    zio->io_child_type == ZIO_CHILD_DDT) {
999 			zio->io_bp_copy = *bp;
1000 			zio->io_bp = &zio->io_bp_copy;	/* so caller can free */
1001 		} else {
1002 			zio->io_bp = (blkptr_t *)bp;
1003 		}
1004 		zio->io_bp_orig = *bp;
1005 		if (zio->io_child_type == ZIO_CHILD_LOGICAL)
1006 			zio->io_logical = zio;
1007 		if (zio->io_child_type > ZIO_CHILD_GANG && BP_IS_GANG(bp))
1008 			pipeline |= ZIO_GANG_STAGES;
1009 		if (flags & ZIO_FLAG_PREALLOCATED) {
1010 			BP_ZERO_DVAS(zio->io_bp);
1011 			BP_SET_BIRTH(zio->io_bp, 0, 0);
1012 		}
1013 	}
1014 
1015 	zio->io_spa = spa;
1016 	zio->io_txg = txg;
1017 	zio->io_done = done;
1018 	zio->io_private = private;
1019 	zio->io_type = type;
1020 	zio->io_priority = priority;
1021 	zio->io_vd = vd;
1022 	zio->io_offset = offset;
1023 	zio->io_orig_abd = zio->io_abd = data;
1024 	zio->io_orig_size = zio->io_size = psize;
1025 	zio->io_lsize = lsize;
1026 	zio->io_orig_flags = zio->io_flags = flags;
1027 	zio->io_orig_stage = zio->io_stage = stage;
1028 	zio->io_orig_pipeline = zio->io_pipeline = pipeline;
1029 	zio->io_pipeline_trace = ZIO_STAGE_OPEN;
1030 	zio->io_allocator = ZIO_ALLOCATOR_NONE;
1031 
1032 	zio->io_state[ZIO_WAIT_READY] = (stage >= ZIO_STAGE_READY) ||
1033 	    (pipeline & ZIO_STAGE_READY) == 0;
1034 	zio->io_state[ZIO_WAIT_DONE] = (stage >= ZIO_STAGE_DONE);
1035 
1036 	if (zb != NULL)
1037 		zio->io_bookmark = *zb;
1038 
1039 	if (pio != NULL) {
1040 		zio->io_metaslab_class = pio->io_metaslab_class;
1041 		if (zio->io_logical == NULL)
1042 			zio->io_logical = pio->io_logical;
1043 		if (zio->io_child_type == ZIO_CHILD_GANG)
1044 			zio->io_gang_leader = pio->io_gang_leader;
1045 		zio_add_child_first(pio, zio);
1046 	}
1047 
1048 	taskq_init_ent(&zio->io_tqent);
1049 
1050 	return (zio);
1051 }
1052 
1053 void
1054 zio_destroy(zio_t *zio)
1055 {
1056 	ASSERT3P(zio->io_batch, ==, NULL);
1057 	ASSERT3P(zio->io_child_batch, ==, NULL);
1058 	ASSERT3P(zio->io_exec_next, ==, NULL);
1059 	metaslab_trace_fini(ZIO_ALLOC_LIST(zio));
1060 	list_destroy(&zio->io_parent_list);
1061 	list_destroy(&zio->io_child_list);
1062 	mutex_destroy(&zio->io_lock);
1063 	cv_destroy(&zio->io_cv);
1064 	kmem_cache_free(zio_cache, zio);
1065 }
1066 
1067 /*
1068  * ZIO intended to be between others.  Provides synchronization at READY
1069  * and DONE pipeline stages and calls the respective callbacks.
1070  */
1071 zio_t *
1072 zio_null(zio_t *pio, spa_t *spa, vdev_t *vd, zio_done_func_t *done,
1073     void *private, zio_flag_t flags)
1074 {
1075 	zio_t *zio;
1076 
1077 	zio = zio_create(pio, spa, 0, NULL, NULL, 0, 0, done, private,
1078 	    ZIO_TYPE_NULL, ZIO_PRIORITY_NOW, flags, vd, 0, NULL,
1079 	    ZIO_STAGE_OPEN, ZIO_INTERLOCK_PIPELINE);
1080 
1081 	return (zio);
1082 }
1083 
1084 /*
1085  * ZIO intended to be a root of a tree.  Unlike null ZIO does not have a
1086  * READY pipeline stage (is ready on creation), so it should not be used
1087  * as child of any ZIO that may need waiting for grandchildren READY stage
1088  * (any other ZIO type).
1089  */
1090 zio_t *
1091 zio_root(spa_t *spa, zio_done_func_t *done, void *private, zio_flag_t flags)
1092 {
1093 	zio_t *zio;
1094 
1095 	zio = zio_create(NULL, spa, 0, NULL, NULL, 0, 0, done, private,
1096 	    ZIO_TYPE_NULL, ZIO_PRIORITY_NOW, flags, NULL, 0, NULL,
1097 	    ZIO_STAGE_OPEN, ZIO_ROOT_PIPELINE);
1098 
1099 	return (zio);
1100 }
1101 
1102 static int
1103 zfs_blkptr_verify_log(spa_t *spa, const blkptr_t *bp,
1104     enum blk_verify_flag blk_verify, const char *fmt, ...)
1105 {
1106 	va_list adx;
1107 	char buf[256];
1108 
1109 	va_start(adx, fmt);
1110 	(void) vsnprintf(buf, sizeof (buf), fmt, adx);
1111 	va_end(adx);
1112 
1113 	zfs_dbgmsg("bad blkptr at %px: "
1114 	    "DVA[0]=%#llx/%#llx "
1115 	    "DVA[1]=%#llx/%#llx "
1116 	    "DVA[2]=%#llx/%#llx "
1117 	    "prop=%#llx "
1118 	    "prop2=%#llx "
1119 	    "pad=%#llx "
1120 	    "phys_birth=%#llx "
1121 	    "birth=%#llx "
1122 	    "fill=%#llx "
1123 	    "cksum=%#llx/%#llx/%#llx/%#llx",
1124 	    bp,
1125 	    (long long)bp->blk_dva[0].dva_word[0],
1126 	    (long long)bp->blk_dva[0].dva_word[1],
1127 	    (long long)bp->blk_dva[1].dva_word[0],
1128 	    (long long)bp->blk_dva[1].dva_word[1],
1129 	    (long long)bp->blk_dva[2].dva_word[0],
1130 	    (long long)bp->blk_dva[2].dva_word[1],
1131 	    (long long)bp->blk_prop,
1132 	    (long long)bp->blk_prop2,
1133 	    (long long)bp->blk_pad,
1134 	    (long long)BP_GET_RAW_PHYSICAL_BIRTH(bp),
1135 	    (long long)BP_GET_LOGICAL_BIRTH(bp),
1136 	    (long long)bp->blk_fill,
1137 	    (long long)bp->blk_cksum.zc_word[0],
1138 	    (long long)bp->blk_cksum.zc_word[1],
1139 	    (long long)bp->blk_cksum.zc_word[2],
1140 	    (long long)bp->blk_cksum.zc_word[3]);
1141 	switch (blk_verify) {
1142 	case BLK_VERIFY_HALT:
1143 		zfs_panic_recover("%s: %s", spa_name(spa), buf);
1144 		break;
1145 	case BLK_VERIFY_LOG:
1146 		zfs_dbgmsg("%s: %s", spa_name(spa), buf);
1147 		break;
1148 	case BLK_VERIFY_ONLY:
1149 		break;
1150 	}
1151 
1152 	return (1);
1153 }
1154 
1155 /*
1156  * Verify the block pointer fields contain reasonable values.  This means
1157  * it only contains known object types, checksum/compression identifiers,
1158  * block sizes within the maximum allowed limits, valid DVAs, etc.
1159  *
1160  * If everything checks out 0 is returned.  The zfs_blkptr_verify
1161  * argument controls the behavior when an invalid field is detected.
1162  *
1163  * Values for blk_verify_flag:
1164  *   BLK_VERIFY_ONLY: evaluate the block
1165  *   BLK_VERIFY_LOG: evaluate the block and log problems
1166  *   BLK_VERIFY_HALT: call zfs_panic_recover on error
1167  *
1168  * Values for blk_config_flag:
1169  *   BLK_CONFIG_HELD: caller holds SCL_VDEV for writer
1170  *   BLK_CONFIG_NEEDED: caller holds no config lock, SCL_VDEV will be
1171  *   obtained for reader
1172  *   BLK_CONFIG_SKIP: skip checks which require SCL_VDEV, for better
1173  *   performance
1174  */
1175 int
1176 zfs_blkptr_verify(spa_t *spa, const blkptr_t *bp,
1177     enum blk_config_flag blk_config, enum blk_verify_flag blk_verify)
1178 {
1179 	int errors = 0;
1180 
1181 	if (unlikely(!DMU_OT_IS_VALID(BP_GET_TYPE(bp)))) {
1182 		errors += zfs_blkptr_verify_log(spa, bp, blk_verify,
1183 		    "blkptr at %px has invalid TYPE %llu",
1184 		    bp, (longlong_t)BP_GET_TYPE(bp));
1185 	}
1186 	if (unlikely(BP_GET_COMPRESS(bp) >= ZIO_COMPRESS_FUNCTIONS)) {
1187 		errors += zfs_blkptr_verify_log(spa, bp, blk_verify,
1188 		    "blkptr at %px has invalid COMPRESS %llu",
1189 		    bp, (longlong_t)BP_GET_COMPRESS(bp));
1190 	}
1191 	if (unlikely(BP_GET_LSIZE(bp) > SPA_MAXBLOCKSIZE)) {
1192 		errors += zfs_blkptr_verify_log(spa, bp, blk_verify,
1193 		    "blkptr at %px has invalid LSIZE %llu",
1194 		    bp, (longlong_t)BP_GET_LSIZE(bp));
1195 	}
1196 	if (BP_IS_EMBEDDED(bp)) {
1197 		if (unlikely(BPE_GET_ETYPE(bp) >= NUM_BP_EMBEDDED_TYPES)) {
1198 			errors += zfs_blkptr_verify_log(spa, bp, blk_verify,
1199 			    "blkptr at %px has invalid ETYPE %llu",
1200 			    bp, (longlong_t)BPE_GET_ETYPE(bp));
1201 		}
1202 		if (unlikely(BPE_GET_PSIZE(bp) > BPE_PAYLOAD_SIZE)) {
1203 			errors += zfs_blkptr_verify_log(spa, bp, blk_verify,
1204 			    "blkptr at %px has invalid PSIZE %llu",
1205 			    bp, (longlong_t)BPE_GET_PSIZE(bp));
1206 		}
1207 		return (errors ? ECKSUM : 0);
1208 	} else if (BP_IS_HOLE(bp)) {
1209 		/*
1210 		 * Holes are allowed (expected, even) to have no DVAs, no
1211 		 * checksum, and no psize.
1212 		 */
1213 		return (errors ? ECKSUM : 0);
1214 	} else if (unlikely(!DVA_IS_VALID(&bp->blk_dva[0]))) {
1215 		/* Non-hole, non-embedded BPs _must_ have at least one DVA */
1216 		errors += zfs_blkptr_verify_log(spa, bp, blk_verify,
1217 		    "blkptr at %px has no valid DVAs", bp);
1218 	}
1219 	if (unlikely(BP_GET_CHECKSUM(bp) >= ZIO_CHECKSUM_FUNCTIONS)) {
1220 		errors += zfs_blkptr_verify_log(spa, bp, blk_verify,
1221 		    "blkptr at %px has invalid CHECKSUM %llu",
1222 		    bp, (longlong_t)BP_GET_CHECKSUM(bp));
1223 	}
1224 	if (unlikely(BP_GET_PSIZE(bp) > SPA_MAXBLOCKSIZE)) {
1225 		errors += zfs_blkptr_verify_log(spa, bp, blk_verify,
1226 		    "blkptr at %px has invalid PSIZE %llu",
1227 		    bp, (longlong_t)BP_GET_PSIZE(bp));
1228 	}
1229 
1230 	/*
1231 	 * Do not verify individual DVAs if the config is not trusted. This
1232 	 * will be done once the zio is executed in vdev_mirror_map_alloc.
1233 	 */
1234 	if (unlikely(!spa->spa_trust_config))
1235 		return (errors ? ECKSUM : 0);
1236 
1237 	switch (blk_config) {
1238 	case BLK_CONFIG_HELD:
1239 		ASSERT(spa_config_held(spa, SCL_VDEV, RW_WRITER));
1240 		break;
1241 	case BLK_CONFIG_NEEDED:
1242 		spa_config_enter(spa, SCL_VDEV, bp, RW_READER);
1243 		break;
1244 	case BLK_CONFIG_NEEDED_TRY:
1245 		if (!spa_config_tryenter(spa, SCL_VDEV, bp, RW_READER))
1246 			return (EBUSY);
1247 		break;
1248 	case BLK_CONFIG_SKIP:
1249 		return (errors ? ECKSUM : 0);
1250 	default:
1251 		panic("invalid blk_config %u", blk_config);
1252 	}
1253 
1254 	/*
1255 	 * Pool-specific checks.
1256 	 *
1257 	 * Note: it would be nice to verify that the logical birth
1258 	 * and physical birth are not too large.  However,
1259 	 * spa_freeze() allows the birth time of log blocks (and
1260 	 * dmu_sync()-ed blocks that are in the log) to be arbitrarily
1261 	 * large.
1262 	 */
1263 	for (int i = 0; i < BP_GET_NDVAS(bp); i++) {
1264 		const dva_t *dva = &bp->blk_dva[i];
1265 		uint64_t vdevid = DVA_GET_VDEV(dva);
1266 
1267 		if (unlikely(vdevid >= spa->spa_root_vdev->vdev_children)) {
1268 			errors += zfs_blkptr_verify_log(spa, bp, blk_verify,
1269 			    "blkptr at %px DVA %u has invalid VDEV %llu",
1270 			    bp, i, (longlong_t)vdevid);
1271 			continue;
1272 		}
1273 		vdev_t *vd = spa->spa_root_vdev->vdev_child[vdevid];
1274 		if (unlikely(vd == NULL)) {
1275 			errors += zfs_blkptr_verify_log(spa, bp, blk_verify,
1276 			    "blkptr at %px DVA %u has invalid VDEV %llu",
1277 			    bp, i, (longlong_t)vdevid);
1278 			continue;
1279 		}
1280 		if (unlikely(vd->vdev_ops == &vdev_hole_ops)) {
1281 			errors += zfs_blkptr_verify_log(spa, bp, blk_verify,
1282 			    "blkptr at %px DVA %u has hole VDEV %llu",
1283 			    bp, i, (longlong_t)vdevid);
1284 			continue;
1285 		}
1286 		if (vd->vdev_ops == &vdev_missing_ops) {
1287 			/*
1288 			 * "missing" vdevs are valid during import, but we
1289 			 * don't have their detailed info (e.g. asize), so
1290 			 * we can't perform any more checks on them.
1291 			 */
1292 			continue;
1293 		}
1294 		uint64_t offset = DVA_GET_OFFSET(dva);
1295 		uint64_t asize = DVA_GET_ASIZE(dva);
1296 		if (DVA_GET_GANG(dva))
1297 			asize = vdev_gang_header_asize(vd);
1298 		if (unlikely(offset + asize > vd->vdev_asize)) {
1299 			errors += zfs_blkptr_verify_log(spa, bp, blk_verify,
1300 			    "blkptr at %px DVA %u has invalid OFFSET %llu",
1301 			    bp, i, (longlong_t)offset);
1302 		}
1303 	}
1304 	if (blk_config == BLK_CONFIG_NEEDED || blk_config ==
1305 	    BLK_CONFIG_NEEDED_TRY)
1306 		spa_config_exit(spa, SCL_VDEV, bp);
1307 
1308 	return (errors ? ECKSUM : 0);
1309 }
1310 
1311 boolean_t
1312 zfs_dva_valid(spa_t *spa, const dva_t *dva, const blkptr_t *bp)
1313 {
1314 	(void) bp;
1315 	uint64_t vdevid = DVA_GET_VDEV(dva);
1316 
1317 	if (vdevid >= spa->spa_root_vdev->vdev_children)
1318 		return (B_FALSE);
1319 
1320 	vdev_t *vd = spa->spa_root_vdev->vdev_child[vdevid];
1321 	if (vd == NULL)
1322 		return (B_FALSE);
1323 
1324 	if (vd->vdev_ops == &vdev_hole_ops)
1325 		return (B_FALSE);
1326 
1327 	if (vd->vdev_ops == &vdev_missing_ops) {
1328 		return (B_FALSE);
1329 	}
1330 
1331 	uint64_t offset = DVA_GET_OFFSET(dva);
1332 	uint64_t asize = DVA_GET_ASIZE(dva);
1333 
1334 	if (DVA_GET_GANG(dva))
1335 		asize = vdev_gang_header_asize(vd);
1336 	if (offset + asize > vd->vdev_asize)
1337 		return (B_FALSE);
1338 
1339 	return (B_TRUE);
1340 }
1341 
1342 zio_t *
1343 zio_read(zio_t *pio, spa_t *spa, const blkptr_t *bp,
1344     abd_t *data, uint64_t size, zio_done_func_t *done, void *private,
1345     zio_priority_t priority, zio_flag_t flags, const zbookmark_phys_t *zb)
1346 {
1347 	zio_t *zio;
1348 
1349 	zio = zio_create(pio, spa, BP_GET_PHYSICAL_BIRTH(bp), bp,
1350 	    data, size, size, done, private,
1351 	    ZIO_TYPE_READ, priority, flags, NULL, 0, zb,
1352 	    ZIO_STAGE_OPEN, (flags & ZIO_FLAG_DDT_CHILD) ?
1353 	    ZIO_DDT_CHILD_READ_PIPELINE : ZIO_READ_PIPELINE);
1354 
1355 	return (zio);
1356 }
1357 
1358 zio_t *
1359 zio_write(zio_t *pio, spa_t *spa, uint64_t txg, blkptr_t *bp,
1360     abd_t *data, uint64_t lsize, uint64_t psize, const zio_prop_t *zp,
1361     zio_done_func_t *ready, zio_done_func_t *children_ready,
1362     zio_done_func_t *done, void *private, zio_priority_t priority,
1363     zio_flag_t flags, const zbookmark_phys_t *zb)
1364 {
1365 	zio_t *zio;
1366 	enum zio_stage pipeline = zp->zp_direct_write == B_TRUE ?
1367 	    ZIO_DIRECT_WRITE_PIPELINE : (flags & ZIO_FLAG_DDT_CHILD) ?
1368 	    ZIO_DDT_CHILD_WRITE_PIPELINE : ZIO_WRITE_PIPELINE;
1369 
1370 
1371 	zio = zio_create(pio, spa, txg, bp, data, lsize, psize, done, private,
1372 	    ZIO_TYPE_WRITE, priority, flags, NULL, 0, zb,
1373 	    ZIO_STAGE_OPEN, pipeline);
1374 
1375 	zio->io_ready = ready;
1376 	zio->io_children_ready = children_ready;
1377 	zio->io_prop = *zp;
1378 
1379 	/*
1380 	 * Data can be NULL if we are going to call zio_write_override() to
1381 	 * provide the already-allocated BP.  But we may need the data to
1382 	 * verify a dedup hit (if requested).  In this case, don't try to
1383 	 * dedup (just take the already-allocated BP verbatim). Encrypted
1384 	 * dedup blocks need data as well so we also disable dedup in this
1385 	 * case.
1386 	 */
1387 	if (data == NULL &&
1388 	    (zio->io_prop.zp_dedup_verify || zio->io_prop.zp_encrypt)) {
1389 		zio->io_prop.zp_dedup = zio->io_prop.zp_dedup_verify = B_FALSE;
1390 	}
1391 
1392 	return (zio);
1393 }
1394 
1395 zio_t *
1396 zio_rewrite(zio_t *pio, spa_t *spa, uint64_t txg, blkptr_t *bp, abd_t *data,
1397     uint64_t size, zio_done_func_t *done, void *private,
1398     zio_priority_t priority, zio_flag_t flags, zbookmark_phys_t *zb)
1399 {
1400 	zio_t *zio;
1401 
1402 	zio = zio_create(pio, spa, txg, bp, data, size, size, done, private,
1403 	    ZIO_TYPE_WRITE, priority, flags | ZIO_FLAG_IO_REWRITE, NULL, 0, zb,
1404 	    ZIO_STAGE_OPEN, ZIO_REWRITE_PIPELINE);
1405 
1406 	return (zio);
1407 }
1408 
1409 void
1410 zio_write_override(zio_t *zio, blkptr_t *bp, int copies, int gang_copies,
1411     boolean_t nopwrite, boolean_t brtwrite)
1412 {
1413 	ASSERT(zio->io_type == ZIO_TYPE_WRITE);
1414 	ASSERT(zio->io_child_type == ZIO_CHILD_LOGICAL);
1415 	ASSERT(zio->io_stage == ZIO_STAGE_OPEN);
1416 	ASSERT(zio->io_txg == spa_syncing_txg(zio->io_spa));
1417 	ASSERT(!brtwrite || !nopwrite);
1418 
1419 	/*
1420 	 * We must reset the io_prop to match the values that existed
1421 	 * when the bp was first written by dmu_sync() keeping in mind
1422 	 * that nopwrite and dedup are mutually exclusive.
1423 	 */
1424 	zio->io_prop.zp_dedup = nopwrite ? B_FALSE : zio->io_prop.zp_dedup;
1425 	zio->io_prop.zp_nopwrite = nopwrite;
1426 	zio->io_prop.zp_brtwrite = brtwrite;
1427 	zio->io_prop.zp_copies = copies;
1428 	zio->io_prop.zp_gang_copies = gang_copies;
1429 	zio->io_bp_override = bp;
1430 }
1431 
1432 void
1433 zio_free(spa_t *spa, uint64_t txg, const blkptr_t *bp)
1434 {
1435 
1436 	(void) zfs_blkptr_verify(spa, bp, BLK_CONFIG_NEEDED, BLK_VERIFY_HALT);
1437 
1438 	/*
1439 	 * The check for EMBEDDED is a performance optimization.  We
1440 	 * process the free here (by ignoring it) rather than
1441 	 * putting it on the list and then processing it in zio_free_sync().
1442 	 */
1443 	if (BP_IS_EMBEDDED(bp))
1444 		return;
1445 
1446 	/*
1447 	 * Frees that are for the currently-syncing txg, are not going to be
1448 	 * deferred, and which will not need to do a read (i.e. not GANG or
1449 	 * DEDUP), can be processed immediately.  Otherwise, put them on the
1450 	 * in-memory list for later processing.
1451 	 *
1452 	 * Note that we only defer frees after zfs_sync_pass_deferred_free
1453 	 * when the log space map feature is disabled. [see relevant comment
1454 	 * in spa_sync_iterate_to_convergence()]
1455 	 */
1456 	if (BP_IS_GANG(bp) ||
1457 	    BP_GET_DEDUP(bp) ||
1458 	    txg != spa->spa_syncing_txg ||
1459 	    (spa_sync_pass(spa) >= zfs_sync_pass_deferred_free &&
1460 	    !spa_feature_is_active(spa, SPA_FEATURE_LOG_SPACEMAP)) ||
1461 	    brt_maybe_exists(spa, bp)) {
1462 		metaslab_check_free(spa, bp);
1463 		bplist_append(&spa->spa_free_bplist[txg & TXG_MASK], bp);
1464 	} else {
1465 		VERIFY0P(zio_free_sync(NULL, spa, txg, bp, 0));
1466 	}
1467 }
1468 
1469 /*
1470  * To improve performance, this function may return NULL if we were able
1471  * to do the free immediately.  This avoids the cost of creating a zio
1472  * (and linking it to the parent, etc).
1473  */
1474 zio_t *
1475 zio_free_sync(zio_t *pio, spa_t *spa, uint64_t txg, const blkptr_t *bp,
1476     zio_flag_t flags)
1477 {
1478 	ASSERT(!BP_IS_HOLE(bp));
1479 	ASSERT(spa_syncing_txg(spa) == txg);
1480 
1481 	if (BP_IS_EMBEDDED(bp))
1482 		return (NULL);
1483 
1484 	metaslab_check_free(spa, bp);
1485 	arc_freed(spa, bp);
1486 	dsl_scan_freed(spa, bp);
1487 
1488 	if (BP_IS_GANG(bp) ||
1489 	    BP_GET_DEDUP(bp) ||
1490 	    brt_maybe_exists(spa, bp)) {
1491 		/*
1492 		 * GANG, DEDUP and BRT blocks can induce a read (for the gang
1493 		 * block header, the DDT or the BRT), so issue them
1494 		 * asynchronously so that this thread is not tied up.
1495 		 */
1496 		enum zio_stage stage =
1497 		    ZIO_FREE_PIPELINE | ZIO_STAGE_ISSUE_ASYNC;
1498 
1499 		return (zio_create(pio, spa, txg, bp, NULL, BP_GET_PSIZE(bp),
1500 		    BP_GET_PSIZE(bp), NULL, NULL,
1501 		    ZIO_TYPE_FREE, ZIO_PRIORITY_NOW,
1502 		    flags, NULL, 0, NULL, ZIO_STAGE_OPEN, stage));
1503 	} else {
1504 		metaslab_free(spa, bp, txg, B_FALSE);
1505 		return (NULL);
1506 	}
1507 }
1508 
1509 zio_t *
1510 zio_claim(zio_t *pio, spa_t *spa, uint64_t txg, const blkptr_t *bp,
1511     zio_done_func_t *done, void *private, zio_flag_t flags)
1512 {
1513 	zio_t *zio;
1514 
1515 	(void) zfs_blkptr_verify(spa, bp, (flags & ZIO_FLAG_CONFIG_WRITER) ?
1516 	    BLK_CONFIG_HELD : BLK_CONFIG_NEEDED, BLK_VERIFY_HALT);
1517 
1518 	if (BP_IS_EMBEDDED(bp))
1519 		return (zio_null(pio, spa, NULL, NULL, NULL, 0));
1520 
1521 	/*
1522 	 * A claim is an allocation of a specific block.  Claims are needed
1523 	 * to support immediate writes in the intent log.  The issue is that
1524 	 * immediate writes contain committed data, but in a txg that was
1525 	 * *not* committed.  Upon opening the pool after an unclean shutdown,
1526 	 * the intent log claims all blocks that contain immediate write data
1527 	 * so that the SPA knows they're in use.
1528 	 *
1529 	 * All claims *must* be resolved in the first txg -- before the SPA
1530 	 * starts allocating blocks -- so that nothing is allocated twice.
1531 	 * If txg == 0 we just verify that the block is claimable.
1532 	 */
1533 	ASSERT3U(BP_GET_LOGICAL_BIRTH(&spa->spa_uberblock.ub_rootbp), <,
1534 	    spa_min_claim_txg(spa));
1535 	ASSERT(txg == spa_min_claim_txg(spa) || txg == 0);
1536 	ASSERT(!BP_GET_DEDUP(bp) || !spa_writeable(spa));	/* zdb(8) */
1537 
1538 	zio = zio_create(pio, spa, txg, bp, NULL, BP_GET_PSIZE(bp),
1539 	    BP_GET_PSIZE(bp), done, private, ZIO_TYPE_CLAIM, ZIO_PRIORITY_NOW,
1540 	    flags, NULL, 0, NULL, ZIO_STAGE_OPEN, ZIO_CLAIM_PIPELINE);
1541 	ASSERT0(zio->io_queued_timestamp);
1542 
1543 	return (zio);
1544 }
1545 
1546 zio_t *
1547 zio_trim(zio_t *pio, vdev_t *vd, uint64_t offset, uint64_t size,
1548     zio_done_func_t *done, void *private, zio_priority_t priority,
1549     zio_flag_t flags, enum trim_flag trim_flags)
1550 {
1551 	zio_t *zio;
1552 
1553 	ASSERT0(vd->vdev_children);
1554 	ASSERT0(P2PHASE(offset, 1ULL << vd->vdev_ashift));
1555 	ASSERT0(P2PHASE(size, 1ULL << vd->vdev_ashift));
1556 	ASSERT3U(size, !=, 0);
1557 
1558 	zio = zio_create(pio, vd->vdev_spa, 0, NULL, NULL, size, size, done,
1559 	    private, ZIO_TYPE_TRIM, priority, flags | ZIO_FLAG_PHYSICAL,
1560 	    vd, offset, NULL, ZIO_STAGE_OPEN, ZIO_TRIM_PIPELINE);
1561 	zio->io_trim_flags = trim_flags;
1562 
1563 	return (zio);
1564 }
1565 
1566 zio_t *
1567 zio_read_phys(zio_t *pio, vdev_t *vd, uint64_t offset, uint64_t size,
1568     abd_t *data, int checksum, zio_done_func_t *done, void *private,
1569     zio_priority_t priority, zio_flag_t flags, boolean_t labels)
1570 {
1571 	zio_t *zio;
1572 
1573 	ASSERT0(vd->vdev_children);
1574 	ASSERT(!labels || offset + size <= VDEV_LABEL_START_SIZE ||
1575 	    offset >= vd->vdev_psize - VDEV_LABEL_END_SIZE);
1576 	ASSERT3U(offset + size, <=, vd->vdev_psize);
1577 
1578 	zio = zio_create(pio, vd->vdev_spa, 0, NULL, data, size, size, done,
1579 	    private, ZIO_TYPE_READ, priority, flags | ZIO_FLAG_PHYSICAL, vd,
1580 	    offset, NULL, ZIO_STAGE_OPEN, ZIO_READ_PHYS_PIPELINE);
1581 
1582 	zio->io_prop.zp_checksum = checksum;
1583 
1584 	return (zio);
1585 }
1586 
1587 zio_t *
1588 zio_write_phys(zio_t *pio, vdev_t *vd, uint64_t offset, uint64_t size,
1589     abd_t *data, int checksum, zio_done_func_t *done, void *private,
1590     zio_priority_t priority, zio_flag_t flags, boolean_t labels)
1591 {
1592 	zio_t *zio;
1593 
1594 	ASSERT0(vd->vdev_children);
1595 	ASSERT(!labels || offset + size <= VDEV_LABEL_START_SIZE ||
1596 	    offset >= vd->vdev_psize - VDEV_LABEL_END_SIZE);
1597 	ASSERT3U(offset + size, <=, vd->vdev_psize);
1598 
1599 	zio = zio_create(pio, vd->vdev_spa, 0, NULL, data, size, size, done,
1600 	    private, ZIO_TYPE_WRITE, priority, flags | ZIO_FLAG_PHYSICAL, vd,
1601 	    offset, NULL, ZIO_STAGE_OPEN, ZIO_WRITE_PHYS_PIPELINE);
1602 
1603 	zio->io_prop.zp_checksum = checksum;
1604 
1605 	if (zio_checksum_table[checksum].ci_flags & ZCHECKSUM_FLAG_EMBEDDED) {
1606 		/*
1607 		 * zec checksums are necessarily destructive -- they modify
1608 		 * the end of the write buffer to hold the verifier/checksum.
1609 		 * Therefore, we must make a local copy in case the data is
1610 		 * being written to multiple places in parallel.
1611 		 */
1612 		abd_t *wbuf = abd_alloc_sametype(data, size);
1613 		abd_copy(wbuf, data, size);
1614 
1615 		zio_push_transform(zio, wbuf, size, size, NULL);
1616 	}
1617 
1618 	return (zio);
1619 }
1620 
1621 /*
1622  * Create a child I/O to do some work for us.
1623  */
1624 zio_t *
1625 zio_vdev_child_io(zio_t *pio, blkptr_t *bp, vdev_t *vd, uint64_t offset,
1626     abd_t *data, uint64_t size, int type, zio_priority_t priority,
1627     zio_flag_t flags, zio_done_func_t *done, void *private)
1628 {
1629 	enum zio_stage pipeline = ZIO_VDEV_CHILD_PIPELINE;
1630 	zio_t *zio;
1631 
1632 	/*
1633 	 * vdev child I/Os do not propagate their error to the parent.
1634 	 * Therefore, for correct operation the caller *must* check for
1635 	 * and handle the error in the child i/o's done callback.
1636 	 * The only exceptions are i/os that we don't care about
1637 	 * (OPTIONAL or REPAIR).
1638 	 */
1639 	ASSERT((flags & ZIO_FLAG_OPTIONAL) || (flags & ZIO_FLAG_IO_REPAIR) ||
1640 	    done != NULL);
1641 
1642 	if (type == ZIO_TYPE_READ && bp != NULL) {
1643 		/*
1644 		 * If we have the bp, then the child should perform the
1645 		 * checksum and the parent need not.  This pushes error
1646 		 * detection as close to the leaves as possible and
1647 		 * eliminates redundant checksums in the interior nodes.
1648 		 */
1649 		pipeline |= ZIO_STAGE_CHECKSUM_VERIFY;
1650 		pio->io_pipeline &= ~ZIO_STAGE_CHECKSUM_VERIFY;
1651 		/*
1652 		 * We never allow the mirror VDEV to attempt reading from any
1653 		 * additional data copies after the first Direct I/O checksum
1654 		 * verify failure. This is to avoid bad data being written out
1655 		 * through the mirror during self healing. See comment in
1656 		 * vdev_mirror_io_done() for more details.
1657 		 */
1658 		ASSERT0(pio->io_post & ZIO_POST_DIO_CHKSUM_ERR);
1659 	} else if (type == ZIO_TYPE_WRITE &&
1660 	    pio->io_prop.zp_direct_write == B_TRUE) {
1661 		/*
1662 		 * By default we only will verify checksums for Direct I/O
1663 		 * writes for Linux. FreeBSD is able to place user pages under
1664 		 * write protection before issuing them to the ZIO pipeline.
1665 		 *
1666 		 * Checksum validation errors will only be reported through
1667 		 * the top-level VDEV, which is set by this child ZIO.
1668 		 */
1669 		ASSERT3P(bp, !=, NULL);
1670 		ASSERT3U(pio->io_child_type, ==, ZIO_CHILD_LOGICAL);
1671 		pipeline |= ZIO_STAGE_DIO_CHECKSUM_VERIFY;
1672 	}
1673 
1674 	if (vd->vdev_ops->vdev_op_leaf) {
1675 		ASSERT0(vd->vdev_children);
1676 		offset += VDEV_LABEL_START_SIZE;
1677 	}
1678 
1679 	flags |= ZIO_VDEV_CHILD_FLAGS(pio);
1680 
1681 	/*
1682 	 * If we've decided to do a repair, the write is not speculative --
1683 	 * even if the original read was. Rebuild is an exception since we
1684 	 * cannot always ensure its data integrity.
1685 	 */
1686 	if ((flags & ZIO_FLAG_IO_REPAIR) &&
1687 	    pio->io_priority != ZIO_PRIORITY_REBUILD)
1688 		flags &= ~ZIO_FLAG_SPECULATIVE;
1689 
1690 	/*
1691 	 * If we're creating a child I/O that is not associated with a
1692 	 * top-level vdev, then the child zio is not an allocating I/O.
1693 	 * If this is a retried I/O then we ignore it since we will
1694 	 * have already processed the original allocating I/O.
1695 	 */
1696 	if (flags & ZIO_FLAG_ALLOC_THROTTLED &&
1697 	    (vd != vd->vdev_top || (flags & ZIO_FLAG_IO_RETRY)) &&
1698 	    type == ZIO_TYPE_WRITE) {
1699 		ASSERT(pio->io_metaslab_class != NULL);
1700 		ASSERT(pio->io_metaslab_class->mc_alloc_throttle_enabled);
1701 		ASSERT(priority == ZIO_PRIORITY_ASYNC_WRITE);
1702 		ASSERT(!(flags & ZIO_FLAG_IO_REPAIR));
1703 		ASSERT(!(pio->io_flags & ZIO_FLAG_IO_REWRITE) ||
1704 		    pio->io_child_type == ZIO_CHILD_GANG);
1705 
1706 		flags &= ~ZIO_FLAG_ALLOC_THROTTLED;
1707 	}
1708 
1709 	zio = zio_create(pio, pio->io_spa, pio->io_txg, bp, data, size, size,
1710 	    done, private, type, priority, flags, vd, offset, &pio->io_bookmark,
1711 	    ZIO_STAGE_VDEV_IO_START >> 1, pipeline);
1712 	ASSERT3U(zio->io_child_type, ==, ZIO_CHILD_VDEV);
1713 
1714 	if (pio->io_child_batch != NULL) {
1715 		/*
1716 		 * Whatever wakes this child up, all it has left to do are the
1717 		 * few cheap stages of ZIO_VDEV_CHILD_PIPELINE, so it is better
1718 		 * run right there than dispatched.
1719 		 */
1720 		zio->io_flags |= ZIO_FLAG_LIGHTWEIGHT;
1721 
1722 		/*
1723 		 * Only children that come back from the block layer gain
1724 		 * anything from a batch.  Interior ones are dispatched by their
1725 		 * own child's zio_notify_parent() instead, as are distributed
1726 		 * spares, which are leaves that issue children of their own.
1727 		 * The scheduler may change before a queue slot is actually
1728 		 * taken, so vdev_should_queue_io() here only keeps the batch
1729 		 * away from vdevs that can never use it; the binding decision
1730 		 * is vdev_queue_io()'s.
1731 		 */
1732 		if (vd->vdev_ops->vdev_op_leaf &&
1733 		    vd->vdev_ops != &vdev_draid_spare_ops &&
1734 		    !vdev_should_queue_io(zio)) {
1735 			/*
1736 			 * The batch is dispatched to the taskq chosen for
1737 			 * whichever member arrives last, so they all have to
1738 			 * choose the same one.  The flags that steer the choice
1739 			 * are vdev-inherited, and type and priority come from
1740 			 * the parent at every call site.
1741 			 */
1742 			ASSERT3U(zio->io_type, ==, pio->io_type);
1743 			ASSERT3U(zio->io_priority, ==, pio->io_priority);
1744 			zio_batch_join(pio->io_child_batch, zio);
1745 		}
1746 	}
1747 
1748 	return (zio);
1749 }
1750 
1751 zio_t *
1752 zio_vdev_delegated_io(vdev_t *vd, uint64_t offset, abd_t *data, uint64_t size,
1753     zio_type_t type, zio_priority_t priority, zio_flag_t flags,
1754     zio_done_func_t *done, void *private)
1755 {
1756 	zio_t *zio;
1757 
1758 	ASSERT(vd->vdev_ops->vdev_op_leaf);
1759 
1760 	zio = zio_create(NULL, vd->vdev_spa, 0, NULL,
1761 	    data, size, size, done, private, type, priority,
1762 	    flags | ZIO_FLAG_CANFAIL | ZIO_FLAG_DONT_RETRY | ZIO_FLAG_DELEGATED,
1763 	    vd, offset, NULL,
1764 	    ZIO_STAGE_VDEV_IO_START >> 1, ZIO_VDEV_CHILD_PIPELINE);
1765 
1766 	return (zio);
1767 }
1768 
1769 
1770 /*
1771  * Send a flush command to the given vdev. Unlike most zio creation functions,
1772  * the flush zios are issued immediately. You can wait on pio to pause until
1773  * the flushes complete.
1774  */
1775 void
1776 zio_flush(zio_t *pio, vdev_t *vd)
1777 {
1778 	const zio_flag_t flags = ZIO_FLAG_CANFAIL | ZIO_FLAG_DONT_PROPAGATE |
1779 	    ZIO_FLAG_DONT_RETRY;
1780 
1781 	if (vd->vdev_nowritecache)
1782 		return;
1783 
1784 	if (vd->vdev_children == 0) {
1785 		/*
1786 		 * A non-concrete vdev (a hole or indirect vdev left behind
1787 		 * by removing a log or data device) has no leaf device to
1788 		 * flush.  Skip it; issuing a flush to an indirect vdev would
1789 		 * trip the ZIO_TYPE_WRITE assertion in
1790 		 * vdev_indirect_io_start().
1791 		 */
1792 		if (!vdev_is_concrete(vd))
1793 			return;
1794 		zio_nowait(zio_create(pio, vd->vdev_spa, 0, NULL, NULL, 0, 0,
1795 		    NULL, NULL, ZIO_TYPE_FLUSH, ZIO_PRIORITY_NOW, flags, vd, 0,
1796 		    NULL, ZIO_STAGE_OPEN, ZIO_FLUSH_PIPELINE));
1797 	} else {
1798 		for (uint64_t c = 0; c < vd->vdev_children; c++)
1799 			zio_flush(pio, vd->vdev_child[c]);
1800 	}
1801 }
1802 
1803 void
1804 zio_shrink(zio_t *zio, uint64_t size)
1805 {
1806 	ASSERT0P(zio->io_executor);
1807 	ASSERT3U(zio->io_orig_size, ==, zio->io_size);
1808 	ASSERT3U(size, <=, zio->io_size);
1809 
1810 	/*
1811 	 * We don't shrink for raidz because of problems with the
1812 	 * reconstruction when reading back less than the block size.
1813 	 * Note, BP_IS_RAIDZ() assumes no compression.
1814 	 */
1815 	ASSERT(BP_GET_COMPRESS(zio->io_bp) == ZIO_COMPRESS_OFF);
1816 	if (!BP_IS_RAIDZ(zio->io_bp)) {
1817 		/* we are not doing a raw write */
1818 		ASSERT3U(zio->io_size, ==, zio->io_lsize);
1819 		zio->io_orig_size = zio->io_size = zio->io_lsize = size;
1820 	}
1821 }
1822 
1823 /*
1824  * Round provided allocation size up to a value that can be allocated
1825  * by at least some vdev(s) in the pool with minimum or no additional
1826  * padding and without extra space usage on others
1827  */
1828 static uint64_t
1829 zio_roundup_alloc_size(spa_t *spa, uint64_t size)
1830 {
1831 	if (size > spa->spa_min_alloc)
1832 		return (roundup(size, spa->spa_gcd_alloc));
1833 	return (spa->spa_min_alloc);
1834 }
1835 
1836 size_t
1837 zio_get_compression_max_size(enum zio_compress compress, uint64_t gcd_alloc,
1838     uint64_t min_alloc, size_t s_len)
1839 {
1840 	size_t d_len;
1841 
1842 	/* minimum 12.5% must be saved (legacy value, may be changed later) */
1843 	d_len = s_len - (s_len >> 3);
1844 
1845 	/* ZLE can't use exactly d_len bytes, it needs more, so ignore it */
1846 	if (compress == ZIO_COMPRESS_ZLE)
1847 		return (d_len);
1848 
1849 	d_len = d_len - d_len % gcd_alloc;
1850 
1851 	if (d_len < min_alloc)
1852 		return (BPE_PAYLOAD_SIZE);
1853 	return (d_len);
1854 }
1855 
1856 /*
1857  * ==========================================================================
1858  * Prepare to read and write logical blocks
1859  * ==========================================================================
1860  */
1861 
1862 static zio_t *
1863 zio_read_bp_init(zio_t *zio)
1864 {
1865 	blkptr_t *bp = zio->io_bp;
1866 	uint64_t psize =
1867 	    BP_IS_EMBEDDED(bp) ? BPE_GET_PSIZE(bp) : BP_GET_PSIZE(bp);
1868 
1869 	ASSERT3P(zio->io_bp, ==, &zio->io_bp_copy);
1870 
1871 	if (BP_GET_COMPRESS(bp) != ZIO_COMPRESS_OFF &&
1872 	    zio->io_child_type == ZIO_CHILD_LOGICAL &&
1873 	    !(zio->io_flags & ZIO_FLAG_RAW_COMPRESS)) {
1874 		zio_push_transform(zio, abd_alloc_sametype(zio->io_abd, psize),
1875 		    psize, psize, zio_decompress);
1876 	}
1877 
1878 	if (((BP_IS_PROTECTED(bp) && !(zio->io_flags & ZIO_FLAG_RAW_ENCRYPT)) ||
1879 	    BP_HAS_INDIRECT_MAC_CKSUM(bp)) &&
1880 	    zio->io_child_type == ZIO_CHILD_LOGICAL) {
1881 		zio_push_transform(zio, abd_alloc_sametype(zio->io_abd, psize),
1882 		    psize, psize, zio_decrypt);
1883 	}
1884 
1885 	if (BP_IS_EMBEDDED(bp) && BPE_GET_ETYPE(bp) == BP_EMBEDDED_TYPE_DATA) {
1886 		int psize = BPE_GET_PSIZE(bp);
1887 		void *data = abd_borrow_buf(zio->io_abd, psize);
1888 
1889 		zio->io_pipeline = ZIO_INTERLOCK_PIPELINE;
1890 		decode_embedded_bp_compressed(bp, data);
1891 		abd_return_buf_copy(zio->io_abd, data, psize);
1892 	} else {
1893 		ASSERT(!BP_IS_EMBEDDED(bp));
1894 	}
1895 
1896 	if (BP_GET_DEDUP(bp) && zio->io_child_type == ZIO_CHILD_LOGICAL)
1897 		zio->io_pipeline = ZIO_DDT_READ_PIPELINE;
1898 
1899 	return (zio);
1900 }
1901 
1902 static zio_t *
1903 zio_write_bp_init(zio_t *zio)
1904 {
1905 	if (!IO_IS_ALLOCATING(zio))
1906 		return (zio);
1907 
1908 	ASSERT(zio->io_child_type != ZIO_CHILD_DDT);
1909 
1910 	if (zio->io_bp_override) {
1911 		blkptr_t *bp = zio->io_bp;
1912 		zio_prop_t *zp = &zio->io_prop;
1913 
1914 		ASSERT(BP_GET_BIRTH(bp) != zio->io_txg);
1915 
1916 		*bp = *zio->io_bp_override;
1917 		zio->io_pipeline = ZIO_INTERLOCK_PIPELINE;
1918 
1919 		if (zp->zp_brtwrite)
1920 			return (zio);
1921 
1922 		ASSERT(!BP_GET_DEDUP(zio->io_bp_override));
1923 
1924 		if (BP_IS_EMBEDDED(bp))
1925 			return (zio);
1926 
1927 		/*
1928 		 * If we've been overridden and nopwrite is set then
1929 		 * set the flag accordingly to indicate that a nopwrite
1930 		 * has already occurred.
1931 		 */
1932 		if (!BP_IS_HOLE(bp) && zp->zp_nopwrite) {
1933 			ASSERT(!zp->zp_dedup);
1934 			ASSERT3U(BP_GET_CHECKSUM(bp), ==, zp->zp_checksum);
1935 			zio->io_flags |= ZIO_FLAG_NOPWRITE;
1936 			return (zio);
1937 		}
1938 
1939 		ASSERT(!zp->zp_nopwrite);
1940 
1941 		if (BP_IS_HOLE(bp) || !zp->zp_dedup)
1942 			return (zio);
1943 
1944 		ASSERT((zio_checksum_table[zp->zp_checksum].ci_flags &
1945 		    ZCHECKSUM_FLAG_DEDUP) || zp->zp_dedup_verify);
1946 
1947 		if (BP_GET_CHECKSUM(bp) == zp->zp_checksum &&
1948 		    !zp->zp_encrypt) {
1949 			BP_SET_DEDUP(bp, 1);
1950 			zio->io_pipeline |= ZIO_STAGE_DDT_WRITE;
1951 			return (zio);
1952 		}
1953 
1954 		/*
1955 		 * We were unable to handle this as an override bp, treat
1956 		 * it as a regular write I/O.
1957 		 */
1958 		zio->io_bp_override = NULL;
1959 		*bp = zio->io_bp_orig;
1960 		zio->io_pipeline = zio->io_orig_pipeline;
1961 	}
1962 
1963 	return (zio);
1964 }
1965 
1966 static zio_t *
1967 zio_write_compress(zio_t *zio)
1968 {
1969 	spa_t *spa = zio->io_spa;
1970 	zio_prop_t *zp = &zio->io_prop;
1971 	enum zio_compress compress = zp->zp_compress;
1972 	blkptr_t *bp = zio->io_bp;
1973 	uint64_t lsize = zio->io_lsize;
1974 	uint64_t psize = zio->io_size;
1975 	uint32_t pass = 1;
1976 
1977 	/*
1978 	 * If our children haven't all reached the ready stage,
1979 	 * wait for them and then repeat this pipeline stage.
1980 	 */
1981 	if (zio_wait_for_children(zio, ZIO_CHILD_LOGICAL_BIT |
1982 	    ZIO_CHILD_GANG_BIT, ZIO_WAIT_READY)) {
1983 		return (NULL);
1984 	}
1985 
1986 	if (!IO_IS_ALLOCATING(zio))
1987 		return (zio);
1988 
1989 	if (zio->io_children_ready != NULL) {
1990 		/*
1991 		 * Now that all our children are ready, run the callback
1992 		 * associated with this zio in case it wants to modify the
1993 		 * data to be written.
1994 		 */
1995 		ASSERT3U(zp->zp_level, >, 0);
1996 		zio->io_children_ready(zio);
1997 	}
1998 
1999 	ASSERT(zio->io_child_type != ZIO_CHILD_DDT);
2000 	ASSERT0P(zio->io_bp_override);
2001 
2002 	if (!BP_IS_HOLE(bp) && BP_GET_BIRTH(bp) == zio->io_txg) {
2003 		/*
2004 		 * We're rewriting an existing block, which means we're
2005 		 * working on behalf of spa_sync().  For spa_sync() to
2006 		 * converge, it must eventually be the case that we don't
2007 		 * have to allocate new blocks.  But compression changes
2008 		 * the blocksize, which forces a reallocate, and makes
2009 		 * convergence take longer.  Therefore, after the first
2010 		 * few passes, stop compressing to ensure convergence.
2011 		 */
2012 		pass = spa_sync_pass(spa);
2013 
2014 		ASSERT(zio->io_txg == spa_syncing_txg(spa));
2015 		ASSERT(zio->io_child_type == ZIO_CHILD_LOGICAL);
2016 		ASSERT(!BP_GET_DEDUP(bp));
2017 
2018 		if (pass >= zfs_sync_pass_dont_compress)
2019 			compress = ZIO_COMPRESS_OFF;
2020 
2021 		/* Make sure someone doesn't change their mind on overwrites */
2022 		ASSERT(BP_IS_EMBEDDED(bp) || BP_IS_GANG(bp) ||
2023 		    MIN(zp->zp_copies, spa_max_replication(spa))
2024 		    == BP_GET_NDVAS(bp));
2025 	}
2026 
2027 	/* If it's a compressed write that is not raw, compress the buffer. */
2028 	if (compress != ZIO_COMPRESS_OFF &&
2029 	    !(zio->io_flags & ZIO_FLAG_RAW_COMPRESS)) {
2030 		abd_t *cabd = NULL;
2031 		if (abd_cmp_zero(zio->io_abd, lsize) == 0)
2032 			psize = 0;
2033 		else if (compress == ZIO_COMPRESS_EMPTY)
2034 			psize = lsize;
2035 		else
2036 			psize = zio_compress_data(compress, zio->io_abd, &cabd,
2037 			    lsize,
2038 			    zio_get_compression_max_size(compress,
2039 			    spa->spa_gcd_alloc, spa->spa_min_alloc, lsize),
2040 			    zp->zp_complevel);
2041 		if (psize == 0) {
2042 			compress = ZIO_COMPRESS_OFF;
2043 		} else if (psize >= lsize) {
2044 			compress = ZIO_COMPRESS_OFF;
2045 			if (cabd != NULL)
2046 				abd_free(cabd);
2047 		} else if (psize <= BPE_PAYLOAD_SIZE && !zp->zp_encrypt &&
2048 		    zp->zp_level == 0 && !DMU_OT_HAS_FILL(zp->zp_type) &&
2049 		    spa_feature_is_enabled(spa, SPA_FEATURE_EMBEDDED_DATA)) {
2050 			void *cbuf = abd_borrow_buf_copy(cabd, lsize);
2051 			encode_embedded_bp_compressed(bp,
2052 			    cbuf, compress, lsize, psize);
2053 			BPE_SET_ETYPE(bp, BP_EMBEDDED_TYPE_DATA);
2054 			BP_SET_TYPE(bp, zio->io_prop.zp_type);
2055 			BP_SET_LEVEL(bp, zio->io_prop.zp_level);
2056 			abd_return_buf(cabd, cbuf, lsize);
2057 			abd_free(cabd);
2058 			BP_SET_LOGICAL_BIRTH(bp, zio->io_txg);
2059 			zio->io_pipeline = ZIO_INTERLOCK_PIPELINE;
2060 			ASSERT(spa_feature_is_active(spa,
2061 			    SPA_FEATURE_EMBEDDED_DATA));
2062 			return (zio);
2063 		} else {
2064 			/*
2065 			 * Round compressed size up to the minimum allocation
2066 			 * size of the smallest-ashift device, and zero the
2067 			 * tail. This ensures that the compressed size of the
2068 			 * BP (and thus compressratio property) are correct,
2069 			 * in that we charge for the padding used to fill out
2070 			 * the last sector.
2071 			 */
2072 			size_t rounded = (size_t)zio_roundup_alloc_size(spa,
2073 			    psize);
2074 			if (rounded >= lsize) {
2075 				compress = ZIO_COMPRESS_OFF;
2076 				abd_free(cabd);
2077 				psize = lsize;
2078 			} else {
2079 				abd_zero_off(cabd, psize, rounded - psize);
2080 				psize = rounded;
2081 				zio_push_transform(zio, cabd,
2082 				    psize, lsize, NULL);
2083 			}
2084 		}
2085 
2086 		/*
2087 		 * We were unable to handle this as an override bp, treat
2088 		 * it as a regular write I/O.
2089 		 */
2090 		zio->io_bp_override = NULL;
2091 		*bp = zio->io_bp_orig;
2092 		zio->io_pipeline = zio->io_orig_pipeline;
2093 
2094 	} else if ((zio->io_flags & ZIO_FLAG_RAW_ENCRYPT) != 0 &&
2095 	    zp->zp_type == DMU_OT_DNODE) {
2096 		/*
2097 		 * The DMU actually relies on the zio layer's compression
2098 		 * to free metadnode blocks that have had all contained
2099 		 * dnodes freed. As a result, even when doing a raw
2100 		 * receive, we must check whether the block can be compressed
2101 		 * to a hole.
2102 		 */
2103 		if (abd_cmp_zero(zio->io_abd, lsize) == 0) {
2104 			psize = 0;
2105 			compress = ZIO_COMPRESS_OFF;
2106 		} else {
2107 			psize = lsize;
2108 		}
2109 	} else if (zio->io_flags & ZIO_FLAG_RAW_COMPRESS &&
2110 	    !(zio->io_flags & ZIO_FLAG_RAW_ENCRYPT)) {
2111 		/*
2112 		 * If we are raw receiving an encrypted dataset we should not
2113 		 * take this codepath because it will change the on-disk block
2114 		 * and decryption will fail.
2115 		 */
2116 		size_t rounded = MIN((size_t)zio_roundup_alloc_size(spa, psize),
2117 		    lsize);
2118 
2119 		if (rounded != psize) {
2120 			abd_t *cdata = abd_alloc_linear(rounded, B_TRUE);
2121 			abd_zero_off(cdata, psize, rounded - psize);
2122 			abd_copy_off(cdata, zio->io_abd, 0, 0, psize);
2123 			psize = rounded;
2124 			zio_push_transform(zio, cdata,
2125 			    psize, rounded, NULL);
2126 		}
2127 	} else {
2128 		ASSERT3U(psize, !=, 0);
2129 	}
2130 
2131 	/*
2132 	 * The final pass of spa_sync() must be all rewrites, but the first
2133 	 * few passes offer a trade-off: allocating blocks defers convergence,
2134 	 * but newly allocated blocks are sequential, so they can be written
2135 	 * to disk faster.  Therefore, we allow the first few passes of
2136 	 * spa_sync() to allocate new blocks, but force rewrites after that.
2137 	 * There should only be a handful of blocks after pass 1 in any case.
2138 	 */
2139 	if (!BP_IS_HOLE(bp) && BP_GET_BIRTH(bp) == zio->io_txg &&
2140 	    BP_GET_PSIZE(bp) == psize &&
2141 	    pass >= zfs_sync_pass_rewrite) {
2142 		VERIFY3U(psize, !=, 0);
2143 		enum zio_stage gang_stages = zio->io_pipeline & ZIO_GANG_STAGES;
2144 
2145 		zio->io_pipeline = ZIO_REWRITE_PIPELINE | gang_stages;
2146 		zio->io_flags |= ZIO_FLAG_IO_REWRITE;
2147 	} else {
2148 		BP_ZERO(bp);
2149 		zio->io_pipeline = ZIO_WRITE_PIPELINE;
2150 	}
2151 
2152 	if (psize == 0) {
2153 		if (BP_GET_LOGICAL_BIRTH(&zio->io_bp_orig) != 0 &&
2154 		    spa_feature_is_active(spa, SPA_FEATURE_HOLE_BIRTH)) {
2155 			BP_SET_LSIZE(bp, lsize);
2156 			BP_SET_TYPE(bp, zp->zp_type);
2157 			BP_SET_LEVEL(bp, zp->zp_level);
2158 			BP_SET_BIRTH(bp, zio->io_txg, 0);
2159 		}
2160 		zio->io_pipeline = ZIO_INTERLOCK_PIPELINE;
2161 	} else {
2162 		ASSERT(zp->zp_checksum != ZIO_CHECKSUM_GANG_HEADER);
2163 		BP_SET_LSIZE(bp, lsize);
2164 		BP_SET_TYPE(bp, zp->zp_type);
2165 		BP_SET_LEVEL(bp, zp->zp_level);
2166 		BP_SET_PSIZE(bp, psize);
2167 		BP_SET_COMPRESS(bp, compress);
2168 		BP_SET_CHECKSUM(bp, zp->zp_checksum);
2169 		BP_SET_DEDUP(bp, zp->zp_dedup);
2170 		BP_SET_BYTEORDER(bp, ZFS_HOST_BYTEORDER);
2171 		if (zp->zp_dedup) {
2172 			ASSERT(zio->io_child_type == ZIO_CHILD_LOGICAL);
2173 			ASSERT(!(zio->io_flags & ZIO_FLAG_IO_REWRITE));
2174 			ASSERT(!zp->zp_encrypt ||
2175 			    DMU_OT_IS_ENCRYPTED(zp->zp_type));
2176 			zio->io_pipeline = ZIO_DDT_WRITE_PIPELINE;
2177 		}
2178 		if (zp->zp_nopwrite) {
2179 			ASSERT(zio->io_child_type == ZIO_CHILD_LOGICAL);
2180 			ASSERT(!(zio->io_flags & ZIO_FLAG_IO_REWRITE));
2181 			zio->io_pipeline |= ZIO_STAGE_NOP_WRITE;
2182 		}
2183 	}
2184 	return (zio);
2185 }
2186 
2187 static zio_t *
2188 zio_free_bp_init(zio_t *zio)
2189 {
2190 	blkptr_t *bp = zio->io_bp;
2191 
2192 	if (zio->io_child_type == ZIO_CHILD_LOGICAL) {
2193 		if (BP_GET_DEDUP(bp))
2194 			/*
2195 			 * Keep the gang stages zio_create() added: if
2196 			 * zio_ddt_free() falls back to a plain free, they
2197 			 * free the gang members along with the header.
2198 			 */
2199 			zio->io_pipeline |= ZIO_DDT_FREE_PIPELINE;
2200 	}
2201 
2202 	ASSERT3P(zio->io_bp, ==, &zio->io_bp_copy);
2203 
2204 	return (zio);
2205 }
2206 
2207 /*
2208  * ==========================================================================
2209  * Execute the I/O pipeline
2210  * ==========================================================================
2211  */
2212 
2213 static void
2214 zio_taskq_dispatch_func(zio_t *zio, zio_taskq_type_t q, boolean_t cutinline,
2215     task_func_t *func)
2216 {
2217 	spa_t *spa = zio->io_spa;
2218 	zio_type_t t = zio->io_type;
2219 
2220 	/*
2221 	 * If we're a config writer or a probe, the normal issue and
2222 	 * interrupt threads may all be blocked waiting for the config lock.
2223 	 * In this case, select the otherwise-unused taskq for ZIO_TYPE_NULL.
2224 	 */
2225 	if (zio->io_flags & (ZIO_FLAG_CONFIG_WRITER | ZIO_FLAG_PROBE))
2226 		t = ZIO_TYPE_NULL;
2227 
2228 	/*
2229 	 * A similar issue exists for the L2ARC write thread until L2ARC 2.0.
2230 	 */
2231 	if (t == ZIO_TYPE_WRITE && zio->io_vd && zio->io_vd->vdev_aux)
2232 		t = ZIO_TYPE_NULL;
2233 
2234 	/*
2235 	 * If this is a high priority I/O, then use the high priority taskq if
2236 	 * available or cut the line otherwise.
2237 	 */
2238 	if (zio->io_priority == ZIO_PRIORITY_SYNC_WRITE) {
2239 		if (spa->spa_zio_taskq[t][q + 1].stqs_count != 0)
2240 			q++;
2241 		else
2242 			cutinline = B_TRUE;
2243 	}
2244 
2245 	ASSERT3U(q, <, ZIO_TASKQ_TYPES);
2246 
2247 	spa_taskq_dispatch(spa, t, q, func, zio, cutinline);
2248 }
2249 
2250 static void
2251 zio_taskq_dispatch(zio_t *zio, zio_taskq_type_t q, boolean_t cutinline)
2252 {
2253 	zio_taskq_dispatch_func(zio, q, cutinline, zio_execute);
2254 }
2255 
2256 static boolean_t
2257 zio_taskq_member(zio_t *zio, zio_taskq_type_t q)
2258 {
2259 	spa_t *spa = zio->io_spa;
2260 
2261 	taskq_t *tq = taskq_of_curthread();
2262 
2263 	for (zio_type_t t = 0; t < ZIO_TYPES; t++) {
2264 		spa_taskqs_t *tqs = &spa->spa_zio_taskq[t][q];
2265 		uint_t i;
2266 		for (i = 0; i < tqs->stqs_count; i++) {
2267 			if (tqs->stqs_taskq[i] == tq)
2268 				return (B_TRUE);
2269 		}
2270 	}
2271 
2272 	return (B_FALSE);
2273 }
2274 
2275 static zio_t *
2276 zio_issue_async(zio_t *zio)
2277 {
2278 	ASSERT((zio->io_type != ZIO_TYPE_WRITE) || ZIO_HAS_ALLOCATOR(zio));
2279 
2280 	/* Whatever may execute this again, it won't be this thread. */
2281 	zio->io_pipeline &= ~ZIO_STAGE_ISSUE_ASYNC;
2282 
2283 	/*
2284 	 * A zio whose children are not ready yet, such as an indirect block
2285 	 * write, has nothing to do in WRITE_COMPRESS but wait for them, so a
2286 	 * thread dispatched for it would only block.  Do that wait here and let
2287 	 * whoever wakes it up carry on, since that is not this thread anymore.
2288 	 */
2289 	if ((zio->io_pipeline & ZIO_STAGE_WRITE_COMPRESS) &&
2290 	    zio_wait_for_children(zio, ZIO_CHILD_LOGICAL_BIT |
2291 	    ZIO_CHILD_GANG_BIT, ZIO_WAIT_READY))
2292 		return (NULL);
2293 
2294 	zio_taskq_dispatch(zio, ZIO_TASKQ_ISSUE, B_FALSE);
2295 	return (NULL);
2296 }
2297 
2298 /*
2299  * ==========================================================================
2300  * Completion batching
2301  * ==========================================================================
2302  *
2303  * A vdev child's entire life after the block layer returns is three pipeline
2304  * stages: VDEV_IO_DONE, VDEV_IO_ASSESS and DONE (ZIO_VDEV_CHILD_PIPELINE).
2305  * For a parent with many children, such as RAIDZ or a mirror, every child but
2306  * the last does nothing there except decrement the parent's child count, yet
2307  * each one costs a taskq dispatch and a context switch to get there.
2308  *
2309  * A batch collects the children of one parent as they return, and once the last
2310  * of them is in, runs all of their completions, and then the parent's, on one
2311  * thread.  Arrival happens in the block layer completion context, so it is
2312  * lock-free: bio_endio() on Linux can run in softirq, where the sleepable
2313  * mutex_t is not usable.
2314  *
2315  * Only children that actually arrive from the block layer join zb_arrived; one
2316  * that reaches its completion on a pipeline thread instead just releases its
2317  * hold and runs that completion itself, as it would have without any of this.
2318  * Building the list on arrival is what allows that, since such a child may run
2319  * all the way to zio_destroy() long before the batch does.
2320  *
2321  * A child that occupies a vdev queue slot must never be a member.  The slot is
2322  * released by vdev_queue_io_done(), part of the deferred completion, while a
2323  * sibling may still be queued for a slot on another vdev whose slots are in
2324  * turn held by the members of other waiting batches -- a cycle that deadlocks.
2325  */
2326 static int zio_batch_enabled = 1;
2327 
2328 /*
2329  * Open a batch collecting the completions of the vdev children this zio is
2330  * about to create, which do little but count down to it.  Every one of those
2331  * children must be created before the matching zio_batch_rele().
2332  */
2333 void
2334 zio_batch_create(zio_t *pio)
2335 {
2336 	zio_batch_t *zb;
2337 
2338 	ASSERT3P(pio->io_child_batch, ==, NULL);
2339 
2340 	if (!zio_batch_enabled)
2341 		return;
2342 
2343 	zb = kmem_alloc(sizeof (*zb), KM_SLEEP);
2344 	zb->zb_arrived = NULL;
2345 	zb->zb_holds = 1;		/* creator's hold */
2346 	pio->io_child_batch = zb;
2347 }
2348 
2349 /*
2350  * Free the batch and return the list of members that arrived on it, for the
2351  * caller to execute.  Members arrive by prepending, and are equal peers of one
2352  * parent, so their order should not matter; the list is reversed into
2353  * completion order only because it is walked here anyway.  Membership is
2354  * dropped in that walk, both because VDEV_IO_ASSESS may reissue a member, which
2355  * must not rejoin, and so that a member's later zio_batch_leave() does not
2356  * touch the batch once it is freed.
2357  */
2358 static zio_t *
2359 zio_batch_run(zio_batch_t *zb)
2360 {
2361 	zio_t *list = NULL, *zio, *next;
2362 
2363 	/* Pairs with zio_batch_arrive(). */
2364 	membar_consumer();
2365 
2366 	for (zio = zb->zb_arrived; zio != NULL; zio = next) {
2367 		next = zio->io_exec_next;
2368 		zio->io_batch = NULL;
2369 		zio->io_exec_next = list;
2370 		list = zio;
2371 	}
2372 
2373 	kmem_free(zb, sizeof (*zb));
2374 
2375 	return (list);
2376 }
2377 
2378 static void
2379 zio_batch_execute(void *arg)
2380 {
2381 	zio_execute(zio_batch_run(((zio_t *)arg)->io_batch));
2382 }
2383 
2384 static void
2385 zio_batch_join(zio_batch_t *zb, zio_t *zio)
2386 {
2387 	ASSERT3P(zio->io_batch, ==, NULL);
2388 	atomic_inc_64(&zb->zb_holds);
2389 	zio->io_batch = zb;
2390 }
2391 
2392 /*
2393  * Close the batch, once all of its members have been created, dropping the hold
2394  * that kept it from running while they were still being created.  Callers do
2395  * this before advancing the parent into VDEV_IO_DONE, where it will wait for
2396  * them; the members go ahead of it in the list, which is harmless, since all
2397  * they do there is decrement its child count.  io_child_batch is cleared, so
2398  * that children created later, such as the repair writes from
2399  * vdev_raidz_io_done(), do not join a batch that is already gone.  Returns the
2400  * parent, preceded by any members that arrived while it was still creating
2401  * them, for the caller to execute.
2402  */
2403 zio_t *
2404 zio_batch_rele(zio_t *pio)
2405 {
2406 	zio_batch_t *zb = pio->io_child_batch;
2407 	zio_t *list, *last;
2408 
2409 	ASSERT3P(pio->io_exec_next, ==, NULL);
2410 
2411 	if (zb == NULL)
2412 		return (pio);
2413 
2414 	pio->io_child_batch = NULL;
2415 	if (atomic_dec_64_nv(&zb->zb_holds) != 0)
2416 		return (pio);
2417 
2418 	if ((list = zio_batch_run(zb)) == NULL)
2419 		return (pio);
2420 
2421 	last = list;
2422 	while (last->io_exec_next != NULL)
2423 		last = last->io_exec_next;
2424 	last->io_exec_next = pio;
2425 	return (list);
2426 }
2427 
2428 /*
2429  * Called in place of a member's taskq dispatch, from the block layer
2430  * completion context.  Returns B_TRUE if the zio was absorbed by a batch, in
2431  * which case the caller must not touch it again.
2432  */
2433 static boolean_t
2434 zio_batch_arrive(zio_t *zio)
2435 {
2436 	zio_batch_t *zb = zio->io_batch;
2437 	zio_t *head;
2438 
2439 	if (zb == NULL)
2440 		return (B_FALSE);
2441 
2442 	/*
2443 	 * The completion is deferred, so take the service time here, while it
2444 	 * still is one: vdev_child_slow_outlier() sits out RAIDZ children based
2445 	 * on io_delta and io_delay.  A non-zero io_delta also tells the stages
2446 	 * below when the block layer returned, as io_timestamp + io_delta.
2447 	 */
2448 	ASSERT3U(zio->io_timestamp, !=, 0);
2449 	zio->io_delta = gethrtime() - zio->io_timestamp;
2450 
2451 	do {
2452 		head = zb->zb_arrived;
2453 		zio->io_exec_next = head;
2454 	} while (atomic_cas_ptr(&zb->zb_arrived, head, zio) != head);
2455 
2456 	/* Publish the arrival before dropping the hold that runs the batch. */
2457 	membar_producer();
2458 
2459 	if (atomic_dec_64_nv(&zb->zb_holds) == 0) {
2460 		zio_taskq_dispatch_func(zio, ZIO_TASKQ_INTERRUPT, B_FALSE,
2461 		    zio_batch_execute);
2462 	}
2463 	return (B_TRUE);
2464 }
2465 
2466 /*
2467  * Give up a membership, either because the zio is about to take a vdev queue
2468  * slot after all, or because it reached its completion on a pipeline thread
2469  * rather than from the block layer, and so will run that completion itself.
2470  * Clearing io_batch makes this idempotent.  Returns the members for the caller
2471  * to execute if this was the last hold on the batch, and NULL otherwise.
2472  */
2473 zio_t *
2474 zio_batch_leave(zio_t *zio)
2475 {
2476 	zio_batch_t *zb = zio->io_batch;
2477 
2478 	if (likely(zb == NULL))
2479 		return (NULL);
2480 
2481 	zio->io_batch = NULL;
2482 	if (atomic_dec_64_nv(&zb->zb_holds) != 0)
2483 		return (NULL);
2484 
2485 	return (zio_batch_run(zb));
2486 }
2487 
2488 void
2489 zio_interrupt(void *zio)
2490 {
2491 	if (zio_batch_arrive(zio))
2492 		return;
2493 	zio_taskq_dispatch(zio, ZIO_TASKQ_INTERRUPT, B_FALSE);
2494 }
2495 
2496 void
2497 zio_delay_interrupt(zio_t *zio)
2498 {
2499 	/*
2500 	 * The timeout_generic() function isn't defined in userspace, so
2501 	 * rather than trying to implement the function, the zio delay
2502 	 * functionality has been disabled for userspace builds.
2503 	 */
2504 
2505 #ifdef _KERNEL
2506 	/*
2507 	 * If io_target_timestamp is zero, then no delay has been registered
2508 	 * for this IO, thus jump to the end of this function and "skip" the
2509 	 * delay; issuing it directly to the zio layer.
2510 	 */
2511 	if (zio->io_target_timestamp != 0) {
2512 		hrtime_t now = gethrtime();
2513 
2514 		if (now >= zio->io_target_timestamp) {
2515 			/*
2516 			 * This IO has already taken longer than the target
2517 			 * delay to complete, so we don't want to delay it
2518 			 * any longer; we "miss" the delay and issue it
2519 			 * directly to the zio layer. This is likely due to
2520 			 * the target latency being set to a value less than
2521 			 * the underlying hardware can satisfy (e.g. delay
2522 			 * set to 1ms, but the disks take 10ms to complete an
2523 			 * IO request).
2524 			 */
2525 
2526 			DTRACE_PROBE2(zio__delay__miss, zio_t *, zio,
2527 			    hrtime_t, now);
2528 
2529 			zio_interrupt(zio);
2530 		} else {
2531 			taskqid_t tid;
2532 			hrtime_t diff = zio->io_target_timestamp - now;
2533 			int ticks = MAX(1, NSEC_TO_TICK(diff));
2534 			clock_t expire_at_tick = ddi_get_lbolt() + ticks;
2535 
2536 			DTRACE_PROBE3(zio__delay__hit, zio_t *, zio,
2537 			    hrtime_t, now, hrtime_t, diff);
2538 
2539 			tid = taskq_dispatch_delay(system_taskq, zio_interrupt,
2540 			    zio, TQ_NOSLEEP, expire_at_tick);
2541 			if (tid == TASKQID_INVALID) {
2542 				/*
2543 				 * Couldn't allocate a task.  Just finish the
2544 				 * zio without a delay.
2545 				 */
2546 				zio_interrupt(zio);
2547 			}
2548 		}
2549 		return;
2550 	}
2551 #endif
2552 	DTRACE_PROBE1(zio__delay__skip, zio_t *, zio);
2553 	zio_interrupt(zio);
2554 }
2555 
2556 static void
2557 zio_deadman_impl(zio_t *pio, int ziodepth)
2558 {
2559 	zio_t *cio, *cio_next;
2560 	zio_link_t *zl = NULL;
2561 	vdev_t *vd = pio->io_vd;
2562 	uint64_t failmode = spa_get_deadman_failmode(pio->io_spa);
2563 
2564 	if (zio_deadman_log_all || (vd != NULL && vd->vdev_ops->vdev_op_leaf)) {
2565 		vdev_queue_t *vq = vd ? &vd->vdev_queue : NULL;
2566 		zbookmark_phys_t *zb = &pio->io_bookmark;
2567 		uint64_t delta = gethrtime() - pio->io_timestamp;
2568 
2569 		zfs_dbgmsg("slow zio[%d]: zio=%px timestamp=%llu "
2570 		    "delta=%llu queued=%llu io=%llu "
2571 		    "path=%s "
2572 		    "last=%llu type=%d "
2573 		    "priority=%d flags=0x%llx stage=0x%x "
2574 		    "pipeline=0x%x pipeline-trace=0x%x "
2575 		    "objset=%llu object=%llu "
2576 		    "level=%llu blkid=%llu "
2577 		    "offset=%llu size=%llu "
2578 		    "error=%d",
2579 		    ziodepth, pio, pio->io_timestamp,
2580 		    (u_longlong_t)delta, pio->io_delta, pio->io_delay,
2581 		    vd ? vd->vdev_path : "NULL",
2582 		    vq ? vq->vq_io_complete_ts : 0, pio->io_type,
2583 		    pio->io_priority, (u_longlong_t)pio->io_flags,
2584 		    pio->io_stage, pio->io_pipeline, pio->io_pipeline_trace,
2585 		    (u_longlong_t)zb->zb_objset, (u_longlong_t)zb->zb_object,
2586 		    (u_longlong_t)zb->zb_level, (u_longlong_t)zb->zb_blkid,
2587 		    (u_longlong_t)pio->io_offset, (u_longlong_t)pio->io_size,
2588 		    pio->io_error);
2589 		(void) zfs_ereport_post(FM_EREPORT_ZFS_DEADMAN,
2590 		    pio->io_spa, vd, zb, pio, 0);
2591 	}
2592 
2593 	if (vd != NULL && vd->vdev_ops->vdev_op_leaf &&
2594 	    list_is_empty(&pio->io_child_list) &&
2595 	    failmode == ZIO_FAILURE_MODE_CONTINUE &&
2596 	    taskq_empty_ent(&pio->io_tqent) &&
2597 	    pio->io_queue_state == ZIO_QS_ACTIVE) {
2598 		pio->io_error = EINTR;
2599 		zio_interrupt(pio);
2600 	}
2601 
2602 	mutex_enter(&pio->io_lock);
2603 	for (cio = zio_walk_children(pio, &zl); cio != NULL; cio = cio_next) {
2604 		cio_next = zio_walk_children(pio, &zl);
2605 		zio_deadman_impl(cio, ziodepth + 1);
2606 	}
2607 	mutex_exit(&pio->io_lock);
2608 }
2609 
2610 /*
2611  * Log the critical information describing this zio and all of its children
2612  * using the zfs_dbgmsg() interface then post deadman event for the ZED.
2613  */
2614 void
2615 zio_deadman(zio_t *pio, const char *tag)
2616 {
2617 	spa_t *spa = pio->io_spa;
2618 	char *name = spa_name(spa);
2619 
2620 	if (!zfs_deadman_enabled || spa_suspended(spa))
2621 		return;
2622 
2623 	zio_deadman_impl(pio, 0);
2624 
2625 	switch (spa_get_deadman_failmode(spa)) {
2626 	case ZIO_FAILURE_MODE_WAIT:
2627 		zfs_dbgmsg("%s waiting for hung I/O to pool '%s'", tag, name);
2628 		break;
2629 
2630 	case ZIO_FAILURE_MODE_CONTINUE:
2631 		zfs_dbgmsg("%s restarting hung I/O for pool '%s'", tag, name);
2632 		break;
2633 
2634 	case ZIO_FAILURE_MODE_PANIC:
2635 		fm_panic("%s determined I/O to pool '%s' is hung.", tag, name);
2636 		break;
2637 	}
2638 }
2639 
2640 /*
2641  * Execute the I/O pipeline until one of the following occurs:
2642  * (1) the I/O completes; (2) the pipeline stalls waiting for
2643  * dependent child I/Os; (3) the I/O issues, so we're waiting
2644  * for an I/O completion interrupt; (4) the I/O is delegated by
2645  * vdev-level caching or aggregation; (5) the I/O is deferred
2646  * due to vdev-level queueing; (6) the I/O is handed off to
2647  * another thread.  In all cases, the pipeline stops whenever
2648  * there's no CPU work; it never burns a thread in cv_wait_io().
2649  *
2650  * There's no locking on io_stage because there's no legitimate way
2651  * for multiple threads to be attempting to process the same I/O.
2652  */
2653 static zio_pipe_stage_t *zio_pipeline[];
2654 
2655 /*
2656  * zio_execute() is a wrapper around the static function
2657  * __zio_execute() so that we can force  __zio_execute() to be
2658  * inlined.  This reduces stack overhead which is important
2659  * because __zio_execute() is called recursively in several zio
2660  * code paths.  zio_execute() itself cannot be inlined because
2661  * it is externally visible.
2662  */
2663 void
2664 zio_execute(void *zio)
2665 {
2666 	fstrans_cookie_t cookie;
2667 
2668 	cookie = spl_fstrans_mark();
2669 	__zio_execute(zio);
2670 	spl_fstrans_unmark(cookie);
2671 }
2672 
2673 /*
2674  * Used to determine if in the current context the stack is sized large
2675  * enough to allow zio_execute() to be called recursively.  A minimum
2676  * stack size of 16K is required to avoid needing to re-dispatch the zio.
2677  */
2678 static boolean_t
2679 zio_execute_stack_check(zio_t *zio)
2680 {
2681 #if !defined(HAVE_LARGE_STACKS)
2682 	dsl_pool_t *dp = spa_get_dsl(zio->io_spa);
2683 
2684 	/* Executing in txg_sync_thread() context. */
2685 	if (dp && curthread == dp->dp_tx.tx_sync_thread)
2686 		return (B_TRUE);
2687 
2688 	/* Pool initialization outside of zio_taskq context. */
2689 	if (dp && spa_is_initializing(dp->dp_spa) &&
2690 	    !zio_taskq_member(zio, ZIO_TASKQ_ISSUE) &&
2691 	    !zio_taskq_member(zio, ZIO_TASKQ_ISSUE_HIGH))
2692 		return (B_TRUE);
2693 #else
2694 	(void) zio;
2695 #endif /* HAVE_LARGE_STACKS */
2696 
2697 	return (B_FALSE);
2698 }
2699 
2700 /*
2701  * Run one pipeline stage, returning a list of zios to continue with, or NULL
2702  * if this thread is done with it.
2703  */
2704 __attribute__((always_inline))
2705 static inline zio_t *
2706 zio_execute_stage(zio_t *zio)
2707 {
2708 	enum zio_stage pipeline = zio->io_pipeline;
2709 	enum zio_stage stage = zio->io_stage;
2710 
2711 	zio->io_executor = curthread;
2712 
2713 	ASSERT(!MUTEX_HELD(&zio->io_lock));
2714 	ASSERT0P(zio->io_stall);
2715 	ASSERT(ISP2(stage));
2716 	ASSERT(pipeline & ~((stage << 1) - 1));
2717 
2718 	do {
2719 		stage <<= 1;
2720 	} while ((stage & pipeline) == 0);
2721 
2722 	ASSERT(stage <= ZIO_STAGE_DONE);
2723 
2724 	/*
2725 	 * If we are in interrupt context and this pipeline stage will grab
2726 	 * a config lock that is held across I/O, or may wait for an I/O that
2727 	 * needs an interrupt thread to complete, issue async to avoid deadlock.
2728 	 *
2729 	 * For VDEV_IO_START, we cut in line so that the io will be sent to
2730 	 * disk promptly.
2731 	 */
2732 	if ((stage & ZIO_BLOCKING_STAGES) && zio->io_vd == NULL &&
2733 	    zio_taskq_member(zio, ZIO_TASKQ_INTERRUPT)) {
2734 		boolean_t cut = (stage == ZIO_STAGE_VDEV_IO_START) ?
2735 		    zio_requeue_io_start_cut_in_line : B_FALSE;
2736 		zio_taskq_dispatch(zio, ZIO_TASKQ_ISSUE, cut);
2737 		return (NULL);
2738 	}
2739 
2740 	/*
2741 	 * If the current context doesn't have large enough stacks
2742 	 * the zio must be issued asynchronously to prevent overflow.
2743 	 */
2744 	if (zio_execute_stack_check(zio)) {
2745 		boolean_t cut = (stage == ZIO_STAGE_VDEV_IO_START) ?
2746 		    zio_requeue_io_start_cut_in_line : B_FALSE;
2747 		zio_taskq_dispatch(zio, ZIO_TASKQ_ISSUE, cut);
2748 		return (NULL);
2749 	}
2750 
2751 	zio->io_stage = stage;
2752 	zio->io_pipeline_trace |= zio->io_stage;
2753 
2754 	/*
2755 	 * The zio pipeline stage returns the next zio to execute (typically
2756 	 * the same as this one), or NULL if we should stop.  It may also
2757 	 * chain more zios to it for us to execute later.
2758 	 */
2759 	return (zio_pipeline[highbit64(stage) - 1](zio));
2760 }
2761 
2762 /*
2763  * Take all but the first of the zios a stage handed back off its head, and
2764  * prepend the rest to those already pending.  Dispatch heavyweight ZIOs except
2765  * the last, so that they could run in parallel.
2766  */
2767 static inline void
2768 zio_execute_defer(zio_t *zio, zio_t **pendingp)
2769 {
2770 	zio_t *list = NULL, **tailp = &list;
2771 	zio_t *next;
2772 
2773 	for (zio_t *cur = zio->io_exec_next; cur != NULL; cur = next) {
2774 		next = cur->io_exec_next;
2775 		cur->io_exec_next = NULL;
2776 		if ((next != NULL || *pendingp != NULL) &&
2777 		    !(cur->io_flags & ZIO_FLAG_LIGHTWEIGHT)) {
2778 			zio_taskq_dispatch(cur,
2779 			    cur->io_stage < ZIO_STAGE_VDEV_IO_START ?
2780 			    ZIO_TASKQ_ISSUE : ZIO_TASKQ_INTERRUPT, B_FALSE);
2781 			continue;
2782 		}
2783 		*tailp = cur;
2784 		tailp = &cur->io_exec_next;
2785 	}
2786 
2787 	*tailp = *pendingp;
2788 	*pendingp = list;
2789 	zio->io_exec_next = NULL;
2790 }
2791 
2792 __attribute__((always_inline))
2793 static inline void
2794 __zio_execute(zio_t *zio)
2795 {
2796 	zio_t *pending = zio->io_exec_next;
2797 	zio->io_exec_next = NULL;
2798 
2799 	for (;;) {
2800 		zio_t *last = zio;
2801 		while ((zio = zio_execute_stage(zio)) != NULL) {
2802 			if (zio->io_exec_next != NULL)
2803 				zio_execute_defer(zio, &pending);
2804 
2805 			/*
2806 			 * A heavyweight zio is dispatched if others are already
2807 			 * waiting for this thread to let them run in parallel.
2808 			 */
2809 			if (zio != last && pending != NULL &&
2810 			    !(zio->io_flags & ZIO_FLAG_LIGHTWEIGHT)) {
2811 				zio_taskq_dispatch(zio,
2812 				    zio->io_stage < ZIO_STAGE_VDEV_IO_START ?
2813 				    ZIO_TASKQ_ISSUE : ZIO_TASKQ_INTERRUPT,
2814 				    B_FALSE);
2815 				break;
2816 			}
2817 			last = zio;
2818 		}
2819 
2820 		if ((zio = pending) == NULL)
2821 			return;
2822 		pending = zio->io_exec_next;
2823 		zio->io_exec_next = NULL;
2824 	}
2825 }
2826 
2827 
2828 /*
2829  * ==========================================================================
2830  * Initiate I/O, either sync or async
2831  * ==========================================================================
2832  */
2833 int
2834 zio_wait(zio_t *zio)
2835 {
2836 	/*
2837 	 * Some routines, like zio_free_sync(), may return a NULL zio
2838 	 * to avoid the performance overhead of creating and then destroying
2839 	 * an unneeded zio.  For the callers' simplicity, we accept a NULL
2840 	 * zio and ignore it.
2841 	 */
2842 	if (zio == NULL)
2843 		return (0);
2844 
2845 	long timeout = MSEC_TO_TICK(zfs_deadman_ziotime_ms);
2846 	int error;
2847 
2848 	ASSERT3S(zio->io_stage, ==, ZIO_STAGE_OPEN);
2849 	ASSERT0P(zio->io_executor);
2850 
2851 	zio->io_waiter = curthread;
2852 	ASSERT0(zio->io_queued_timestamp);
2853 	zio->io_queued_timestamp = gethrtime();
2854 
2855 	if (zio->io_type == ZIO_TYPE_WRITE) {
2856 		spa_select_allocator(zio);
2857 	}
2858 	__zio_execute(zio);
2859 
2860 	mutex_enter(&zio->io_lock);
2861 	while (zio->io_executor != NULL) {
2862 		error = cv_timedwait_io(&zio->io_cv, &zio->io_lock,
2863 		    ddi_get_lbolt() + timeout);
2864 
2865 		if (zfs_deadman_enabled && error == -1 &&
2866 		    gethrtime() - zio->io_queued_timestamp >
2867 		    spa_deadman_ziotime(zio->io_spa)) {
2868 			mutex_exit(&zio->io_lock);
2869 			timeout = MSEC_TO_TICK(zfs_deadman_checktime_ms);
2870 			zio_deadman(zio, FTAG);
2871 			mutex_enter(&zio->io_lock);
2872 		}
2873 	}
2874 	mutex_exit(&zio->io_lock);
2875 
2876 	error = zio->io_error;
2877 	zio_destroy(zio);
2878 
2879 	return (error);
2880 }
2881 
2882 void
2883 zio_nowait(zio_t *zio)
2884 {
2885 	/*
2886 	 * See comment in zio_wait().
2887 	 */
2888 	if (zio == NULL)
2889 		return;
2890 
2891 	ASSERT0P(zio->io_executor);
2892 
2893 	if (zio->io_child_type == ZIO_CHILD_LOGICAL &&
2894 	    list_is_empty(&zio->io_parent_list)) {
2895 		zio_t *pio;
2896 
2897 		/*
2898 		 * This is a logical async I/O with no parent to wait for it.
2899 		 * We add it to the spa_async_root_zio "Godfather" I/O which
2900 		 * will ensure they complete prior to unloading the pool.
2901 		 */
2902 		spa_t *spa = zio->io_spa;
2903 		pio = spa->spa_async_zio_root[CPU_SEQID_UNSTABLE];
2904 
2905 		zio_add_child(pio, zio);
2906 	}
2907 
2908 	ASSERT0(zio->io_queued_timestamp);
2909 	zio->io_queued_timestamp = gethrtime();
2910 	if (zio->io_type == ZIO_TYPE_WRITE) {
2911 		spa_select_allocator(zio);
2912 	}
2913 	__zio_execute(zio);
2914 }
2915 
2916 /*
2917  * ==========================================================================
2918  * Reexecute, cancel, or suspend/resume failed I/O
2919  * ==========================================================================
2920  */
2921 
2922 static void
2923 zio_reexecute(void *arg)
2924 {
2925 	zio_t *pio = arg;
2926 	zio_t *cio, *cio_next, *gio;
2927 
2928 	ASSERT(pio->io_child_type == ZIO_CHILD_LOGICAL);
2929 	ASSERT(pio->io_orig_stage == ZIO_STAGE_OPEN);
2930 	ASSERT0P(pio->io_gang_leader);
2931 	ASSERT0P(pio->io_gang_tree);
2932 
2933 	mutex_enter(&pio->io_lock);
2934 	pio->io_flags = pio->io_orig_flags;
2935 	pio->io_stage = pio->io_orig_stage;
2936 	pio->io_pipeline = pio->io_orig_pipeline;
2937 	pio->io_post = 0;
2938 	pio->io_flags |= ZIO_FLAG_REEXECUTED;
2939 	pio->io_pipeline_trace = 0;
2940 	pio->io_error = 0;
2941 	pio->io_state[ZIO_WAIT_READY] = (pio->io_stage >= ZIO_STAGE_READY) ||
2942 	    (pio->io_pipeline & ZIO_STAGE_READY) == 0;
2943 	pio->io_state[ZIO_WAIT_DONE] = (pio->io_stage >= ZIO_STAGE_DONE);
2944 
2945 	/*
2946 	 * It's possible for a failed ZIO to be a descendant of more than one
2947 	 * ZIO tree. When reexecuting it, we have to be sure to add its wait
2948 	 * states to all parent wait counts.
2949 	 *
2950 	 * Those parents, in turn, may have other children that are currently
2951 	 * active, usually because they've already been reexecuted after
2952 	 * resuming. Those children may be executing and may call
2953 	 * zio_notify_parent() at the same time as we're updating our parent's
2954 	 * counts. To avoid races while updating the counts, we take
2955 	 * gio->io_lock before each update.
2956 	 */
2957 	zio_link_t *zl = NULL;
2958 	while ((gio = zio_walk_parents(pio, &zl)) != NULL) {
2959 		mutex_enter(&gio->io_lock);
2960 		for (int w = 0; w < ZIO_WAIT_TYPES; w++) {
2961 			gio->io_children[pio->io_child_type][w] +=
2962 			    !pio->io_state[w];
2963 		}
2964 		mutex_exit(&gio->io_lock);
2965 	}
2966 
2967 	for (int c = 0; c < ZIO_CHILD_TYPES; c++)
2968 		pio->io_child_error[c] = 0;
2969 
2970 	if (IO_IS_ALLOCATING(pio))
2971 		BP_ZERO(pio->io_bp);
2972 
2973 	/*
2974 	 * As we reexecute pio's children, new children could be created.
2975 	 * New children go to the head of pio's io_child_list, however,
2976 	 * so we will (correctly) not reexecute them.  The key is that
2977 	 * the remainder of pio's io_child_list, from 'cio_next' onward,
2978 	 * cannot be affected by any side effects of reexecuting 'cio'.
2979 	 */
2980 	zl = NULL;
2981 	for (cio = zio_walk_children(pio, &zl); cio != NULL; cio = cio_next) {
2982 		cio_next = zio_walk_children(pio, &zl);
2983 		mutex_exit(&pio->io_lock);
2984 		zio_reexecute(cio);
2985 		mutex_enter(&pio->io_lock);
2986 	}
2987 	mutex_exit(&pio->io_lock);
2988 
2989 	/*
2990 	 * Now that all children have been reexecuted, execute the parent.
2991 	 * We don't reexecute "The Godfather" I/O here as it's the
2992 	 * responsibility of the caller to wait on it.
2993 	 */
2994 	if (!(pio->io_flags & ZIO_FLAG_GODFATHER)) {
2995 		pio->io_queued_timestamp = gethrtime();
2996 		__zio_execute(pio);
2997 	}
2998 }
2999 
3000 void
3001 zio_suspend(spa_t *spa, zio_t *zio, zio_suspend_reason_t reason)
3002 {
3003 	if (spa_get_failmode(spa) == ZIO_FAILURE_MODE_PANIC)
3004 		fm_panic("Pool '%s' has encountered an uncorrectable I/O "
3005 		    "failure and the failure mode property for this pool "
3006 		    "is set to panic.", spa_name(spa));
3007 
3008 	if (reason != ZIO_SUSPEND_MMP) {
3009 		cmn_err(CE_WARN, "Pool '%s' has encountered an uncorrectable "
3010 		    "I/O failure and has been suspended.", spa_name(spa));
3011 	}
3012 
3013 	(void) zfs_ereport_post(FM_EREPORT_ZFS_IO_FAILURE, spa, NULL,
3014 	    NULL, NULL, 0);
3015 
3016 	mutex_enter(&spa->spa_suspend_lock);
3017 
3018 	if (spa->spa_suspend_zio_root == NULL)
3019 		spa->spa_suspend_zio_root = zio_root(spa, NULL, NULL,
3020 		    ZIO_FLAG_CANFAIL | ZIO_FLAG_SPECULATIVE |
3021 		    ZIO_FLAG_GODFATHER);
3022 
3023 	spa->spa_suspended = reason;
3024 
3025 	if (zio != NULL) {
3026 		ASSERT(!(zio->io_flags & ZIO_FLAG_GODFATHER));
3027 		ASSERT(zio != spa->spa_suspend_zio_root);
3028 		ASSERT(zio->io_child_type == ZIO_CHILD_LOGICAL);
3029 		ASSERT0P(zio_unique_parent(zio));
3030 		ASSERT(zio->io_stage == ZIO_STAGE_DONE);
3031 		zio_add_child(spa->spa_suspend_zio_root, zio);
3032 	}
3033 
3034 	mutex_exit(&spa->spa_suspend_lock);
3035 
3036 	txg_wait_kick(spa->spa_dsl_pool);
3037 }
3038 
3039 int
3040 zio_resume(spa_t *spa)
3041 {
3042 	zio_t *pio;
3043 
3044 	/*
3045 	 * Reexecute all previously suspended i/o.
3046 	 */
3047 	mutex_enter(&spa->spa_suspend_lock);
3048 	if (spa->spa_suspended != ZIO_SUSPEND_NONE)
3049 		cmn_err(CE_WARN, "Pool '%s' was suspended and is being "
3050 		    "resumed. Failed I/O will be retried.",
3051 		    spa_name(spa));
3052 	spa->spa_suspended = ZIO_SUSPEND_NONE;
3053 	cv_broadcast(&spa->spa_suspend_cv);
3054 	pio = spa->spa_suspend_zio_root;
3055 	spa->spa_suspend_zio_root = NULL;
3056 	mutex_exit(&spa->spa_suspend_lock);
3057 
3058 	if (pio == NULL)
3059 		return (0);
3060 
3061 	zio_reexecute(pio);
3062 	return (zio_wait(pio));
3063 }
3064 
3065 void
3066 zio_resume_wait(spa_t *spa)
3067 {
3068 	mutex_enter(&spa->spa_suspend_lock);
3069 	while (spa_suspended(spa))
3070 		cv_wait(&spa->spa_suspend_cv, &spa->spa_suspend_lock);
3071 	mutex_exit(&spa->spa_suspend_lock);
3072 }
3073 
3074 /*
3075  * ==========================================================================
3076  * Gang blocks.
3077  *
3078  * A gang block is a collection of small blocks that looks to the DMU
3079  * like one large block.  When zio_dva_allocate() cannot find a block
3080  * of the requested size, due to either severe fragmentation or the pool
3081  * being nearly full, it calls zio_write_gang_block() to construct the
3082  * block from smaller fragments.
3083  *
3084  * A gang block consists of a a gang header and up to gbh_nblkptrs(size)
3085  * gang members. The gang header is like an indirect block: it's an array
3086  * of block pointers, though the header has a small tail (a zio_eck_t)
3087  * that stores an embedded checksum. It is allocated using only a single
3088  * sector as the requested size, and hence is allocatable regardless of
3089  * fragmentation. Its size is determined by the smallest allocatable
3090  * asize of the vdevs it was allocated on. The gang header's bps point
3091  * to its gang members, which hold the data.
3092  *
3093  * Gang blocks are self-checksumming, using the bp's <vdev, offset, txg>
3094  * as the verifier to ensure uniqueness of the SHA256 checksum.
3095  * Critically, the gang block bp's blk_cksum is the checksum of the data,
3096  * not the gang header.  This ensures that data block signatures (needed for
3097  * deduplication) are independent of how the block is physically stored.
3098  *
3099  * Gang blocks can be nested: a gang member may itself be a gang block.
3100  * Thus every gang block is a tree in which root and all interior nodes are
3101  * gang headers, and the leaves are normal blocks that contain user data.
3102  * The root of the gang tree is called the gang leader.
3103  *
3104  * To perform any operation (read, rewrite, free, claim) on a gang block,
3105  * zio_gang_assemble() first assembles the gang tree (minus data leaves)
3106  * in the io_gang_tree field of the original logical i/o by recursively
3107  * reading the gang leader and all gang headers below it.  This yields
3108  * an in-core tree containing the contents of every gang header and the
3109  * bps for every constituent of the gang block.
3110  *
3111  * With the gang tree now assembled, zio_gang_issue() just walks the gang tree
3112  * and invokes a callback on each bp.  To free a gang block, zio_gang_issue()
3113  * calls zio_free_gang() -- a trivial wrapper around zio_free() -- for each bp.
3114  * zio_claim_gang() provides a similarly trivial wrapper for zio_claim().
3115  * zio_read_gang() is a wrapper around zio_read() that omits reading gang
3116  * headers, since we already have those in io_gang_tree.  zio_rewrite_gang()
3117  * performs a zio_rewrite() of the data or, for gang headers, a zio_rewrite()
3118  * of the gang header plus zio_checksum_compute() of the data to update the
3119  * gang header's blk_cksum as described above.
3120  *
3121  * The two-phase assemble/issue model solves the problem of partial failure --
3122  * what if you'd freed part of a gang block but then couldn't read the
3123  * gang header for another part?  Assembling the entire gang tree first
3124  * ensures that all the necessary gang header I/O has succeeded before
3125  * starting the actual work of free, claim, or write.  Once the gang tree
3126  * is assembled, free and claim are in-memory operations that cannot fail.
3127  *
3128  * In the event that a gang write fails, zio_dva_unallocate() walks the
3129  * gang tree to immediately free (i.e. insert back into the space map)
3130  * everything we've allocated.  This ensures that we don't get ENOSPC
3131  * errors during repeated suspend/resume cycles due to a flaky device.
3132  *
3133  * Gang rewrites only happen during sync-to-convergence.  If we can't assemble
3134  * the gang tree, we won't modify the block, so we can safely defer the free
3135  * (knowing that the block is still intact).  If we *can* assemble the gang
3136  * tree, then even if some of the rewrites fail, zio_dva_unallocate() will free
3137  * each constituent bp and we can allocate a new block on the next sync pass.
3138  *
3139  * In all cases, the gang tree allows complete recovery from partial failure.
3140  * ==========================================================================
3141  */
3142 
3143 static void
3144 zio_gang_issue_func_done(zio_t *zio)
3145 {
3146 	abd_free(zio->io_abd);
3147 }
3148 
3149 static zio_t *
3150 zio_read_gang(zio_t *pio, blkptr_t *bp, zio_gang_node_t *gn, abd_t *data,
3151     uint64_t offset)
3152 {
3153 	if (gn != NULL)
3154 		return (pio);
3155 
3156 	return (zio_read(pio, pio->io_spa, bp, abd_get_offset(data, offset),
3157 	    BP_GET_PSIZE(bp), zio_gang_issue_func_done,
3158 	    NULL, pio->io_priority, ZIO_GANG_CHILD_FLAGS(pio),
3159 	    &pio->io_bookmark));
3160 }
3161 
3162 static zio_t *
3163 zio_rewrite_gang(zio_t *pio, blkptr_t *bp, zio_gang_node_t *gn, abd_t *data,
3164     uint64_t offset)
3165 {
3166 	zio_t *zio;
3167 
3168 	if (gn != NULL) {
3169 		abd_t *gbh_abd =
3170 		    abd_get_from_buf(gn->gn_gbh, gn->gn_gangblocksize);
3171 		zio = zio_rewrite(pio, pio->io_spa, pio->io_txg, bp,
3172 		    gbh_abd, gn->gn_gangblocksize, zio_gang_issue_func_done,
3173 		    NULL, pio->io_priority, ZIO_GANG_CHILD_FLAGS(pio),
3174 		    &pio->io_bookmark);
3175 		/*
3176 		 * As we rewrite each gang header, the pipeline will compute
3177 		 * a new gang block header checksum for it; but no one will
3178 		 * compute a new data checksum, so we do that here.  The one
3179 		 * exception is the gang leader: the pipeline already computed
3180 		 * its data checksum because that stage precedes gang assembly.
3181 		 * (Presently, nothing actually uses interior data checksums;
3182 		 * this is just good hygiene.)
3183 		 */
3184 		if (gn != pio->io_gang_leader->io_gang_tree) {
3185 			abd_t *buf = abd_get_offset(data, offset);
3186 
3187 			zio_checksum_compute(zio, BP_GET_CHECKSUM(bp),
3188 			    buf, BP_GET_PSIZE(bp));
3189 
3190 			abd_free(buf);
3191 		}
3192 		/*
3193 		 * If we are here to damage data for testing purposes,
3194 		 * leave the GBH alone so that we can detect the damage.
3195 		 */
3196 		if (pio->io_gang_leader->io_flags & ZIO_FLAG_INDUCE_DAMAGE)
3197 			zio->io_pipeline &= ~ZIO_VDEV_IO_STAGES;
3198 	} else {
3199 		zio = zio_rewrite(pio, pio->io_spa, pio->io_txg, bp,
3200 		    abd_get_offset(data, offset), BP_GET_PSIZE(bp),
3201 		    zio_gang_issue_func_done, NULL, pio->io_priority,
3202 		    ZIO_GANG_CHILD_FLAGS(pio), &pio->io_bookmark);
3203 	}
3204 
3205 	return (zio);
3206 }
3207 
3208 static zio_t *
3209 zio_free_gang(zio_t *pio, blkptr_t *bp, zio_gang_node_t *gn, abd_t *data,
3210     uint64_t offset)
3211 {
3212 	(void) gn, (void) data, (void) offset;
3213 
3214 	zio_t *zio = zio_free_sync(pio, pio->io_spa, pio->io_txg, bp,
3215 	    ZIO_GANG_CHILD_FLAGS(pio));
3216 	if (zio == NULL) {
3217 		zio = zio_null(pio, pio->io_spa,
3218 		    NULL, NULL, NULL, ZIO_GANG_CHILD_FLAGS(pio));
3219 	}
3220 	return (zio);
3221 }
3222 
3223 static zio_t *
3224 zio_claim_gang(zio_t *pio, blkptr_t *bp, zio_gang_node_t *gn, abd_t *data,
3225     uint64_t offset)
3226 {
3227 	(void) gn, (void) data, (void) offset;
3228 	return (zio_claim(pio, pio->io_spa, pio->io_txg, bp,
3229 	    NULL, NULL, ZIO_GANG_CHILD_FLAGS(pio)));
3230 }
3231 
3232 static zio_gang_issue_func_t *zio_gang_issue_func[ZIO_TYPES] = {
3233 	NULL,
3234 	zio_read_gang,
3235 	zio_rewrite_gang,
3236 	zio_free_gang,
3237 	zio_claim_gang,
3238 	NULL
3239 };
3240 
3241 static void zio_gang_tree_assemble_done(zio_t *zio);
3242 
3243 static zio_gang_node_t *
3244 zio_gang_node_alloc(zio_gang_node_t **gnpp, uint64_t gangblocksize)
3245 {
3246 	zio_gang_node_t *gn;
3247 
3248 	ASSERT0P(*gnpp);
3249 
3250 	gn = kmem_zalloc(sizeof (*gn) +
3251 	    (gbh_nblkptrs(gangblocksize) * sizeof (gn)), KM_SLEEP);
3252 	gn->gn_gangblocksize = gn->gn_allocsize = gangblocksize;
3253 	gn->gn_gbh = zio_buf_alloc(gangblocksize);
3254 	*gnpp = gn;
3255 
3256 	return (gn);
3257 }
3258 
3259 static void
3260 zio_gang_node_free(zio_gang_node_t **gnpp)
3261 {
3262 	zio_gang_node_t *gn = *gnpp;
3263 
3264 	for (int g = 0; g < gbh_nblkptrs(gn->gn_allocsize); g++)
3265 		ASSERT0P(gn->gn_child[g]);
3266 
3267 	zio_buf_free(gn->gn_gbh, gn->gn_allocsize);
3268 	kmem_free(gn, sizeof (*gn) +
3269 	    (gbh_nblkptrs(gn->gn_allocsize) * sizeof (gn)));
3270 	*gnpp = NULL;
3271 }
3272 
3273 static void
3274 zio_gang_tree_free(zio_gang_node_t **gnpp)
3275 {
3276 	zio_gang_node_t *gn = *gnpp;
3277 
3278 	if (gn == NULL)
3279 		return;
3280 
3281 	for (int g = 0; g < gbh_nblkptrs(gn->gn_allocsize); g++)
3282 		zio_gang_tree_free(&gn->gn_child[g]);
3283 
3284 	zio_gang_node_free(gnpp);
3285 }
3286 
3287 static void
3288 zio_gang_tree_assemble(zio_t *gio, blkptr_t *bp, zio_gang_node_t **gnpp)
3289 {
3290 	uint64_t gangblocksize = UINT64_MAX;
3291 	if (spa_feature_is_active(gio->io_spa,
3292 	    SPA_FEATURE_DYNAMIC_GANG_HEADER)) {
3293 		spa_config_enter(gio->io_spa, SCL_VDEV, FTAG, RW_READER);
3294 		for (int dva = 0; dva < BP_GET_NDVAS(bp); dva++) {
3295 			vdev_t *vd = vdev_lookup_top(gio->io_spa,
3296 			    DVA_GET_VDEV(&bp->blk_dva[dva]));
3297 			uint64_t psize = vdev_gang_header_psize(vd);
3298 			gangblocksize = MIN(gangblocksize, psize);
3299 		}
3300 		spa_config_exit(gio->io_spa, SCL_VDEV, FTAG);
3301 	} else {
3302 		gangblocksize = SPA_OLD_GANGBLOCKSIZE;
3303 	}
3304 	ASSERT3U(gangblocksize, !=, UINT64_MAX);
3305 	zio_gang_node_t *gn = zio_gang_node_alloc(gnpp, gangblocksize);
3306 	abd_t *gbh_abd = abd_get_from_buf(gn->gn_gbh, gangblocksize);
3307 
3308 	ASSERT(gio->io_gang_leader == gio);
3309 	ASSERT(BP_IS_GANG(bp));
3310 
3311 	zio_nowait(zio_read(gio, gio->io_spa, bp, gbh_abd, gangblocksize,
3312 	    zio_gang_tree_assemble_done, gn, gio->io_priority,
3313 	    ZIO_GANG_CHILD_FLAGS(gio), &gio->io_bookmark));
3314 }
3315 
3316 static void
3317 zio_gang_tree_assemble_done(zio_t *zio)
3318 {
3319 	zio_t *gio = zio->io_gang_leader;
3320 	zio_gang_node_t *gn = zio->io_private;
3321 	blkptr_t *bp = zio->io_bp;
3322 
3323 	ASSERT(gio == zio_unique_parent(zio));
3324 	ASSERT(list_is_empty(&zio->io_child_list));
3325 
3326 	if (zio->io_error)
3327 		return;
3328 
3329 	/* this ABD was created from a linear buf in zio_gang_tree_assemble */
3330 	if (BP_SHOULD_BYTESWAP(bp))
3331 		byteswap_uint64_array(abd_to_buf(zio->io_abd), zio->io_size);
3332 
3333 	ASSERT3P(abd_to_buf(zio->io_abd), ==, gn->gn_gbh);
3334 	/*
3335 	 * If this was an old-style gangblock, the gangblocksize should have
3336 	 * been updated in zio_checksum_error to reflect that.
3337 	 */
3338 	ASSERT3U(gbh_eck(gn->gn_gbh, gn->gn_gangblocksize)->zec_magic,
3339 	    ==, ZEC_MAGIC);
3340 
3341 	abd_free(zio->io_abd);
3342 
3343 	for (int g = 0; g < gbh_nblkptrs(gn->gn_gangblocksize); g++) {
3344 		blkptr_t *gbp = gbh_bp(gn->gn_gbh, g);
3345 		if (!BP_IS_GANG(gbp))
3346 			continue;
3347 		zio_gang_tree_assemble(gio, gbp, &gn->gn_child[g]);
3348 	}
3349 }
3350 
3351 static void
3352 zio_gang_tree_issue(zio_t *pio, zio_gang_node_t *gn, blkptr_t *bp, abd_t *data,
3353     uint64_t offset)
3354 {
3355 	zio_t *gio = pio->io_gang_leader;
3356 	zio_t *zio;
3357 
3358 	ASSERT(BP_IS_GANG(bp) == !!gn);
3359 	ASSERT(BP_GET_CHECKSUM(bp) == BP_GET_CHECKSUM(gio->io_bp));
3360 	ASSERT(BP_GET_LSIZE(bp) == BP_GET_PSIZE(bp) || gn == gio->io_gang_tree);
3361 
3362 	/*
3363 	 * If you're a gang header, your data is in gn->gn_gbh.
3364 	 * If you're a gang member, your data is in 'data' and gn == NULL.
3365 	 */
3366 	zio = zio_gang_issue_func[gio->io_type](pio, bp, gn, data, offset);
3367 
3368 	if (gn != NULL) {
3369 		ASSERT3U(gbh_eck(gn->gn_gbh,
3370 		    gn->gn_gangblocksize)->zec_magic, ==, ZEC_MAGIC);
3371 
3372 		for (int g = 0; g < gbh_nblkptrs(gn->gn_gangblocksize); g++) {
3373 			blkptr_t *gbp = gbh_bp(gn->gn_gbh, g);
3374 			if (BP_IS_HOLE(gbp))
3375 				continue;
3376 			zio_gang_tree_issue(zio, gn->gn_child[g], gbp, data,
3377 			    offset);
3378 			offset += BP_GET_PSIZE(gbp);
3379 		}
3380 	}
3381 
3382 	if (gn == gio->io_gang_tree)
3383 		ASSERT3U(gio->io_size, ==, offset);
3384 
3385 	if (zio != pio)
3386 		zio_nowait(zio);
3387 }
3388 
3389 static zio_t *
3390 zio_gang_assemble(zio_t *zio)
3391 {
3392 	blkptr_t *bp = zio->io_bp;
3393 
3394 	ASSERT(BP_IS_GANG(bp) && zio->io_gang_leader == NULL);
3395 	ASSERT(zio->io_child_type > ZIO_CHILD_GANG);
3396 
3397 	zio->io_gang_leader = zio;
3398 
3399 	zio_gang_tree_assemble(zio, bp, &zio->io_gang_tree);
3400 
3401 	return (zio);
3402 }
3403 
3404 static zio_t *
3405 zio_gang_issue(zio_t *zio)
3406 {
3407 	blkptr_t *bp = zio->io_bp;
3408 
3409 	if (zio_wait_for_children(zio, ZIO_CHILD_GANG_BIT, ZIO_WAIT_DONE)) {
3410 		return (NULL);
3411 	}
3412 
3413 	ASSERT(BP_IS_GANG(bp) && zio->io_gang_leader == zio);
3414 	ASSERT(zio->io_child_type > ZIO_CHILD_GANG);
3415 
3416 	if (zio->io_child_error[ZIO_CHILD_GANG] == 0)
3417 		zio_gang_tree_issue(zio, zio->io_gang_tree, bp, zio->io_abd,
3418 		    0);
3419 	else
3420 		zio_gang_tree_free(&zio->io_gang_tree);
3421 
3422 	zio->io_pipeline = ZIO_INTERLOCK_PIPELINE;
3423 
3424 	return (zio);
3425 }
3426 
3427 static void
3428 zio_inherit_allocator(zio_t *pio, zio_t *cio)
3429 {
3430 	cio->io_allocator = pio->io_allocator;
3431 }
3432 
3433 static void
3434 zio_write_gang_member_ready(zio_t *zio)
3435 {
3436 	zio_t *pio = zio_unique_parent(zio);
3437 	dva_t *cdva = zio->io_bp->blk_dva;
3438 	dva_t *pdva = pio->io_bp->blk_dva;
3439 	uint64_t asize;
3440 	zio_t *gio __maybe_unused = zio->io_gang_leader;
3441 
3442 	if (BP_IS_HOLE(zio->io_bp))
3443 		return;
3444 
3445 	/*
3446 	 * If we're getting direct-invoked from zio_write_gang_block(),
3447 	 * the bp_orig will be set.
3448 	 */
3449 	ASSERT(BP_IS_HOLE(&zio->io_bp_orig) ||
3450 	    zio->io_flags & ZIO_FLAG_PREALLOCATED);
3451 
3452 	ASSERT(zio->io_child_type == ZIO_CHILD_GANG);
3453 	ASSERT3U(zio->io_prop.zp_copies, ==, gio->io_prop.zp_copies);
3454 	ASSERT3U(zio->io_prop.zp_copies, <=, BP_GET_NDVAS(zio->io_bp));
3455 	ASSERT3U(pio->io_prop.zp_copies, <=, BP_GET_NDVAS(pio->io_bp));
3456 	VERIFY3U(BP_GET_NDVAS(zio->io_bp), <=, BP_GET_NDVAS(pio->io_bp));
3457 
3458 	mutex_enter(&pio->io_lock);
3459 	for (int d = 0; d < BP_GET_NDVAS(zio->io_bp); d++) {
3460 		ASSERT(DVA_GET_GANG(&pdva[d]));
3461 		asize = DVA_GET_ASIZE(&pdva[d]);
3462 		asize += DVA_GET_ASIZE(&cdva[d]);
3463 		DVA_SET_ASIZE(&pdva[d], asize);
3464 	}
3465 	mutex_exit(&pio->io_lock);
3466 }
3467 
3468 static void
3469 zio_write_gang_done(zio_t *zio)
3470 {
3471 	/*
3472 	 * The io_abd field will be NULL for a zio with no data.  The io_flags
3473 	 * will initially have the ZIO_FLAG_NODATA bit flag set, but we can't
3474 	 * check for it here as it is cleared in zio_ready.
3475 	 */
3476 	if (zio->io_abd != NULL)
3477 		abd_free(zio->io_abd);
3478 }
3479 
3480 static void
3481 zio_update_feature(void *arg, dmu_tx_t *tx)
3482 {
3483 	spa_t *spa = dmu_tx_pool(tx)->dp_spa;
3484 	spa_feature_incr(spa, (spa_feature_t)(uintptr_t)arg, tx);
3485 }
3486 
3487 static zio_t *
3488 zio_write_gang_block(zio_t *pio, metaslab_class_t *mc)
3489 {
3490 	spa_t *spa = pio->io_spa;
3491 	blkptr_t *bp = pio->io_bp;
3492 	zio_t *gio = pio->io_gang_leader;
3493 	zio_t *zio;
3494 	zio_gang_node_t *gn, **gnpp;
3495 	zio_gbh_phys_t *gbh;
3496 	abd_t *gbh_abd;
3497 	uint64_t txg = pio->io_txg;
3498 	uint64_t resid = pio->io_size;
3499 	zio_prop_t zp;
3500 	int error;
3501 	boolean_t has_data = !(pio->io_flags & ZIO_FLAG_NODATA);
3502 
3503 	/*
3504 	 * Store multiple copies of the GBH, so that we can still traverse
3505 	 * all the data (e.g. to free or scrub) even if a block is damaged.
3506 	 * This value respects the redundant_metadata property.
3507 	 */
3508 	int gbh_copies = gio->io_prop.zp_gang_copies;
3509 	if (gbh_copies == 0) {
3510 		/*
3511 		 * This should only happen in the case where we're filling in
3512 		 * DDT entries for a parent that wants more copies than the DDT
3513 		 * has.  In that case, we cannot gang without creating a mixed
3514 		 * blkptr, which is illegal.
3515 		 */
3516 		ASSERT3U(gio->io_child_type, ==, ZIO_CHILD_DDT);
3517 		pio->io_error = EAGAIN;
3518 		return (pio);
3519 	}
3520 	ASSERT3S(gbh_copies, >, 0);
3521 	ASSERT3S(gbh_copies, <=, SPA_DVAS_PER_BP);
3522 
3523 	ASSERT(ZIO_HAS_ALLOCATOR(pio));
3524 	int flags = METASLAB_GANG_HEADER;
3525 	if (pio->io_flags & ZIO_FLAG_ALLOC_THROTTLED) {
3526 		ASSERT(pio->io_priority == ZIO_PRIORITY_ASYNC_WRITE);
3527 		ASSERT(has_data);
3528 
3529 		flags |= METASLAB_ASYNC_ALLOC;
3530 	}
3531 
3532 	uint64_t gangblocksize = SPA_OLD_GANGBLOCKSIZE;
3533 	uint64_t candidate = gangblocksize;
3534 	error = metaslab_alloc_range(spa, mc, gangblocksize, gangblocksize,
3535 	    bp, gbh_copies, txg, pio == gio ? NULL : gio->io_bp, flags,
3536 	    ZIO_ALLOC_LIST(pio), pio->io_allocator, pio, &candidate);
3537 	if (error) {
3538 		pio->io_error = error;
3539 		return (pio);
3540 	}
3541 	if (spa_feature_is_active(spa, SPA_FEATURE_DYNAMIC_GANG_HEADER))
3542 		gangblocksize = candidate;
3543 
3544 	if (pio == gio) {
3545 		gnpp = &gio->io_gang_tree;
3546 	} else {
3547 		gnpp = pio->io_private;
3548 		ASSERT(pio->io_ready == zio_write_gang_member_ready);
3549 	}
3550 
3551 	gn = zio_gang_node_alloc(gnpp, gangblocksize);
3552 	gbh = gn->gn_gbh;
3553 	memset(gbh, 0, gangblocksize);
3554 	gbh_abd = abd_get_from_buf(gbh, gangblocksize);
3555 
3556 	/*
3557 	 * Create the gang header.
3558 	 */
3559 	zio = zio_rewrite(pio, spa, txg, bp, gbh_abd, gangblocksize,
3560 	    zio_write_gang_done, NULL, pio->io_priority,
3561 	    ZIO_GANG_CHILD_FLAGS(pio), &pio->io_bookmark);
3562 
3563 	zio_inherit_allocator(pio, zio);
3564 	if (pio->io_flags & ZIO_FLAG_ALLOC_THROTTLED) {
3565 		boolean_t more;
3566 		VERIFY(metaslab_class_throttle_reserve(mc, zio->io_allocator,
3567 		    gbh_copies, zio->io_size, B_TRUE, &more));
3568 		zio->io_flags |= ZIO_FLAG_ALLOC_THROTTLED;
3569 	}
3570 
3571 	/*
3572 	 * Create and nowait the gang children. First, we try to do
3573 	 * opportunistic allocations. If that fails to generate enough
3574 	 * space, we fall back to normal zio_write calls for nested gang.
3575 	 */
3576 	int g;
3577 	boolean_t any_failed = B_FALSE;
3578 	for (g = 0; resid != 0; g++) {
3579 		flags &= METASLAB_ASYNC_ALLOC;
3580 		flags |= METASLAB_GANG_CHILD;
3581 		zp.zp_checksum = gio->io_prop.zp_checksum;
3582 		zp.zp_compress = ZIO_COMPRESS_OFF;
3583 		zp.zp_complevel = gio->io_prop.zp_complevel;
3584 		zp.zp_type = zp.zp_storage_type = DMU_OT_NONE;
3585 		zp.zp_level = 0;
3586 		zp.zp_copies = gio->io_prop.zp_copies;
3587 		zp.zp_gang_copies = gio->io_prop.zp_gang_copies;
3588 		zp.zp_dedup = B_FALSE;
3589 		zp.zp_dedup_verify = B_FALSE;
3590 		zp.zp_nopwrite = B_FALSE;
3591 		zp.zp_encrypt = gio->io_prop.zp_encrypt;
3592 		zp.zp_byteorder = gio->io_prop.zp_byteorder;
3593 		zp.zp_direct_write = B_FALSE;
3594 		memset(zp.zp_salt, 0, ZIO_DATA_SALT_LEN);
3595 		memset(zp.zp_iv, 0, ZIO_DATA_IV_LEN);
3596 		memset(zp.zp_mac, 0, ZIO_DATA_MAC_LEN);
3597 
3598 		uint64_t min_size = zio_roundup_alloc_size(spa,
3599 		    resid / (gbh_nblkptrs(gangblocksize) - g));
3600 		min_size = MIN(min_size, resid);
3601 		bp = &((blkptr_t *)gbh)[g];
3602 
3603 		zio_alloc_list_t cio_list;
3604 		metaslab_trace_init(&cio_list);
3605 		uint64_t allocated_size = UINT64_MAX;
3606 		error = metaslab_alloc_range(spa, mc, min_size, resid,
3607 		    bp, gio->io_prop.zp_copies, txg, NULL,
3608 		    flags, &cio_list, zio->io_allocator, NULL, &allocated_size);
3609 
3610 		boolean_t allocated = error == 0;
3611 		any_failed |= !allocated;
3612 
3613 		uint64_t psize = allocated ? MIN(resid, allocated_size) :
3614 		    min_size;
3615 		ASSERT3U(psize, >=, min_size);
3616 
3617 		zio_t *cio = zio_write(zio, spa, txg, bp, has_data ?
3618 		    abd_get_offset(pio->io_abd, pio->io_size - resid) : NULL,
3619 		    psize, psize, &zp, zio_write_gang_member_ready, NULL,
3620 		    zio_write_gang_done, &gn->gn_child[g], pio->io_priority,
3621 		    ZIO_GANG_CHILD_FLAGS(pio) |
3622 		    (allocated ? ZIO_FLAG_PREALLOCATED : 0), &pio->io_bookmark);
3623 
3624 		resid -= psize;
3625 		zio_inherit_allocator(zio, cio);
3626 		if (allocated) {
3627 			metaslab_trace_move(&cio_list, ZIO_ALLOC_LIST(cio));
3628 			metaslab_group_alloc_increment_all(spa,
3629 			    &cio->io_bp_orig, zio->io_allocator, flags, psize,
3630 			    cio);
3631 		}
3632 		/*
3633 		 * We do not reserve for the child writes, since we already
3634 		 * reserved for the parent.  Unreserve though will be called
3635 		 * for individual children.  We can do this since sum of all
3636 		 * child's physical sizes is equal to parent's physical size.
3637 		 * It would not work for potentially bigger allocation sizes.
3638 		 */
3639 
3640 		zio_nowait(cio);
3641 	}
3642 
3643 	/*
3644 	 * If we used more gang children than the old limit, we must already be
3645 	 * using the new headers. No need to update anything, just move on.
3646 	 *
3647 	 * Otherwise, we might be in a case where we need to turn on the new
3648 	 * feature, so we check that. We enable the new feature if we didn't
3649 	 * manage to fit everything into 3 gang children and we could have
3650 	 * written more than that.
3651 	 */
3652 	if (g > gbh_nblkptrs(SPA_OLD_GANGBLOCKSIZE)) {
3653 		ASSERT(spa_feature_is_active(spa,
3654 		    SPA_FEATURE_DYNAMIC_GANG_HEADER));
3655 	} else if (any_failed && candidate > SPA_OLD_GANGBLOCKSIZE &&
3656 	    spa_feature_is_enabled(spa, SPA_FEATURE_DYNAMIC_GANG_HEADER) &&
3657 	    !spa_feature_is_active(spa, SPA_FEATURE_DYNAMIC_GANG_HEADER)) {
3658 		dmu_tx_t *tx = dmu_tx_create_assigned(spa->spa_dsl_pool,
3659 		    MAX(txg, spa_syncing_txg(spa) + 1));
3660 		dsl_sync_task_nowait(spa->spa_dsl_pool,
3661 		    zio_update_feature,
3662 		    (void *)SPA_FEATURE_DYNAMIC_GANG_HEADER, tx);
3663 		dmu_tx_commit(tx);
3664 	}
3665 
3666 	/*
3667 	 * Set pio's pipeline to just wait for zio to finish.
3668 	 */
3669 	pio->io_pipeline = ZIO_INTERLOCK_PIPELINE;
3670 
3671 	zio_nowait(zio);
3672 
3673 	return (pio);
3674 }
3675 
3676 /*
3677  * The zio_nop_write stage in the pipeline determines if allocating a
3678  * new bp is necessary.  The nopwrite feature can handle writes in
3679  * either syncing or open context (i.e. zil writes) and as a result is
3680  * mutually exclusive with dedup.
3681  *
3682  * By leveraging a cryptographically secure checksum, such as SHA256, we
3683  * can compare the checksums of the new data and the old to determine if
3684  * allocating a new block is required.  Note that our requirements for
3685  * cryptographic strength are fairly weak: there can't be any accidental
3686  * hash collisions, but we don't need to be secure against intentional
3687  * (malicious) collisions.  To trigger a nopwrite, you have to be able
3688  * to write the file to begin with, and triggering an incorrect (hash
3689  * collision) nopwrite is no worse than simply writing to the file.
3690  * That said, there are no known attacks against the checksum algorithms
3691  * used for nopwrite, assuming that the salt and the checksums
3692  * themselves remain secret.
3693  */
3694 static zio_t *
3695 zio_nop_write(zio_t *zio)
3696 {
3697 	blkptr_t *bp = zio->io_bp;
3698 	blkptr_t *bp_orig = &zio->io_bp_orig;
3699 	zio_prop_t *zp = &zio->io_prop;
3700 
3701 	ASSERT(BP_IS_HOLE(bp));
3702 	ASSERT0(BP_GET_LEVEL(bp));
3703 	ASSERT(!(zio->io_flags & ZIO_FLAG_IO_REWRITE));
3704 	ASSERT(zp->zp_nopwrite);
3705 	ASSERT(!zp->zp_dedup);
3706 	ASSERT0P(zio->io_bp_override);
3707 	ASSERT(IO_IS_ALLOCATING(zio));
3708 
3709 	/*
3710 	 * Check to see if the original bp and the new bp have matching
3711 	 * characteristics (i.e. same checksum, compression algorithms, etc).
3712 	 * If they don't then just continue with the pipeline which will
3713 	 * allocate a new bp.
3714 	 */
3715 	if (BP_IS_HOLE(bp_orig) ||
3716 	    !(zio_checksum_table[BP_GET_CHECKSUM(bp)].ci_flags &
3717 	    ZCHECKSUM_FLAG_NOPWRITE) ||
3718 	    BP_IS_ENCRYPTED(bp) || BP_IS_ENCRYPTED(bp_orig) ||
3719 	    BP_GET_CHECKSUM(bp) != BP_GET_CHECKSUM(bp_orig) ||
3720 	    BP_GET_COMPRESS(bp) != BP_GET_COMPRESS(bp_orig) ||
3721 	    BP_GET_DEDUP(bp) != BP_GET_DEDUP(bp_orig) ||
3722 	    zp->zp_copies != BP_GET_NDVAS(bp_orig))
3723 		return (zio);
3724 
3725 	/*
3726 	 * If the checksums match then reset the pipeline so that we
3727 	 * avoid allocating a new bp and issuing any I/O.
3728 	 */
3729 	if (ZIO_CHECKSUM_EQUAL(bp->blk_cksum, bp_orig->blk_cksum)) {
3730 		ASSERT(zio_checksum_table[zp->zp_checksum].ci_flags &
3731 		    ZCHECKSUM_FLAG_NOPWRITE);
3732 		ASSERT3U(BP_GET_PSIZE(bp), ==, BP_GET_PSIZE(bp_orig));
3733 		ASSERT3U(BP_GET_LSIZE(bp), ==, BP_GET_LSIZE(bp_orig));
3734 		ASSERT(zp->zp_compress != ZIO_COMPRESS_OFF);
3735 		ASSERT3U(bp->blk_prop, ==, bp_orig->blk_prop);
3736 
3737 		/*
3738 		 * If we're overwriting a block that is currently on an
3739 		 * indirect vdev, then ignore the nopwrite request and
3740 		 * allow a new block to be allocated on a concrete vdev.
3741 		 */
3742 		spa_config_enter(zio->io_spa, SCL_VDEV, FTAG, RW_READER);
3743 		for (int d = 0; d < BP_GET_NDVAS(bp_orig); d++) {
3744 			vdev_t *tvd = vdev_lookup_top(zio->io_spa,
3745 			    DVA_GET_VDEV(&bp_orig->blk_dva[d]));
3746 			if (tvd->vdev_ops == &vdev_indirect_ops) {
3747 				spa_config_exit(zio->io_spa, SCL_VDEV, FTAG);
3748 				return (zio);
3749 			}
3750 		}
3751 		spa_config_exit(zio->io_spa, SCL_VDEV, FTAG);
3752 
3753 		*bp = *bp_orig;
3754 		zio->io_pipeline = ZIO_INTERLOCK_PIPELINE;
3755 		zio->io_flags |= ZIO_FLAG_NOPWRITE;
3756 	}
3757 
3758 	return (zio);
3759 }
3760 
3761 /*
3762  * ==========================================================================
3763  * Block Reference Table
3764  * ==========================================================================
3765  */
3766 static zio_t *
3767 zio_brt_free(zio_t *zio)
3768 {
3769 	blkptr_t *bp;
3770 
3771 	bp = zio->io_bp;
3772 
3773 	if (BP_GET_LEVEL(bp) > 0 ||
3774 	    BP_IS_METADATA(bp) ||
3775 	    !brt_maybe_exists(zio->io_spa, bp)) {
3776 		return (zio);
3777 	}
3778 
3779 	if (!brt_entry_decref(zio->io_spa, bp)) {
3780 		/*
3781 		 * This isn't the last reference, so we cannot free
3782 		 * the data yet.
3783 		 */
3784 		zio->io_pipeline = ZIO_INTERLOCK_PIPELINE;
3785 	}
3786 
3787 	return (zio);
3788 }
3789 
3790 /*
3791  * ==========================================================================
3792  * Dedup
3793  * ==========================================================================
3794  */
3795 static void
3796 zio_ddt_child_read_done(zio_t *zio)
3797 {
3798 	blkptr_t *bp = zio->io_bp;
3799 	ddt_t *ddt;
3800 	ddt_entry_t *dde = zio->io_private;
3801 	zio_t *pio = zio_unique_parent(zio);
3802 
3803 	mutex_enter(&pio->io_lock);
3804 	ddt = ddt_select(zio->io_spa, bp);
3805 
3806 	if (zio->io_error == 0) {
3807 		ddt_phys_variant_t v = ddt_phys_select(ddt, dde, bp);
3808 		/* this phys variant doesn't need repair */
3809 		ddt_phys_clear(dde->dde_phys, v);
3810 	}
3811 
3812 	if (zio->io_error == 0 && dde->dde_io->dde_repair_abd == NULL)
3813 		dde->dde_io->dde_repair_abd = zio->io_abd;
3814 	else
3815 		abd_free(zio->io_abd);
3816 	mutex_exit(&pio->io_lock);
3817 }
3818 
3819 static zio_t *
3820 zio_ddt_read_start(zio_t *zio)
3821 {
3822 	blkptr_t *bp = zio->io_bp;
3823 
3824 	ASSERT(BP_GET_DEDUP(bp));
3825 	ASSERT(BP_GET_PSIZE(bp) == zio->io_size);
3826 	ASSERT(zio->io_child_type == ZIO_CHILD_LOGICAL);
3827 
3828 	if (zio->io_child_error[ZIO_CHILD_DDT]) {
3829 		ddt_t *ddt = ddt_select(zio->io_spa, bp);
3830 		ddt_entry_t *dde = ddt_repair_start(ddt, bp);
3831 		ddt_phys_variant_t v_self = ddt_phys_select(ddt, dde, bp);
3832 		ddt_univ_phys_t *ddp = dde->dde_phys;
3833 		blkptr_t blk;
3834 
3835 		ASSERT0P(zio->io_vsd);
3836 		zio->io_vsd = dde;
3837 
3838 		if (v_self == DDT_PHYS_NONE)
3839 			return (zio);
3840 
3841 		/* issue I/O for the other copies */
3842 		for (int p = 0; p < DDT_NPHYS(ddt); p++) {
3843 			ddt_phys_variant_t v = DDT_PHYS_VARIANT(ddt, p);
3844 
3845 			if (ddt_phys_birth(ddp, v) == 0 || v == v_self)
3846 				continue;
3847 
3848 			ddt_bp_create(ddt->ddt_checksum, &dde->dde_key,
3849 			    ddp, v, &blk);
3850 			zio_nowait(zio_read(zio, zio->io_spa, &blk,
3851 			    abd_alloc_for_io(zio->io_size, B_TRUE),
3852 			    zio->io_size, zio_ddt_child_read_done, dde,
3853 			    zio->io_priority, ZIO_DDT_CHILD_FLAGS(zio) |
3854 			    ZIO_FLAG_DONT_PROPAGATE, &zio->io_bookmark));
3855 		}
3856 		return (zio);
3857 	}
3858 
3859 	zio_nowait(zio_read(zio, zio->io_spa, bp,
3860 	    zio->io_abd, zio->io_size, NULL, NULL, zio->io_priority,
3861 	    ZIO_DDT_CHILD_FLAGS(zio), &zio->io_bookmark));
3862 
3863 	return (zio);
3864 }
3865 
3866 static zio_t *
3867 zio_ddt_read_done(zio_t *zio)
3868 {
3869 	blkptr_t *bp = zio->io_bp;
3870 
3871 	if (zio_wait_for_children(zio, ZIO_CHILD_DDT_BIT, ZIO_WAIT_DONE)) {
3872 		return (NULL);
3873 	}
3874 
3875 	ASSERT(BP_GET_DEDUP(bp));
3876 	ASSERT(BP_GET_PSIZE(bp) == zio->io_size);
3877 	ASSERT(zio->io_child_type == ZIO_CHILD_LOGICAL);
3878 
3879 	if (zio->io_child_error[ZIO_CHILD_DDT]) {
3880 		ddt_t *ddt = ddt_select(zio->io_spa, bp);
3881 		ddt_entry_t *dde = zio->io_vsd;
3882 		if (ddt == NULL) {
3883 			ASSERT(spa_load_state(zio->io_spa) != SPA_LOAD_NONE);
3884 			return (zio);
3885 		}
3886 		if (dde == NULL) {
3887 			zio->io_stage = ZIO_STAGE_DDT_READ_START >> 1;
3888 			zio_taskq_dispatch(zio, ZIO_TASKQ_ISSUE, B_FALSE);
3889 			return (NULL);
3890 		}
3891 		if (dde->dde_io->dde_repair_abd != NULL) {
3892 			abd_copy(zio->io_abd, dde->dde_io->dde_repair_abd,
3893 			    zio->io_size);
3894 			zio->io_child_error[ZIO_CHILD_DDT] = 0;
3895 		}
3896 		ddt_repair_done(ddt, dde);
3897 		zio->io_vsd = NULL;
3898 	}
3899 
3900 	ASSERT0P(zio->io_vsd);
3901 
3902 	return (zio);
3903 }
3904 
3905 static boolean_t
3906 zio_ddt_collision(zio_t *zio, ddt_t *ddt, ddt_entry_t *dde)
3907 {
3908 	spa_t *spa = zio->io_spa;
3909 	boolean_t do_raw = !!(zio->io_flags & ZIO_FLAG_RAW);
3910 
3911 	ASSERT(!(zio->io_bp_override && do_raw));
3912 
3913 	/*
3914 	 * Note: we compare the original data, not the transformed data,
3915 	 * because when zio->io_bp is an override bp, we will not have
3916 	 * pushed the I/O transforms.  That's an important optimization
3917 	 * because otherwise we'd compress/encrypt all dmu_sync() data twice.
3918 	 * However, we should never get a raw, override zio so in these
3919 	 * cases we can compare the io_abd directly. This is useful because
3920 	 * it allows us to do dedup verification even if we don't have access
3921 	 * to the original data (for instance, if the encryption keys aren't
3922 	 * loaded).
3923 	 */
3924 
3925 	for (int p = 0; p < DDT_NPHYS(ddt); p++) {
3926 		if (DDT_PHYS_IS_DITTO(ddt, p))
3927 			continue;
3928 
3929 		if (dde->dde_io == NULL)
3930 			continue;
3931 
3932 		/*
3933 		 * Lock dde_io to prevent the lead zio from completing
3934 		 * and freeing its ABD while we compare against it.
3935 		 */
3936 		mutex_enter(&dde->dde_io->dde_io_lock);
3937 		zio_t *lio = dde->dde_io->dde_lead_zio[p];
3938 		if (lio == NULL) {
3939 			mutex_exit(&dde->dde_io->dde_io_lock);
3940 			continue;
3941 		}
3942 		boolean_t collision;
3943 		if (do_raw) {
3944 			collision = lio->io_size != zio->io_size ||
3945 			    abd_cmp(zio->io_abd, lio->io_abd) != 0;
3946 		} else {
3947 			collision = lio->io_orig_size != zio->io_orig_size ||
3948 			    abd_cmp(zio->io_orig_abd, lio->io_orig_abd) != 0;
3949 		}
3950 		mutex_exit(&dde->dde_io->dde_io_lock);
3951 		return (collision);
3952 	}
3953 
3954 	for (int p = 0; p < DDT_NPHYS(ddt); p++) {
3955 		ddt_phys_variant_t v = DDT_PHYS_VARIANT(ddt, p);
3956 		uint64_t phys_birth = ddt_phys_birth(dde->dde_phys, v);
3957 
3958 		if (phys_birth != 0 && do_raw) {
3959 			blkptr_t blk = *zio->io_bp;
3960 			uint64_t psize;
3961 			abd_t *tmpabd;
3962 			int error;
3963 
3964 			ddt_bp_fill(dde->dde_phys, v, &blk, phys_birth);
3965 			psize = BP_GET_PSIZE(&blk);
3966 
3967 			if (psize != zio->io_size)
3968 				return (B_TRUE);
3969 
3970 			ddt_exit(ddt);
3971 
3972 			tmpabd = abd_alloc_for_io(psize, B_TRUE);
3973 
3974 			error = zio_wait(zio_read(NULL, spa, &blk, tmpabd,
3975 			    psize, NULL, NULL, ZIO_PRIORITY_SYNC_READ,
3976 			    ZIO_FLAG_CANFAIL | ZIO_FLAG_SPECULATIVE |
3977 			    ZIO_FLAG_RAW, &zio->io_bookmark));
3978 
3979 			if (error == 0) {
3980 				if (abd_cmp(tmpabd, zio->io_abd) != 0)
3981 					error = SET_ERROR(ENOENT);
3982 			}
3983 
3984 			abd_free(tmpabd);
3985 			ddt_enter(ddt);
3986 			return (error != 0);
3987 		} else if (phys_birth != 0) {
3988 			arc_buf_t *abuf = NULL;
3989 			arc_flags_t aflags = ARC_FLAG_WAIT;
3990 			blkptr_t blk = *zio->io_bp;
3991 			int error;
3992 
3993 			ddt_bp_fill(dde->dde_phys, v, &blk, phys_birth);
3994 
3995 			if (BP_GET_LSIZE(&blk) != zio->io_orig_size)
3996 				return (B_TRUE);
3997 
3998 			ddt_exit(ddt);
3999 
4000 			error = arc_read(NULL, spa, &blk,
4001 			    arc_getbuf_func, &abuf, ZIO_PRIORITY_SYNC_READ,
4002 			    ZIO_FLAG_CANFAIL | ZIO_FLAG_SPECULATIVE,
4003 			    &aflags, &zio->io_bookmark);
4004 
4005 			if (error == 0) {
4006 				if (abd_cmp_buf(zio->io_orig_abd, abuf->b_data,
4007 				    zio->io_orig_size) != 0)
4008 					error = SET_ERROR(ENOENT);
4009 				arc_buf_destroy(abuf, &abuf);
4010 			}
4011 
4012 			ddt_enter(ddt);
4013 			return (error != 0);
4014 		}
4015 	}
4016 
4017 	return (B_FALSE);
4018 }
4019 
4020 static void
4021 zio_ddt_child_write_done(zio_t *zio)
4022 {
4023 	ddt_t *ddt = ddt_select(zio->io_spa, zio->io_bp);
4024 	ddt_entry_t *dde = zio->io_private;
4025 
4026 	zio_link_t *zl = NULL;
4027 	ASSERT3P(zio_walk_parents(zio, &zl), !=, NULL);
4028 
4029 	int p = DDT_PHYS_FOR_COPIES(ddt, zio->io_prop.zp_copies);
4030 	ddt_phys_variant_t v = DDT_PHYS_VARIANT(ddt, p);
4031 	ddt_univ_phys_t *ddp = dde->dde_phys;
4032 
4033 	mutex_enter(&dde->dde_io->dde_io_lock);
4034 
4035 	/* we're the lead, so once we're done there's no one else outstanding */
4036 	if (dde->dde_io->dde_lead_zio[p] == zio)
4037 		dde->dde_io->dde_lead_zio[p] = NULL;
4038 
4039 	ddt_univ_phys_t *orig = &dde->dde_io->dde_orig_phys;
4040 
4041 	if (zio->io_error != 0) {
4042 		/*
4043 		 * The write failed, so we're about to abort the entire IO
4044 		 * chain. We need to revert the entry back to what it was at
4045 		 * the last time it was successfully extended.
4046 		 */
4047 		ddt_phys_unextend(ddp, orig, v);
4048 		ddt_phys_clear(orig, v);
4049 
4050 		mutex_exit(&dde->dde_io->dde_io_lock);
4051 
4052 		/*
4053 		 * Undo the optimistic refcount increments that were done in
4054 		 * zio_ddt_write() for all non-DDT-child parents. Since errors
4055 		 * are rare, taking the global lock here is acceptable.
4056 		 */
4057 		ddt_enter(ddt);
4058 		zio_t *pio;
4059 		zl = NULL;
4060 		while ((pio = zio_walk_parents(zio, &zl)) != NULL) {
4061 			if (!(pio->io_flags & ZIO_FLAG_DDT_CHILD))
4062 				ddt_phys_decref(ddp, v);
4063 		}
4064 		ddt_exit(ddt);
4065 		return;
4066 	}
4067 
4068 	/*
4069 	 * We've successfully added new DVAs to the entry. Clear the saved
4070 	 * state or, if there's still outstanding IO, remember it so we can
4071 	 * revert to a known good state if that IO fails.
4072 	 */
4073 	if (dde->dde_io->dde_lead_zio[p] == NULL)
4074 		ddt_phys_clear(orig, v);
4075 	else
4076 		ddt_phys_copy(orig, ddp, v);
4077 
4078 	mutex_exit(&dde->dde_io->dde_io_lock);
4079 }
4080 
4081 static void
4082 zio_ddt_child_write_ready(zio_t *zio)
4083 {
4084 	ddt_t *ddt = ddt_select(zio->io_spa, zio->io_bp);
4085 	ddt_entry_t *dde = zio->io_private;
4086 
4087 	zio_link_t *zl = NULL;
4088 	ASSERT3P(zio_walk_parents(zio, &zl), !=, NULL);
4089 
4090 	int p = DDT_PHYS_FOR_COPIES(ddt, zio->io_prop.zp_copies);
4091 	ddt_phys_variant_t v = DDT_PHYS_VARIANT(ddt, p);
4092 
4093 	if (ddt_phys_is_gang(dde->dde_phys, v)) {
4094 		for (int i = 0; i < BP_GET_NDVAS(zio->io_bp); i++) {
4095 			dva_t *d = &zio->io_bp->blk_dva[i];
4096 			metaslab_group_alloc_decrement(zio->io_spa,
4097 			    DVA_GET_VDEV(d), zio->io_allocator,
4098 			    METASLAB_ASYNC_ALLOC, zio->io_size, zio);
4099 		}
4100 		zio->io_error = EAGAIN;
4101 	}
4102 
4103 	if (zio->io_error != 0)
4104 		return;
4105 
4106 	mutex_enter(&dde->dde_io->dde_io_lock);
4107 
4108 	ddt_phys_extend(dde->dde_phys, v, zio->io_bp);
4109 
4110 	zio_t *pio;
4111 	zl = NULL;
4112 	while ((pio = zio_walk_parents(zio, &zl)) != NULL) {
4113 		if (!(pio->io_flags & ZIO_FLAG_DDT_CHILD))
4114 			ddt_bp_fill(dde->dde_phys, v, pio->io_bp, zio->io_txg);
4115 	}
4116 
4117 	mutex_exit(&dde->dde_io->dde_io_lock);
4118 }
4119 
4120 static zio_t *
4121 zio_ddt_write(zio_t *zio)
4122 {
4123 	spa_t *spa = zio->io_spa;
4124 	blkptr_t *bp = zio->io_bp;
4125 	uint64_t txg = zio->io_txg;
4126 	zio_prop_t *zp = &zio->io_prop;
4127 	ddt_t *ddt = ddt_select(spa, bp);
4128 	ddt_entry_t *dde;
4129 
4130 	ASSERT(BP_GET_DEDUP(bp));
4131 	ASSERT(BP_GET_CHECKSUM(bp) == zp->zp_checksum);
4132 	ASSERT(BP_IS_HOLE(bp) || zio->io_bp_override);
4133 	ASSERT(!(zio->io_bp_override && (zio->io_flags & ZIO_FLAG_RAW)));
4134 	/*
4135 	 * Deduplication will not take place for Direct I/O writes. The
4136 	 * ddt_tree will be emptied in syncing context. Direct I/O writes take
4137 	 * place in the open-context. Direct I/O write can not attempt to
4138 	 * modify the ddt_tree while issuing out a write.
4139 	 */
4140 	ASSERT3B(zio->io_prop.zp_direct_write, ==, B_FALSE);
4141 
4142 	ddt_enter(ddt);
4143 	/*
4144 	 * Search DDT for matching entry.  Skip DVAs verification here, since
4145 	 * they can go only from override, and once we get here the override
4146 	 * pointer can't have "D" flag to be confused with pruned DDT entries.
4147 	 */
4148 	IMPLY(zio->io_bp_override, !BP_GET_DEDUP(zio->io_bp_override));
4149 	dde = ddt_lookup(ddt, bp, B_FALSE);
4150 	if (dde == NULL) {
4151 		/* DDT size is over its quota so no new entries */
4152 		ddt_exit(ddt);
4153 		zp->zp_dedup = B_FALSE;
4154 		BP_SET_DEDUP(bp, B_FALSE);
4155 		if (zio->io_bp_override == NULL)
4156 			zio->io_pipeline = ZIO_WRITE_PIPELINE;
4157 		return (zio);
4158 	}
4159 
4160 	if (zp->zp_dedup_verify && zio_ddt_collision(zio, ddt, dde)) {
4161 		/*
4162 		 * If we're using a weak checksum, upgrade to a strong checksum
4163 		 * and try again.  If we're already using a strong checksum,
4164 		 * we can't resolve it, so just convert to an ordinary write.
4165 		 * (And automatically e-mail a paper to Nature?)
4166 		 */
4167 		ddt_exit(ddt);
4168 		if (!(zio_checksum_table[zp->zp_checksum].ci_flags &
4169 		    ZCHECKSUM_FLAG_DEDUP)) {
4170 			zp->zp_checksum = spa_dedup_checksum(spa);
4171 			zio_pop_transforms(zio);
4172 			zio->io_stage = ZIO_STAGE_OPEN;
4173 			BP_ZERO(bp);
4174 		} else {
4175 			zp->zp_dedup = B_FALSE;
4176 			BP_SET_DEDUP(bp, B_FALSE);
4177 		}
4178 		ASSERT(!BP_GET_DEDUP(bp));
4179 		zio->io_pipeline = ZIO_WRITE_PIPELINE;
4180 		return (zio);
4181 	}
4182 
4183 	int p = DDT_PHYS_FOR_COPIES(ddt, zp->zp_copies);
4184 	ddt_phys_variant_t v = DDT_PHYS_VARIANT(ddt, p);
4185 
4186 	/*
4187 	 * In the common cases, at this point we have a regular BP with no
4188 	 * allocated DVAs, and the corresponding DDT entry for its checksum.
4189 	 * Our goal is to fill the BP with enough DVAs to satisfy its copies=
4190 	 * requirement.
4191 	 *
4192 	 * One of three things needs to happen to fulfill this:
4193 	 *
4194 	 * - if the DDT entry has enough DVAs to satisfy the BP, we just copy
4195 	 *   them out of the entry and return;
4196 	 *
4197 	 * - if the DDT entry has no DVAs (ie its brand new), then we have to
4198 	 *   issue the write as normal so that DVAs can be allocated and the
4199 	 *   data land on disk. We then copy the DVAs into the DDT entry on
4200 	 *   return.
4201 	 *
4202 	 * - if the DDT entry has some DVAs, but too few, we have to issue the
4203 	 *   write, adjusted to have allocate fewer copies. When it returns, we
4204 	 *   add the new DVAs to the DDT entry, and update the BP to have the
4205 	 *   full amount it originally requested.
4206 	 *
4207 	 * In all cases, if there's already a writing IO in flight, we need to
4208 	 * defer the action until after the write is done. If our action is to
4209 	 * write, we need to adjust our request for additional DVAs to match
4210 	 * what will be in the DDT entry after it completes. In this way every
4211 	 * IO can be guaranteed to recieve enough DVAs simply by joining the
4212 	 * end of the chain and letting the sequence play out.
4213 	 */
4214 
4215 	/* Number of DVAs requested by the IO. */
4216 	uint8_t need_dvas = zp->zp_copies;
4217 	/* Number of DVAs in outstanding writes for this dde. */
4218 	uint8_t parent_dvas = 0;
4219 
4220 	/*
4221 	 * What we do next depends on whether or not there's IO outstanding
4222 	 * that will update this entry. If dde_io exists, we need to hold
4223 	 * its lock to safely check and use dde_lead_zio.
4224 	 */
4225 	ddt_entry_io_t *dde_io = dde->dde_io;
4226 	if (dde_io != NULL)
4227 		mutex_enter(&dde_io->dde_io_lock);
4228 
4229 	/*
4230 	 * Number of DVAs in the DDT entry. If the BP is encrypted we ignore
4231 	 * the third one as normal.
4232 	 *
4233 	 * Must be computed after taking dde_io_lock (if held) to avoid
4234 	 * racing with ddt_phys_unextend() in zio_ddt_child_write_done()
4235 	 * error path, which can zero DVAs under dde_io_lock. Without the
4236 	 * lock, a stale have_dvas causes ddt_bp_fill() to copy a zeroed
4237 	 * DVA into the BP, producing a hole that reads back as zeros.
4238 	 */
4239 	ddt_univ_phys_t *ddp = dde->dde_phys;
4240 	int have_dvas = ddt_phys_dva_count(ddp, v, BP_IS_ENCRYPTED(bp));
4241 	IMPLY(have_dvas == 0, ddt_phys_birth(ddp, v) == 0);
4242 	boolean_t is_ganged = ddt_phys_is_gang(ddp, v);
4243 
4244 	if (dde_io == NULL || dde_io->dde_lead_zio[p] == NULL) {
4245 		/*
4246 		 * No IO outstanding, so we only need to worry about ourselves.
4247 		 */
4248 
4249 		/*
4250 		 * Override BPs bring their own DVAs and their own problems.
4251 		 */
4252 		if (zio->io_bp_override) {
4253 			/*
4254 			 * For a brand-new entry, all the work has been done
4255 			 * for us, and we can just fill it out from the provided
4256 			 * block and leave.
4257 			 */
4258 			if (have_dvas == 0) {
4259 				if (dde_io != NULL)
4260 					mutex_exit(&dde_io->dde_io_lock);
4261 				ASSERT(BP_GET_BIRTH(bp) == txg);
4262 				ASSERT(BP_EQUAL(bp, zio->io_bp_override));
4263 				ddt_phys_extend(ddp, v, bp);
4264 				ddt_phys_addref(ddp, v);
4265 				ddt_exit(ddt);
4266 				return (zio);
4267 			}
4268 
4269 			/*
4270 			 * If we already have this entry, then we want to treat
4271 			 * it like a regular write. To do this we just wipe
4272 			 * them out and proceed like a regular write.
4273 			 *
4274 			 * Even if there are some DVAs in the entry, we still
4275 			 * have to clear them out. We can't use them to fill
4276 			 * out the dedup entry, as they are all referenced
4277 			 * together by a bp already on disk, and will be freed
4278 			 * as a group.
4279 			 */
4280 			BP_ZERO_DVAS(bp);
4281 			BP_SET_BIRTH(bp, 0, 0);
4282 		}
4283 
4284 		/*
4285 		 * If there are enough DVAs in the entry to service our request,
4286 		 * then we can just use them as-is.
4287 		 */
4288 		if (have_dvas >= need_dvas) {
4289 			if (dde_io != NULL)
4290 				mutex_exit(&dde_io->dde_io_lock);
4291 
4292 			/*
4293 			 * For rewrite operations, try preserving the original
4294 			 * logical birth time.  If the result matches the
4295 			 * original BP, this becomes a NOP.
4296 			 */
4297 			if (zp->zp_rewrite) {
4298 				uint64_t orig_logical_birth =
4299 				    BP_GET_LOGICAL_BIRTH(&zio->io_bp_orig);
4300 				ddt_bp_fill(ddp, v, bp, orig_logical_birth);
4301 				if (BP_EQUAL(bp, &zio->io_bp_orig)) {
4302 					/* We can skip accounting. */
4303 					ddt_exit(ddt);
4304 					zio->io_flags |= ZIO_FLAG_NOPWRITE;
4305 					return (zio);
4306 				}
4307 			}
4308 
4309 			ddt_bp_fill(ddp, v, bp, txg);
4310 			ddt_phys_addref(ddp, v);
4311 			ddt_exit(ddt);
4312 			return (zio);
4313 		}
4314 
4315 		/*
4316 		 * Otherwise, we have to issue IO to fill the entry up to the
4317 		 * amount we need.
4318 		 */
4319 		need_dvas -= have_dvas;
4320 	} else {
4321 		/*
4322 		 * There's a write in-flight. If there's already enough DVAs on
4323 		 * the entry, then either there were already enough to start
4324 		 * with, or the in-flight IO is between READY and DONE, and so
4325 		 * has extended the entry with new DVAs. Either way, we don't
4326 		 * need to do anything, we can just slot in behind it.
4327 		 */
4328 
4329 		if (zio->io_bp_override) {
4330 			/*
4331 			 * If there's a write out, then we're soon going to
4332 			 * have our own copies of this block, so clear out the
4333 			 * override block and treat it as a regular dedup
4334 			 * write. See comment above.
4335 			 */
4336 			BP_ZERO_DVAS(bp);
4337 			BP_SET_BIRTH(bp, 0, 0);
4338 		}
4339 
4340 		if (have_dvas >= need_dvas) {
4341 			/*
4342 			 * A minor point: there might already be enough
4343 			 * committed DVAs in the entry to service our request,
4344 			 * but we don't know which are completed and which are
4345 			 * allocated but not yet written. In this case, should
4346 			 * the IO for the new DVAs fail, we will be on the end
4347 			 * of the IO chain and will also recieve an error, even
4348 			 * though our request could have been serviced.
4349 			 *
4350 			 * This is an extremely rare case, as it requires the
4351 			 * original block to be copied with a request for a
4352 			 * larger number of DVAs, then copied again requesting
4353 			 * the same (or already fulfilled) number of DVAs while
4354 			 * the first request is active, and then that first
4355 			 * request errors. In return, the logic required to
4356 			 * catch and handle it is complex. For now, I'm just
4357 			 * not going to bother with it.
4358 			 */
4359 
4360 			/*
4361 			 * We always fill the bp here as we may have arrived
4362 			 * after the in-flight write has passed READY, and so
4363 			 * missed out.
4364 			 */
4365 			ddt_bp_fill(ddp, v, bp, txg);
4366 piggyback:
4367 			zio_add_child(zio, dde_io->dde_lead_zio[p]);
4368 
4369 			/*
4370 			 * Optimistically increment refcount for this parent.
4371 			 * If the write fails, zio_ddt_child_write_done() will
4372 			 * decrement for all non-DDT-child parents.
4373 			 */
4374 			ddt_phys_addref(ddp, v);
4375 			mutex_exit(&dde_io->dde_io_lock);
4376 			ddt_exit(ddt);
4377 			return (zio);
4378 		}
4379 
4380 		/*
4381 		 * There's not enough in the entry yet, so we need to look at
4382 		 * the write in-flight and see how many DVAs it will have once
4383 		 * it completes.
4384 		 *
4385 		 * The in-flight write has potentially had its copies request
4386 		 * reduced (if we're filling out an existing entry), so we need
4387 		 * to reach in and get the original write to find out what it is
4388 		 * expecting.
4389 		 *
4390 		 * Note that the parent of the lead zio will always have the
4391 		 * highest zp_copies of any zio in the chain, because ones that
4392 		 * can be serviced without additional IO are always added to
4393 		 * the back of the chain.
4394 		 */
4395 		zio_link_t *zl = NULL;
4396 		zio_t *pio =
4397 		    zio_walk_parents(dde->dde_io->dde_lead_zio[p], &zl);
4398 		ASSERT(pio);
4399 		parent_dvas = pio->io_prop.zp_copies;
4400 
4401 		if (parent_dvas >= need_dvas)
4402 			goto piggyback;
4403 
4404 		/*
4405 		 * Still not enough, so we will need to issue to get the
4406 		 * shortfall.
4407 		 */
4408 		need_dvas -= parent_dvas;
4409 	}
4410 
4411 	if (is_ganged) {
4412 		if (dde_io != NULL)
4413 			mutex_exit(&dde_io->dde_io_lock);
4414 		ddt_exit(ddt);
4415 		zp->zp_dedup = B_FALSE;
4416 		BP_SET_DEDUP(bp, B_FALSE);
4417 		zio->io_pipeline = ZIO_WRITE_PIPELINE;
4418 		return (zio);
4419 	}
4420 
4421 	/*
4422 	 * We need to write. We will create a new write with the copies
4423 	 * property adjusted to match the number of DVAs we need to grow
4424 	 * the DDT entry by to satisfy the request.
4425 	 */
4426 	zio_prop_t czp;
4427 	if (have_dvas > 0 || parent_dvas > 0) {
4428 		czp = *zp;
4429 		czp.zp_copies = need_dvas;
4430 		czp.zp_gang_copies = 0;
4431 		zp = &czp;
4432 	} else {
4433 		ASSERT3U(zp->zp_copies, ==, need_dvas);
4434 	}
4435 
4436 	zio_t *cio = zio_write(zio, spa, txg, bp, zio->io_orig_abd,
4437 	    zio->io_orig_size, zio->io_orig_size, zp,
4438 	    zio_ddt_child_write_ready, NULL,
4439 	    zio_ddt_child_write_done, dde, zio->io_priority,
4440 	    ZIO_DDT_CHILD_FLAGS(zio), &zio->io_bookmark);
4441 	zio_inherit_allocator(zio, cio);
4442 
4443 	zio_push_transform(cio, zio->io_abd, zio->io_size, 0, NULL);
4444 
4445 	/*
4446 	 * We are the new lead zio, because our parent has the highest
4447 	 * zp_copies that has been requested for this entry so far.
4448 	 */
4449 	if (dde_io == NULL) {
4450 		/*
4451 		 * New dde_io.  No lock needed since no other thread can have
4452 		 * a reference yet.
4453 		 */
4454 		ddt_alloc_entry_io(dde);
4455 		dde_io = dde->dde_io;
4456 		/*
4457 		 * First time out, take a copy of the stable entry to revert
4458 		 * to if there's an error (see zio_ddt_child_write_done())
4459 		 */
4460 		ddt_phys_copy(&dde_io->dde_orig_phys, dde->dde_phys, v);
4461 		dde_io->dde_lead_zio[p] = cio;
4462 	} else {
4463 		if (dde_io->dde_lead_zio[p] == NULL) {
4464 			/*
4465 			 * First time out, take a copy of the stable entry
4466 			 * to revert to if there's an error (see
4467 			 * zio_ddt_child_write_done())
4468 			 */
4469 			ddt_phys_copy(&dde_io->dde_orig_phys, dde->dde_phys,
4470 			    v);
4471 		} else {
4472 			/*
4473 			 * Make the existing chain our child, because it
4474 			 * cannot complete until we have.
4475 			 */
4476 			zio_add_child(cio, dde_io->dde_lead_zio[p]);
4477 		}
4478 		dde_io->dde_lead_zio[p] = cio;
4479 		mutex_exit(&dde_io->dde_io_lock);
4480 	}
4481 
4482 	/*
4483 	 * Optimistically increment the refcount for this dedup write.
4484 	 * If the write fails, zio_ddt_child_write_done() will decrement
4485 	 * for all non-DDT-child parents.
4486 	 */
4487 	ddt_phys_addref(ddp, v);
4488 
4489 	ddt_exit(ddt);
4490 
4491 	zio_nowait(cio);
4492 
4493 	return (zio);
4494 }
4495 
4496 static ddt_entry_t *freedde; /* for debugging */
4497 
4498 static zio_t *
4499 zio_ddt_free(zio_t *zio)
4500 {
4501 	spa_t *spa = zio->io_spa;
4502 	blkptr_t *bp = zio->io_bp;
4503 	ddt_t *ddt = ddt_select(spa, bp);
4504 	ddt_entry_t *dde = NULL;
4505 
4506 	ASSERT(BP_GET_DEDUP(bp));
4507 	ASSERT(zio->io_child_type == ZIO_CHILD_LOGICAL);
4508 
4509 	ddt_enter(ddt);
4510 	freedde = dde = ddt_lookup(ddt, bp, B_TRUE);
4511 	if (dde) {
4512 		ddt_phys_variant_t v = ddt_phys_select(ddt, dde, bp);
4513 		if (v != DDT_PHYS_NONE)
4514 			ddt_phys_decref(dde->dde_phys, v);
4515 		else
4516 			/*
4517 			 * No phys matches this BP; ddt_lookup() returned a
4518 			 * fresh, empty entry because the key is not in the
4519 			 * table at all (eg the original entry was pruned).
4520 			 * There is no reference to release, so we need to do
4521 			 * a normal (not dedup) free. Clear dde so we fall
4522 			 * into the block below.
4523 			 */
4524 			dde = NULL;
4525 	}
4526 	ddt_exit(ddt);
4527 
4528 	if (dde) {
4529 		/*
4530 		 * DDT entry found and the refcount has been decremented.
4531 		 * Stop the pipeline — there is nothing more to do right now.
4532 		 */
4533 		zio->io_pipeline = ZIO_INTERLOCK_PIPELINE;
4534 	} else {
4535 		/*
4536 		 * No DDT entry; the block must have been pruned from the
4537 		 * table.  Clear the DEDUP bit so it is treated as a normal
4538 		 * block from here on.  BRT_FREE and DVA_FREE follow in the
4539 		 * pipeline and will handle any cloned references and the
4540 		 * actual block free respectively, along with the gang stages
4541 		 * for a gang BP.
4542 		 *
4543 		 * Only flat (FDT) tables are ever pruned, so a miss against
4544 		 * a traditional table means the table and the BP disagree,
4545 		 * which should not be possible. The plain free below is
4546 		 * still the best we can do for this BP, but leave a trace.
4547 		 */
4548 		if (!(ddt->ddt_flags & DDT_FLAG_FLAT)) {
4549 			zfs_dbgmsg("%s: no matching traditional DDT phys for "
4550 			    "dedup BP DVA[0]=<%llu:%llx:%llx> phys_birth=%llu; "
4551 			    "freeing without a refcount decrement",
4552 			    spa_name(spa),
4553 			    (u_longlong_t)DVA_GET_VDEV(&bp->blk_dva[0]),
4554 			    (u_longlong_t)DVA_GET_OFFSET(&bp->blk_dva[0]),
4555 			    (u_longlong_t)DVA_GET_ASIZE(&bp->blk_dva[0]),
4556 			    (u_longlong_t)BP_GET_PHYSICAL_BIRTH(bp));
4557 		}
4558 		BP_SET_DEDUP(bp, 0);
4559 	}
4560 
4561 	return (zio);
4562 }
4563 
4564 /*
4565  * ==========================================================================
4566  * Allocate and free blocks
4567  * ==========================================================================
4568  */
4569 
4570 static zio_t *
4571 zio_io_to_allocate(metaslab_class_allocator_t *mca, boolean_t *more)
4572 {
4573 	zio_t *zio;
4574 
4575 	ASSERT(MUTEX_HELD(&mca->mca_lock));
4576 
4577 	zio = avl_first(&mca->mca_tree);
4578 	if (zio == NULL) {
4579 		*more = B_FALSE;
4580 		return (NULL);
4581 	}
4582 
4583 	ASSERT(IO_IS_ALLOCATING(zio));
4584 	ASSERT(ZIO_HAS_ALLOCATOR(zio));
4585 
4586 	/*
4587 	 * Try to place a reservation for this zio. If we're unable to
4588 	 * reserve then we throttle.
4589 	 */
4590 	if (!metaslab_class_throttle_reserve(zio->io_metaslab_class,
4591 	    zio->io_allocator, zio->io_prop.zp_copies, zio->io_size,
4592 	    B_FALSE, more)) {
4593 		return (NULL);
4594 	}
4595 	zio->io_flags |= ZIO_FLAG_ALLOC_THROTTLED;
4596 
4597 	avl_remove(&mca->mca_tree, zio);
4598 	ASSERT3U(zio->io_stage, <, ZIO_STAGE_DVA_ALLOCATE);
4599 
4600 	if (avl_is_empty(&mca->mca_tree))
4601 		*more = B_FALSE;
4602 	return (zio);
4603 }
4604 
4605 static zio_t *
4606 zio_dva_throttle(zio_t *zio)
4607 {
4608 	spa_t *spa = zio->io_spa;
4609 	zio_t *nio;
4610 	metaslab_class_t *mc;
4611 	boolean_t more;
4612 
4613 	/*
4614 	 * If not already chosen, choose an appropriate allocation class.
4615 	 */
4616 	mc = zio->io_metaslab_class;
4617 	if (mc == NULL)
4618 		mc = spa_preferred_class(spa, zio);
4619 
4620 	if (zio->io_priority == ZIO_PRIORITY_SYNC_WRITE ||
4621 	    !mc->mc_alloc_throttle_enabled ||
4622 	    zio->io_child_type == ZIO_CHILD_GANG ||
4623 	    zio->io_flags & ZIO_FLAG_NODATA) {
4624 		return (zio);
4625 	}
4626 
4627 	ASSERT(zio->io_type == ZIO_TYPE_WRITE);
4628 	ASSERT(ZIO_HAS_ALLOCATOR(zio));
4629 	ASSERT(zio->io_child_type > ZIO_CHILD_GANG);
4630 	ASSERT3U(zio->io_queued_timestamp, >, 0);
4631 	ASSERT(zio->io_stage == ZIO_STAGE_DVA_THROTTLE);
4632 
4633 	zio->io_metaslab_class = mc;
4634 	metaslab_class_allocator_t *mca = &mc->mc_allocator[zio->io_allocator];
4635 	mutex_enter(&mca->mca_lock);
4636 	avl_add(&mca->mca_tree, zio);
4637 	nio = zio_io_to_allocate(mca, &more);
4638 	mutex_exit(&mca->mca_lock);
4639 	return (nio);
4640 }
4641 
4642 static void
4643 zio_allocate_dispatch(metaslab_class_t *mc, int allocator)
4644 {
4645 	metaslab_class_allocator_t *mca = &mc->mc_allocator[allocator];
4646 	zio_t *zio;
4647 	boolean_t more;
4648 
4649 	do {
4650 		mutex_enter(&mca->mca_lock);
4651 		zio = zio_io_to_allocate(mca, &more);
4652 		mutex_exit(&mca->mca_lock);
4653 		if (zio == NULL)
4654 			return;
4655 
4656 		ASSERT3U(zio->io_stage, ==, ZIO_STAGE_DVA_THROTTLE);
4657 		ASSERT0(zio->io_error);
4658 		zio_taskq_dispatch(zio, ZIO_TASKQ_ISSUE, B_TRUE);
4659 	} while (more);
4660 }
4661 
4662 static zio_t *
4663 zio_dva_allocate(zio_t *zio)
4664 {
4665 	spa_t *spa = zio->io_spa;
4666 	metaslab_class_t *mc, *newmc;
4667 	blkptr_t *bp = zio->io_bp;
4668 	int error;
4669 	int flags = 0;
4670 
4671 	if (zio->io_gang_leader == NULL) {
4672 		ASSERT(zio->io_child_type > ZIO_CHILD_GANG);
4673 		zio->io_gang_leader = zio;
4674 	}
4675 	if (zio->io_flags & ZIO_FLAG_PREALLOCATED) {
4676 		ASSERT3U(zio->io_child_type, ==, ZIO_CHILD_GANG);
4677 		memcpy(zio->io_bp->blk_dva, zio->io_bp_orig.blk_dva,
4678 		    3 * sizeof (dva_t));
4679 		BP_SET_LOGICAL_BIRTH(zio->io_bp,
4680 		    BP_GET_LOGICAL_BIRTH(&zio->io_bp_orig));
4681 		BP_SET_PHYSICAL_BIRTH(zio->io_bp,
4682 		    BP_GET_RAW_PHYSICAL_BIRTH(&zio->io_bp_orig));
4683 		return (zio);
4684 	}
4685 
4686 	ASSERT(BP_IS_HOLE(bp));
4687 	ASSERT0(BP_GET_NDVAS(bp));
4688 	ASSERT3U(zio->io_prop.zp_copies, >, 0);
4689 
4690 	ASSERT3U(zio->io_prop.zp_copies, <=, spa_max_replication(spa));
4691 	ASSERT3U(zio->io_size, ==, BP_GET_PSIZE(bp));
4692 
4693 	if (zio->io_flags & ZIO_FLAG_GANG_CHILD)
4694 		flags |= METASLAB_GANG_CHILD;
4695 	if (zio->io_priority == ZIO_PRIORITY_ASYNC_WRITE)
4696 		flags |= METASLAB_ASYNC_ALLOC;
4697 
4698 	/*
4699 	 * If not already chosen, choose an appropriate allocation class.
4700 	 */
4701 	mc = zio->io_metaslab_class;
4702 	if (mc == NULL) {
4703 		mc = spa_preferred_class(spa, zio);
4704 		zio->io_metaslab_class = mc;
4705 	}
4706 	ZIOSTAT_BUMP(ziostat_total_allocations);
4707 
4708 again:
4709 	/*
4710 	 * Try allocating the block in the usual metaslab class.
4711 	 * If that's full, allocate it in some other class(es).
4712 	 * If that's full, allocate as a gang block,
4713 	 * and if all are full, the allocation fails (which shouldn't happen).
4714 	 *
4715 	 * Note that we do not fall back on embedded slog (ZIL) space, to
4716 	 * preserve unfragmented slog space, which is critical for decent
4717 	 * sync write performance.  If a log allocation fails, we will fall
4718 	 * back to spa_sync() which is abysmal for performance.
4719 	 */
4720 	ASSERT(ZIO_HAS_ALLOCATOR(zio));
4721 	error = metaslab_alloc(spa, mc, zio->io_size, bp,
4722 	    zio->io_prop.zp_copies, zio->io_txg, NULL, flags,
4723 	    ZIO_ALLOC_LIST(zio), zio->io_allocator, zio);
4724 
4725 	/*
4726 	 * When the dedup or special class is spilling into the normal class,
4727 	 * there can still be significant space available due to deferred
4728 	 * frees that are in-flight.  We track the txg when this occurred and
4729 	 * back off adding new DDT entries for a few txgs to allow the free
4730 	 * blocks to be processed.
4731 	 */
4732 	if (error == ENOSPC && spa->spa_dedup_class_full_txg != zio->io_txg &&
4733 	    (mc == spa_dedup_class(spa) || (mc == spa_special_class(spa) &&
4734 	    !spa_has_dedup(spa) && spa_special_has_ddt(spa)))) {
4735 		spa->spa_dedup_class_full_txg = zio->io_txg;
4736 		zfs_dbgmsg("%s[%llu]: %s class spilling, req size %llu, "
4737 		    "%llu allocated of %llu",
4738 		    spa_name(spa), (u_longlong_t)zio->io_txg,
4739 		    metaslab_class_get_name(mc),
4740 		    (u_longlong_t)zio->io_size,
4741 		    (u_longlong_t)metaslab_class_get_alloc(mc),
4742 		    (u_longlong_t)metaslab_class_get_space(mc));
4743 	}
4744 
4745 	/*
4746 	 * Fall back to some other class when this one is full.
4747 	 */
4748 	if (error == ENOSPC && (newmc = spa_preferred_class(spa, zio)) != mc) {
4749 		/*
4750 		 * If we are holding old class reservation, drop it.
4751 		 * Dispatch the next ZIO(s) there if some are waiting.
4752 		 */
4753 		if (zio->io_flags & ZIO_FLAG_ALLOC_THROTTLED) {
4754 			if (metaslab_class_throttle_unreserve(mc,
4755 			    zio->io_allocator, zio->io_prop.zp_copies,
4756 			    zio->io_size)) {
4757 				zio_allocate_dispatch(zio->io_metaslab_class,
4758 				    zio->io_allocator);
4759 			}
4760 			zio->io_flags &= ~ZIO_FLAG_ALLOC_THROTTLED;
4761 		}
4762 
4763 		if (zfs_flags & ZFS_DEBUG_METASLAB_ALLOC) {
4764 			zfs_dbgmsg("%s: metaslab allocation failure in %s "
4765 			    "class, trying fallback to %s class: zio %px, "
4766 			    "size %llu, error %d", spa_name(spa),
4767 			    metaslab_class_get_name(mc),
4768 			    metaslab_class_get_name(newmc),
4769 			    zio, (u_longlong_t)zio->io_size, error);
4770 		}
4771 		zio->io_metaslab_class = mc = newmc;
4772 		ZIOSTAT_BUMP(ziostat_alloc_class_fallbacks);
4773 
4774 		/*
4775 		 * If the new class uses throttling, return to that pipeline
4776 		 * stage.  Otherwise just do another allocation attempt.
4777 		 */
4778 		if (zio->io_priority != ZIO_PRIORITY_SYNC_WRITE &&
4779 		    mc->mc_alloc_throttle_enabled &&
4780 		    zio->io_child_type != ZIO_CHILD_GANG &&
4781 		    !(zio->io_flags & ZIO_FLAG_NODATA)) {
4782 			zio->io_stage = ZIO_STAGE_DVA_THROTTLE >> 1;
4783 			return (zio);
4784 		}
4785 		goto again;
4786 	}
4787 
4788 	if (error == ENOSPC && zio->io_size > spa->spa_min_alloc) {
4789 		if (zfs_flags & ZFS_DEBUG_METASLAB_ALLOC) {
4790 			zfs_dbgmsg("%s: metaslab allocation failure, "
4791 			    "trying ganging: zio %px, size %llu, error %d",
4792 			    spa_name(spa), zio, (u_longlong_t)zio->io_size,
4793 			    error);
4794 		}
4795 		ZIOSTAT_BUMP(ziostat_gang_writes);
4796 		if (flags & METASLAB_GANG_CHILD)
4797 			ZIOSTAT_BUMP(ziostat_gang_multilevel);
4798 		return (zio_write_gang_block(zio, mc));
4799 	}
4800 	if (error != 0) {
4801 		if (error != ENOSPC ||
4802 		    (zfs_flags & ZFS_DEBUG_METASLAB_ALLOC)) {
4803 			zfs_dbgmsg("%s: metaslab allocation failure: zio %px, "
4804 			    "size %llu, error %d",
4805 			    spa_name(spa), zio, (u_longlong_t)zio->io_size,
4806 			    error);
4807 		}
4808 		zio->io_error = error;
4809 	} else if (zio->io_prop.zp_rewrite) {
4810 		/*
4811 		 * For rewrite operations, preserve the logical birth time
4812 		 * but set the physical birth time to the current txg.
4813 		 */
4814 		uint64_t logical_birth = BP_GET_LOGICAL_BIRTH(&zio->io_bp_orig);
4815 		ASSERT3U(logical_birth, <=, zio->io_txg);
4816 		BP_SET_BIRTH(zio->io_bp, logical_birth, zio->io_txg);
4817 		BP_SET_REWRITE(zio->io_bp, 1);
4818 	}
4819 
4820 	return (zio);
4821 }
4822 
4823 static zio_t *
4824 zio_dva_free(zio_t *zio)
4825 {
4826 	metaslab_free(zio->io_spa, zio->io_bp, zio->io_txg, B_FALSE);
4827 
4828 	return (zio);
4829 }
4830 
4831 static zio_t *
4832 zio_dva_claim(zio_t *zio)
4833 {
4834 	int error;
4835 
4836 	error = metaslab_claim(zio->io_spa, zio->io_bp, zio->io_txg);
4837 	if (error)
4838 		zio->io_error = error;
4839 
4840 	return (zio);
4841 }
4842 
4843 /*
4844  * Undo an allocation.  This is used by zio_done() when an I/O fails
4845  * and we want to give back the block we just allocated.
4846  * This handles both normal blocks and gang blocks.
4847  */
4848 static void
4849 zio_dva_unallocate(zio_t *zio, zio_gang_node_t *gn, blkptr_t *bp)
4850 {
4851 	ASSERT(BP_GET_BIRTH(bp) == zio->io_txg || BP_IS_HOLE(bp));
4852 	ASSERT0P(zio->io_bp_override);
4853 
4854 	if (!BP_IS_HOLE(bp)) {
4855 		metaslab_free(zio->io_spa, bp, BP_GET_BIRTH(bp), B_TRUE);
4856 	}
4857 
4858 	if (gn != NULL) {
4859 		for (int g = 0; g < gbh_nblkptrs(gn->gn_gangblocksize); g++) {
4860 			zio_dva_unallocate(zio, gn->gn_child[g],
4861 			    gbh_bp(gn->gn_gbh, g));
4862 		}
4863 	}
4864 }
4865 
4866 /*
4867  * Try to allocate an intent log block.  Return 0 on success, errno on failure.
4868  */
4869 int
4870 zio_alloc_zil(spa_t *spa, objset_t *os, uint64_t txg, blkptr_t *new_bp,
4871     uint64_t min_size, uint64_t max_size, boolean_t *slog,
4872     boolean_t allow_larger)
4873 {
4874 	int error;
4875 	zio_alloc_list_t io_alloc_list;
4876 	uint64_t alloc_size = 0;
4877 
4878 	ASSERT(txg > spa_syncing_txg(spa));
4879 	ASSERT3U(min_size, <=, max_size);
4880 
4881 	metaslab_trace_init(&io_alloc_list);
4882 
4883 	/*
4884 	 * Block pointer fields are useful to metaslabs for stats and debugging.
4885 	 * Fill in the obvious ones before calling into metaslab_alloc().
4886 	 */
4887 	BP_SET_TYPE(new_bp, DMU_OT_INTENT_LOG);
4888 	BP_SET_PSIZE(new_bp, max_size);
4889 	BP_SET_LEVEL(new_bp, 0);
4890 
4891 	/*
4892 	 * When allocating a zil block, we don't have information about
4893 	 * the final destination of the block except the objset it's part
4894 	 * of, so we just hash the objset ID to pick the allocator to get
4895 	 * some parallelism.
4896 	 */
4897 	int flags = METASLAB_ZIL;
4898 	int allocator = (uint_t)cityhash1(os->os_dsl_dataset->ds_object)
4899 	    % spa->spa_alloc_count;
4900 	ZIOSTAT_BUMP(ziostat_total_allocations);
4901 
4902 	/* Try log class (dedicated slog devices) first */
4903 	error = metaslab_alloc_range(spa, spa_log_class(spa), min_size,
4904 	    max_size, new_bp, 1, txg, NULL, flags, &io_alloc_list, allocator,
4905 	    NULL, &alloc_size);
4906 	*slog = (error == 0);
4907 
4908 	/* Try special_embedded_log class (reserved on special vdevs) */
4909 	if (error != 0) {
4910 		error = metaslab_alloc_range(spa,
4911 		    spa_special_embedded_log_class(spa), min_size, max_size,
4912 		    new_bp, 1, txg, NULL, flags, &io_alloc_list, allocator,
4913 		    NULL, &alloc_size);
4914 	}
4915 
4916 	/* Try special class (general special vdev allocation) */
4917 	if (error != 0) {
4918 		error = metaslab_alloc_range(spa, spa_special_class(spa),
4919 		    min_size, max_size, new_bp, 1, txg, NULL, flags,
4920 		    &io_alloc_list, allocator, NULL, &alloc_size);
4921 	}
4922 
4923 	/* Try embedded_log class (reserved on normal vdevs) */
4924 	if (error != 0) {
4925 		error = metaslab_alloc_range(spa, spa_embedded_log_class(spa),
4926 		    min_size, max_size, new_bp, 1, txg, NULL, flags,
4927 		    &io_alloc_list, allocator, NULL, &alloc_size);
4928 	}
4929 
4930 	/* Finally fall back to normal class */
4931 	if (error != 0) {
4932 		ZIOSTAT_BUMP(ziostat_alloc_class_fallbacks);
4933 		error = metaslab_alloc_range(spa, spa_normal_class(spa),
4934 		    min_size, max_size, new_bp, 1, txg, NULL, flags,
4935 		    &io_alloc_list, allocator, NULL, &alloc_size);
4936 	}
4937 	metaslab_trace_fini(&io_alloc_list);
4938 
4939 	if (error == 0) {
4940 		if (!allow_larger)
4941 			alloc_size = MIN(alloc_size, max_size);
4942 		else if (max_size <= SPA_OLD_MAXBLOCKSIZE)
4943 			alloc_size = MIN(alloc_size, SPA_OLD_MAXBLOCKSIZE);
4944 		alloc_size = P2ALIGN_TYPED(alloc_size, ZIL_MIN_BLKSZ, uint64_t);
4945 
4946 		BP_SET_LSIZE(new_bp, alloc_size);
4947 		BP_SET_PSIZE(new_bp, alloc_size);
4948 		BP_SET_COMPRESS(new_bp, ZIO_COMPRESS_OFF);
4949 		BP_SET_CHECKSUM(new_bp,
4950 		    spa_version(spa) >= SPA_VERSION_SLIM_ZIL
4951 		    ? ZIO_CHECKSUM_ZILOG2 : ZIO_CHECKSUM_ZILOG);
4952 		BP_SET_TYPE(new_bp, DMU_OT_INTENT_LOG);
4953 		BP_SET_LEVEL(new_bp, 0);
4954 		BP_SET_DEDUP(new_bp, 0);
4955 		BP_SET_BYTEORDER(new_bp, ZFS_HOST_BYTEORDER);
4956 
4957 		/*
4958 		 * encrypted blocks will require an IV and salt. We generate
4959 		 * these now since we will not be rewriting the bp at
4960 		 * rewrite time.
4961 		 */
4962 		if (os->os_encrypted) {
4963 			uint8_t iv[ZIO_DATA_IV_LEN];
4964 			uint8_t salt[ZIO_DATA_SALT_LEN];
4965 
4966 			BP_SET_CRYPT(new_bp, B_TRUE);
4967 			VERIFY0(spa_crypt_get_salt(spa,
4968 			    dmu_objset_id(os), salt));
4969 			VERIFY0(zio_crypt_generate_iv(iv));
4970 
4971 			zio_crypt_encode_params_bp(new_bp, salt, iv);
4972 		}
4973 	} else {
4974 		zfs_dbgmsg("%s: zil block allocation failure: "
4975 		    "min_size %llu, max_size %llu, error %d", spa_name(spa),
4976 		    (u_longlong_t)min_size, (u_longlong_t)max_size, error);
4977 	}
4978 
4979 	return (error);
4980 }
4981 
4982 /*
4983  * ==========================================================================
4984  * Read and write to physical devices
4985  * ==========================================================================
4986  */
4987 
4988 /*
4989  * Issue an I/O to the underlying vdev. Typically the issue pipeline
4990  * stops after this stage and will resume upon I/O completion.
4991  * However, there are instances where the vdev layer may need to
4992  * continue the pipeline when an I/O was not issued. Since the I/O
4993  * that was sent to the vdev layer might be different than the one
4994  * currently active in the pipeline (see vdev_queue_io()), we explicitly
4995  * force the underlying vdev layers to call either zio_execute() or
4996  * zio_interrupt() to ensure that the pipeline continues with the correct I/O.
4997  */
4998 static zio_t *
4999 zio_vdev_io_start(zio_t *zio)
5000 {
5001 	vdev_t *vd = zio->io_vd;
5002 	uint64_t align;
5003 	spa_t *spa = zio->io_spa;
5004 
5005 	zio->io_delta = 0;
5006 	zio->io_delay = 0;
5007 
5008 	ASSERT0(zio->io_error);
5009 	ASSERT0(zio->io_child_error[ZIO_CHILD_VDEV]);
5010 
5011 	if (vd == NULL) {
5012 		if (!(zio->io_flags & ZIO_FLAG_CONFIG_WRITER)) {
5013 			/*
5014 			 * A deadlock workaround. The ddt_prune_unique_entries()
5015 			 * -> prune_candidates_sync() code path takes the
5016 			 * SCL_ZIO reader lock and may request it again here.
5017 			 * If there is another thread who wants the SCL_ZIO
5018 			 * writer lock, then scl_write_wanted will be set.
5019 			 * Thus, the spa_config_enter_priority() is used to
5020 			 * ignore pending writer requests.
5021 			 *
5022 			 * The locking should be revised to remove the need
5023 			 * for this workaround.  If that's not workable then
5024 			 * it should only be applied to the zios involved in
5025 			 * the pruning process.  This impacts the read/write
5026 			 * I/O balance while pruning.
5027 			 */
5028 			if (spa->spa_active_ddt_prune)
5029 				spa_config_enter_priority(spa, SCL_ZIO, zio,
5030 				    RW_READER);
5031 			else
5032 				spa_config_enter(spa, SCL_ZIO, zio,
5033 				    RW_READER);
5034 		}
5035 
5036 		/*
5037 		 * The mirror_ops handle multiple DVAs in a single BP.
5038 		 */
5039 		vdev_mirror_ops.vdev_op_io_start(zio);
5040 		return (NULL);
5041 	}
5042 
5043 	ASSERT3P(zio->io_logical, !=, zio);
5044 	if (zio->io_type == ZIO_TYPE_WRITE) {
5045 		ASSERT(spa->spa_trust_config);
5046 
5047 		/*
5048 		 * Note: the code can handle other kinds of writes,
5049 		 * but we don't expect them.
5050 		 */
5051 		if (zio->io_vd->vdev_noalloc) {
5052 			ASSERT(zio->io_flags &
5053 			    (ZIO_FLAG_PHYSICAL | ZIO_FLAG_SELF_HEAL |
5054 			    ZIO_FLAG_RESILVER | ZIO_FLAG_INDUCE_DAMAGE));
5055 		}
5056 	}
5057 
5058 	align = 1ULL << vd->vdev_top->vdev_ashift;
5059 
5060 	if (!(zio->io_flags & ZIO_FLAG_PHYSICAL) &&
5061 	    P2PHASE(zio->io_size, align) != 0) {
5062 		/* Transform logical writes to be a full physical block size. */
5063 		uint64_t asize = P2ROUNDUP(zio->io_size, align);
5064 		abd_t *abuf = abd_alloc_sametype(zio->io_abd, asize);
5065 		ASSERT(vd == vd->vdev_top);
5066 		if (zio->io_type == ZIO_TYPE_WRITE) {
5067 			abd_copy(abuf, zio->io_abd, zio->io_size);
5068 			abd_zero_off(abuf, zio->io_size, asize - zio->io_size);
5069 		}
5070 		zio_push_transform(zio, abuf, asize, asize, zio_subblock);
5071 	}
5072 
5073 	/*
5074 	 * If this is not a physical io, make sure that it is properly aligned
5075 	 * before proceeding.
5076 	 */
5077 	if (!(zio->io_flags & ZIO_FLAG_PHYSICAL)) {
5078 		ASSERT0(P2PHASE(zio->io_offset, align));
5079 		ASSERT0(P2PHASE(zio->io_size, align));
5080 	} else {
5081 		/*
5082 		 * For physical writes, we allow 512b aligned writes and assume
5083 		 * the device will perform a read-modify-write as necessary.
5084 		 */
5085 		ASSERT0(P2PHASE(zio->io_offset, SPA_MINBLOCKSIZE));
5086 		ASSERT0(P2PHASE(zio->io_size, SPA_MINBLOCKSIZE));
5087 	}
5088 
5089 	VERIFY(zio->io_type != ZIO_TYPE_WRITE || spa_writeable(spa));
5090 
5091 	/*
5092 	 * If this is a repair I/O, and there's no self-healing involved --
5093 	 * that is, we're just resilvering what we expect to resilver --
5094 	 * then don't do the I/O unless zio's txg is actually in vd's DTL.
5095 	 * This prevents spurious resilvering.
5096 	 *
5097 	 * There are a few ways that we can end up creating these spurious
5098 	 * resilver i/os:
5099 	 *
5100 	 * 1. A resilver i/o will be issued if any DVA in the BP has a
5101 	 * dirty DTL.  The mirror code will issue resilver writes to
5102 	 * each DVA, including the one(s) that are not on vdevs with dirty
5103 	 * DTLs.
5104 	 *
5105 	 * 2. With nested replication, which happens when we have a
5106 	 * "replacing" or "spare" vdev that's a child of a mirror or raidz.
5107 	 * For example, given mirror(replacing(A+B), C), it's likely that
5108 	 * only A is out of date (it's the new device). In this case, we'll
5109 	 * read from C, then use the data to resilver A+B -- but we don't
5110 	 * actually want to resilver B, just A. The top-level mirror has no
5111 	 * way to know this, so instead we just discard unnecessary repairs
5112 	 * as we work our way down the vdev tree.
5113 	 *
5114 	 * 3. ZTEST also creates mirrors of mirrors, mirrors of raidz, etc.
5115 	 * The same logic applies to any form of nested replication: ditto
5116 	 * + mirror, RAID-Z + replacing, etc.
5117 	 *
5118 	 * However, indirect vdevs point off to other vdevs which may have
5119 	 * DTL's, so we never bypass them.  The child i/os on concrete vdevs
5120 	 * will be properly bypassed instead.
5121 	 *
5122 	 * Leaf DTL_PARTIAL can be empty when a legitimate write comes from
5123 	 * a dRAID spare vdev. For example, when a dRAID spare is first
5124 	 * used, its spare blocks need to be written to but the leaf vdev's
5125 	 * of such blocks can have empty DTL_PARTIAL.
5126 	 *
5127 	 * There seemed no clean way to allow such writes while bypassing
5128 	 * spurious ones. At this point, just avoid all bypassing for dRAID
5129 	 * for correctness.
5130 	 */
5131 	if ((zio->io_flags & ZIO_FLAG_IO_REPAIR) &&
5132 	    !(zio->io_flags & ZIO_FLAG_SELF_HEAL) &&
5133 	    zio->io_txg != 0 &&	/* not a delegated i/o */
5134 	    vd->vdev_ops != &vdev_indirect_ops &&
5135 	    vd->vdev_top->vdev_ops != &vdev_draid_ops &&
5136 	    !vdev_dtl_contains(vd, DTL_PARTIAL, zio->io_txg, 1)) {
5137 		ASSERT(zio->io_type == ZIO_TYPE_WRITE);
5138 		zio_vdev_io_bypass(zio);
5139 		return (zio);
5140 	}
5141 
5142 	/*
5143 	 * Select the next best leaf I/O to process.  Distributed spares are
5144 	 * excluded since they dispatch the I/O directly to a leaf vdev after
5145 	 * applying the dRAID mapping.
5146 	 */
5147 	if (vd->vdev_ops->vdev_op_leaf &&
5148 	    vd->vdev_ops != &vdev_draid_spare_ops &&
5149 	    (zio->io_type == ZIO_TYPE_READ ||
5150 	    zio->io_type == ZIO_TYPE_WRITE ||
5151 	    zio->io_type == ZIO_TYPE_TRIM)) {
5152 
5153 		if ((zio = vdev_queue_io(zio)) == NULL)
5154 			return (NULL);
5155 
5156 		if (!vdev_accessible(vd, zio)) {
5157 			zio->io_error = SET_ERROR(ENXIO);
5158 			zio_interrupt(zio);
5159 			return (NULL);
5160 		}
5161 		zio->io_delay = gethrtime();
5162 
5163 		int error = zio_handle_device_injections(vd, zio, ENOSYS,
5164 		    EFAULT);
5165 		if (error == ENOSYS || (error == EFAULT &&
5166 		    !(zio->io_flags & ZIO_FLAG_IO_REPAIR))) {
5167 			/*
5168 			 * "no-op" injections return success, but do no actual
5169 			 * work. Just return it. "io-prefail" injections are
5170 			 * similar, but don't return success.
5171 			 */
5172 			if (error == EFAULT)
5173 				zio->io_error = EIO;
5174 			zio_delay_interrupt(zio);
5175 			return (NULL);
5176 		}
5177 	}
5178 
5179 	vd->vdev_ops->vdev_op_io_start(zio);
5180 	return (NULL);
5181 }
5182 
5183 static zio_t *
5184 zio_vdev_io_done(zio_t *zio)
5185 {
5186 	vdev_t *vd = zio->io_vd;
5187 	vdev_ops_t *ops = vd ? vd->vdev_ops : &vdev_mirror_ops;
5188 	boolean_t unexpected_error = B_FALSE;
5189 
5190 	if (zio_wait_for_children(zio, ZIO_CHILD_VDEV_BIT, ZIO_WAIT_DONE)) {
5191 		return (NULL);
5192 	}
5193 
5194 	ASSERT(zio->io_type == ZIO_TYPE_READ ||
5195 	    zio->io_type == ZIO_TYPE_WRITE ||
5196 	    zio->io_type == ZIO_TYPE_FLUSH ||
5197 	    zio->io_type == ZIO_TYPE_TRIM);
5198 
5199 	if (zio->io_delay) {
5200 		/* io_delta is set only if the completion was deferred. */
5201 		zio->io_delay = (zio->io_delta != 0 ?
5202 		    zio->io_timestamp + zio->io_delta : gethrtime()) -
5203 		    zio->io_delay;
5204 	}
5205 
5206 	if (vd != NULL && vd->vdev_ops->vdev_op_leaf &&
5207 	    vd->vdev_ops != &vdev_draid_spare_ops) {
5208 		if (zio->io_type != ZIO_TYPE_FLUSH)
5209 			vdev_queue_io_done(zio);
5210 
5211 		if (zio_injection_enabled && zio->io_error == 0)
5212 			zio->io_error = zio_handle_device_injections(vd, zio,
5213 			    EIO, EILSEQ);
5214 
5215 		if (zio_injection_enabled && zio->io_error == 0)
5216 			zio->io_error = zio_handle_label_injection(zio, EIO);
5217 
5218 		if (zio->io_error && zio->io_type != ZIO_TYPE_FLUSH &&
5219 		    zio->io_type != ZIO_TYPE_TRIM) {
5220 			if (!vdev_accessible(vd, zio)) {
5221 				zio->io_error = SET_ERROR(ENXIO);
5222 			} else {
5223 				unexpected_error = B_TRUE;
5224 			}
5225 		}
5226 	}
5227 
5228 	/*
5229 	 * This zio got here on a pipeline thread rather than from the block
5230 	 * layer, so it runs its own completion and gives up its membership.
5231 	 * The batch is chained only below, to keep it clear of whatever
5232 	 * vdev_op_io_done() may do with this zio.
5233 	 */
5234 	zio_t *batch = zio_batch_leave(zio);
5235 
5236 	ops->vdev_op_io_done(zio);
5237 
5238 	if (unexpected_error && vd->vdev_remove_wanted == B_FALSE)
5239 		VERIFY0P(vdev_probe(vd, zio));
5240 
5241 	zio->io_exec_next = batch;
5242 	return (zio);
5243 }
5244 
5245 /*
5246  * This function is used to change the priority of an existing zio that is
5247  * currently in-flight. This is used by the arc to upgrade priority in the
5248  * event that a demand read is made for a block that is currently queued
5249  * as a scrub or async read IO. Otherwise, the high priority read request
5250  * would end up having to wait for the lower priority IO.
5251  */
5252 void
5253 zio_change_priority(zio_t *pio, zio_priority_t priority)
5254 {
5255 	zio_t *cio, *cio_next;
5256 	zio_link_t *zl = NULL;
5257 
5258 	ASSERT3U(priority, <, ZIO_PRIORITY_NUM_QUEUEABLE);
5259 
5260 	if (pio->io_vd != NULL && pio->io_vd->vdev_ops->vdev_op_leaf) {
5261 		vdev_queue_change_io_priority(pio, priority);
5262 	} else {
5263 		pio->io_priority = priority;
5264 	}
5265 
5266 	mutex_enter(&pio->io_lock);
5267 	for (cio = zio_walk_children(pio, &zl); cio != NULL; cio = cio_next) {
5268 		cio_next = zio_walk_children(pio, &zl);
5269 		zio_change_priority(cio, priority);
5270 	}
5271 	mutex_exit(&pio->io_lock);
5272 }
5273 
5274 /*
5275  * For non-raidz ZIOs, we can just copy aside the bad data read from the
5276  * disk, and use that to finish the checksum ereport later.
5277  */
5278 static void
5279 zio_vsd_default_cksum_finish(zio_cksum_report_t *zcr,
5280     const abd_t *good_buf)
5281 {
5282 	/* no processing needed */
5283 	zfs_ereport_finish_checksum(zcr, good_buf, zcr->zcr_cbdata, B_FALSE);
5284 }
5285 
5286 void
5287 zio_vsd_default_cksum_report(zio_t *zio, zio_cksum_report_t *zcr)
5288 {
5289 	void *abd = abd_alloc_sametype(zio->io_abd, zio->io_size);
5290 
5291 	abd_copy(abd, zio->io_abd, zio->io_size);
5292 
5293 	zcr->zcr_cbinfo = zio->io_size;
5294 	zcr->zcr_cbdata = abd;
5295 	zcr->zcr_finish = zio_vsd_default_cksum_finish;
5296 	zcr->zcr_free = zio_abd_free;
5297 }
5298 
5299 static zio_t *
5300 zio_vdev_io_assess(zio_t *zio)
5301 {
5302 	vdev_t *vd = zio->io_vd;
5303 
5304 	if (zio_wait_for_children(zio, ZIO_CHILD_VDEV_BIT, ZIO_WAIT_DONE)) {
5305 		return (NULL);
5306 	}
5307 
5308 	/* A repair write bypass skips VDEV_IO_DONE entirely. */
5309 	zio->io_exec_next = zio_batch_leave(zio);
5310 
5311 	if (vd == NULL && !(zio->io_flags & ZIO_FLAG_CONFIG_WRITER))
5312 		spa_config_exit(zio->io_spa, SCL_ZIO, zio);
5313 
5314 	if (zio->io_vsd != NULL) {
5315 		zio->io_vsd_ops->vsd_free(zio);
5316 		zio->io_vsd = NULL;
5317 	}
5318 
5319 	/*
5320 	 * If a Direct I/O operation has a checksum verify error then this I/O
5321 	 * should not attempt to be issued again.
5322 	 */
5323 	if (zio->io_post & ZIO_POST_DIO_CHKSUM_ERR) {
5324 		if (zio->io_type == ZIO_TYPE_WRITE) {
5325 			ASSERT3U(zio->io_child_type, ==, ZIO_CHILD_LOGICAL);
5326 			ASSERT3U(zio->io_error, ==, EIO);
5327 		}
5328 		zio->io_pipeline = ZIO_INTERLOCK_PIPELINE;
5329 		return (zio);
5330 	}
5331 
5332 	if (zio_injection_enabled && zio->io_error == 0)
5333 		zio->io_error = zio_handle_fault_injection(zio, EIO);
5334 
5335 	/*
5336 	 * If the I/O failed, determine whether we should attempt to retry it.
5337 	 *
5338 	 * On retry, we cut in line in the issue queue, since we don't want
5339 	 * compression/checksumming/etc. work to prevent our (cheap) IO reissue.
5340 	 */
5341 	if (zio->io_error && vd == NULL &&
5342 	    !(zio->io_flags & (ZIO_FLAG_DONT_RETRY | ZIO_FLAG_IO_RETRY))) {
5343 		ASSERT(!(zio->io_flags & ZIO_FLAG_DONT_QUEUE));	/* not a leaf */
5344 		ASSERT(!(zio->io_flags & ZIO_FLAG_IO_BYPASS));	/* not a leaf */
5345 		zio->io_error = 0;
5346 		zio->io_flags |= ZIO_FLAG_IO_RETRY | ZIO_FLAG_DONT_AGGREGATE;
5347 		zio->io_stage = ZIO_STAGE_VDEV_IO_START >> 1;
5348 		zio_taskq_dispatch(zio, ZIO_TASKQ_ISSUE,
5349 		    zio_requeue_io_start_cut_in_line);
5350 		return (NULL);
5351 	}
5352 
5353 	/*
5354 	 * If we got an error on a leaf device, convert it to ENXIO
5355 	 * if the device is not accessible at all.
5356 	 */
5357 	if (zio->io_error && vd != NULL && vd->vdev_ops->vdev_op_leaf &&
5358 	    !vdev_accessible(vd, zio))
5359 		zio->io_error = SET_ERROR(ENXIO);
5360 
5361 	/*
5362 	 * If we can't write to an interior vdev (mirror or RAID-Z),
5363 	 * set vdev_cant_write so that we stop trying to allocate from it.
5364 	 */
5365 	if (zio->io_error == ENXIO && zio->io_type == ZIO_TYPE_WRITE &&
5366 	    vd != NULL && !vd->vdev_ops->vdev_op_leaf) {
5367 		vdev_dbgmsg(vd, "zio_vdev_io_assess(zio=%px) setting "
5368 		    "cant_write=TRUE due to write failure with ENXIO",
5369 		    zio);
5370 		vd->vdev_cant_write = B_TRUE;
5371 	}
5372 
5373 	/*
5374 	 * If a cache flush returns ENOTSUP we know that no future
5375 	 * attempts will ever succeed. In this case we set a persistent
5376 	 * boolean flag so that we don't bother with it in the future, and
5377 	 * then we act like the flush succeeded.
5378 	 */
5379 	if (zio->io_error == ENOTSUP && zio->io_type == ZIO_TYPE_FLUSH &&
5380 	    vd != NULL) {
5381 		vd->vdev_nowritecache = B_TRUE;
5382 		zio->io_error = 0;
5383 	}
5384 
5385 	if (zio->io_error)
5386 		zio->io_pipeline = ZIO_INTERLOCK_PIPELINE;
5387 
5388 	return (zio);
5389 }
5390 
5391 void
5392 zio_vdev_io_reissue(zio_t *zio)
5393 {
5394 	ASSERT(zio->io_stage == ZIO_STAGE_VDEV_IO_START);
5395 	ASSERT0(zio->io_error);
5396 
5397 	zio->io_stage >>= 1;
5398 }
5399 
5400 void
5401 zio_vdev_io_redone(zio_t *zio)
5402 {
5403 	ASSERT(zio->io_stage == ZIO_STAGE_VDEV_IO_DONE);
5404 
5405 	zio->io_stage >>= 1;
5406 }
5407 
5408 void
5409 zio_vdev_io_bypass(zio_t *zio)
5410 {
5411 	ASSERT(zio->io_stage == ZIO_STAGE_VDEV_IO_START);
5412 	ASSERT0(zio->io_error);
5413 
5414 	zio->io_flags |= ZIO_FLAG_IO_BYPASS;
5415 	zio->io_stage = ZIO_STAGE_VDEV_IO_ASSESS >> 1;
5416 }
5417 
5418 /*
5419  * ==========================================================================
5420  * Encrypt and store encryption parameters
5421  * ==========================================================================
5422  */
5423 
5424 
5425 /*
5426  * This function is used for ZIO_STAGE_ENCRYPT. It is responsible for
5427  * managing the storage of encryption parameters and passing them to the
5428  * lower-level encryption functions.
5429  */
5430 static zio_t *
5431 zio_encrypt(zio_t *zio)
5432 {
5433 	zio_prop_t *zp = &zio->io_prop;
5434 	spa_t *spa = zio->io_spa;
5435 	blkptr_t *bp = zio->io_bp;
5436 	uint64_t psize = BP_GET_PSIZE(bp);
5437 	uint64_t dsobj = zio->io_bookmark.zb_objset;
5438 	dmu_object_type_t ot = BP_GET_TYPE(bp);
5439 	void *enc_buf = NULL;
5440 	abd_t *eabd = NULL;
5441 	uint8_t salt[ZIO_DATA_SALT_LEN];
5442 	uint8_t iv[ZIO_DATA_IV_LEN];
5443 	uint8_t mac[ZIO_DATA_MAC_LEN];
5444 	boolean_t no_crypt = B_FALSE;
5445 
5446 	/* the root zio already encrypted the data */
5447 	if (zio->io_child_type == ZIO_CHILD_GANG)
5448 		return (zio);
5449 
5450 	/* only ZIL blocks are re-encrypted on rewrite */
5451 	if (!IO_IS_ALLOCATING(zio) && ot != DMU_OT_INTENT_LOG)
5452 		return (zio);
5453 
5454 	if (!(zp->zp_encrypt || BP_IS_ENCRYPTED(bp))) {
5455 		BP_SET_CRYPT(bp, B_FALSE);
5456 		return (zio);
5457 	}
5458 
5459 	/* if we are doing raw encryption set the provided encryption params */
5460 	if (zio->io_flags & ZIO_FLAG_RAW_ENCRYPT) {
5461 		ASSERT0(BP_GET_LEVEL(bp));
5462 		BP_SET_CRYPT(bp, B_TRUE);
5463 		BP_SET_BYTEORDER(bp, zp->zp_byteorder);
5464 		if (ot != DMU_OT_OBJSET)
5465 			zio_crypt_encode_mac_bp(bp, zp->zp_mac);
5466 
5467 		/* dnode blocks must be written out in the provided byteorder */
5468 		if (zp->zp_byteorder != ZFS_HOST_BYTEORDER &&
5469 		    ot == DMU_OT_DNODE) {
5470 			void *bswap_buf = zio_buf_alloc(psize);
5471 			abd_t *babd = abd_get_from_buf(bswap_buf, psize);
5472 
5473 			ASSERT3U(BP_GET_COMPRESS(bp), ==, ZIO_COMPRESS_OFF);
5474 			abd_copy_to_buf(bswap_buf, zio->io_abd, psize);
5475 			dmu_ot_byteswap[DMU_OT_BYTESWAP(ot)].ob_func(bswap_buf,
5476 			    psize);
5477 
5478 			abd_take_ownership_of_buf(babd, B_TRUE);
5479 			zio_push_transform(zio, babd, psize, psize, NULL);
5480 		}
5481 
5482 		if (DMU_OT_IS_ENCRYPTED(ot))
5483 			zio_crypt_encode_params_bp(bp, zp->zp_salt, zp->zp_iv);
5484 		return (zio);
5485 	}
5486 
5487 	/* indirect blocks only maintain a cksum of the lower level MACs */
5488 	if (BP_GET_LEVEL(bp) > 0) {
5489 		BP_SET_CRYPT(bp, B_TRUE);
5490 		VERIFY0(zio_crypt_do_indirect_mac_checksum_abd(B_TRUE,
5491 		    zio->io_orig_abd, BP_GET_LSIZE(bp), BP_SHOULD_BYTESWAP(bp),
5492 		    mac));
5493 		zio_crypt_encode_mac_bp(bp, mac);
5494 		return (zio);
5495 	}
5496 
5497 	/*
5498 	 * Objset blocks are a special case since they have 2 256-bit MACs
5499 	 * embedded within them.
5500 	 */
5501 	if (ot == DMU_OT_OBJSET) {
5502 		ASSERT0(DMU_OT_IS_ENCRYPTED(ot));
5503 		ASSERT3U(BP_GET_COMPRESS(bp), ==, ZIO_COMPRESS_OFF);
5504 		BP_SET_CRYPT(bp, B_TRUE);
5505 		VERIFY0(spa_do_crypt_objset_mac_abd(B_TRUE, spa, dsobj,
5506 		    zio->io_abd, psize, BP_SHOULD_BYTESWAP(bp)));
5507 		return (zio);
5508 	}
5509 
5510 	/* unencrypted object types are only authenticated with a MAC */
5511 	if (!DMU_OT_IS_ENCRYPTED(ot)) {
5512 		BP_SET_CRYPT(bp, B_TRUE);
5513 		VERIFY0(spa_do_crypt_mac_abd(B_TRUE, spa, dsobj,
5514 		    zio->io_abd, psize, mac));
5515 		zio_crypt_encode_mac_bp(bp, mac);
5516 		return (zio);
5517 	}
5518 
5519 	/*
5520 	 * Later passes of sync-to-convergence may decide to rewrite data
5521 	 * in place to avoid more disk reallocations. This presents a problem
5522 	 * for encryption because this constitutes rewriting the new data with
5523 	 * the same encryption key and IV. However, this only applies to blocks
5524 	 * in the MOS (particularly the spacemaps) and we do not encrypt the
5525 	 * MOS. We assert that the zio is allocating or an intent log write
5526 	 * to enforce this.
5527 	 */
5528 	ASSERT(IO_IS_ALLOCATING(zio) || ot == DMU_OT_INTENT_LOG);
5529 	ASSERT(BP_GET_LEVEL(bp) == 0 || ot == DMU_OT_INTENT_LOG);
5530 	ASSERT(spa_feature_is_active(spa, SPA_FEATURE_ENCRYPTION));
5531 	ASSERT3U(psize, !=, 0);
5532 
5533 	enc_buf = zio_buf_alloc(psize);
5534 	eabd = abd_get_from_buf(enc_buf, psize);
5535 	abd_take_ownership_of_buf(eabd, B_TRUE);
5536 
5537 	/*
5538 	 * For an explanation of what encryption parameters are stored
5539 	 * where, see the block comment in zio_crypt.c.
5540 	 */
5541 	if (ot == DMU_OT_INTENT_LOG) {
5542 		zio_crypt_decode_params_bp(bp, salt, iv);
5543 	} else {
5544 		BP_SET_CRYPT(bp, B_TRUE);
5545 	}
5546 
5547 	/* Perform the encryption. This should not fail */
5548 	VERIFY0(spa_do_crypt_abd(B_TRUE, spa, &zio->io_bookmark,
5549 	    BP_GET_TYPE(bp), BP_GET_DEDUP(bp), BP_SHOULD_BYTESWAP(bp),
5550 	    salt, iv, mac, psize, zio->io_abd, eabd, &no_crypt));
5551 
5552 	/* encode encryption metadata into the bp */
5553 	if (ot == DMU_OT_INTENT_LOG) {
5554 		/*
5555 		 * ZIL blocks store the MAC in the embedded checksum, so the
5556 		 * transform must always be applied.
5557 		 */
5558 		zio_crypt_encode_mac_zil(enc_buf, mac);
5559 		zio_push_transform(zio, eabd, psize, psize, NULL);
5560 	} else {
5561 		BP_SET_CRYPT(bp, B_TRUE);
5562 		zio_crypt_encode_params_bp(bp, salt, iv);
5563 		zio_crypt_encode_mac_bp(bp, mac);
5564 
5565 		if (no_crypt) {
5566 			ASSERT3U(ot, ==, DMU_OT_DNODE);
5567 			abd_free(eabd);
5568 		} else {
5569 			zio_push_transform(zio, eabd, psize, psize, NULL);
5570 		}
5571 	}
5572 
5573 	return (zio);
5574 }
5575 
5576 /*
5577  * ==========================================================================
5578  * Generate and verify checksums
5579  * ==========================================================================
5580  */
5581 static zio_t *
5582 zio_checksum_generate(zio_t *zio)
5583 {
5584 	blkptr_t *bp = zio->io_bp;
5585 	enum zio_checksum checksum;
5586 
5587 	if (bp == NULL) {
5588 		/*
5589 		 * This is zio_write_phys().
5590 		 * We're either generating a label checksum, or none at all.
5591 		 */
5592 		checksum = zio->io_prop.zp_checksum;
5593 
5594 		if (checksum == ZIO_CHECKSUM_OFF)
5595 			return (zio);
5596 
5597 		ASSERT(checksum == ZIO_CHECKSUM_LABEL);
5598 	} else {
5599 		if (BP_IS_GANG(bp) && zio->io_child_type == ZIO_CHILD_GANG) {
5600 			ASSERT(!IO_IS_ALLOCATING(zio));
5601 			checksum = ZIO_CHECKSUM_GANG_HEADER;
5602 		} else {
5603 			checksum = BP_GET_CHECKSUM(bp);
5604 		}
5605 	}
5606 
5607 	zio_checksum_compute(zio, checksum, zio->io_abd, zio->io_size);
5608 
5609 	return (zio);
5610 }
5611 
5612 static zio_t *
5613 zio_checksum_verify(zio_t *zio)
5614 {
5615 	zio_bad_cksum_t info;
5616 	blkptr_t *bp = zio->io_bp;
5617 	int error;
5618 
5619 	ASSERT(zio->io_vd != NULL);
5620 
5621 	if (bp == NULL) {
5622 		/*
5623 		 * This is zio_read_phys().
5624 		 * We're either verifying a label checksum, or nothing at all.
5625 		 */
5626 		if (zio->io_prop.zp_checksum == ZIO_CHECKSUM_OFF)
5627 			return (zio);
5628 
5629 		ASSERT3U(zio->io_prop.zp_checksum, ==, ZIO_CHECKSUM_LABEL);
5630 	}
5631 
5632 	ASSERT0(zio->io_post & ZIO_POST_DIO_CHKSUM_ERR);
5633 	IMPLY(zio->io_flags & ZIO_FLAG_DIO_READ,
5634 	    !(zio->io_flags & ZIO_FLAG_SPECULATIVE));
5635 
5636 	if ((error = zio_checksum_error(zio, &info)) != 0) {
5637 		zio->io_error = error;
5638 		if (error == ECKSUM &&
5639 		    !(zio->io_flags & ZIO_FLAG_SPECULATIVE)) {
5640 			if (zio->io_flags & ZIO_FLAG_DIO_READ) {
5641 				zio->io_post |= ZIO_POST_DIO_CHKSUM_ERR;
5642 				zio_t *pio = zio_unique_parent(zio);
5643 				/*
5644 				 * Any Direct I/O read that has a checksum
5645 				 * error must be treated as suspicous as the
5646 				 * contents of the buffer could be getting
5647 				 * manipulated while the I/O is taking place.
5648 				 *
5649 				 * The checksum verify error will only be
5650 				 * reported here for disk and file VDEV's and
5651 				 * will be reported on those that the failure
5652 				 * occurred on. Other types of VDEV's report the
5653 				 * verify failure in their own code paths.
5654 				 */
5655 				if (pio->io_child_type == ZIO_CHILD_LOGICAL) {
5656 					zio_dio_chksum_verify_error_report(zio);
5657 				}
5658 			} else {
5659 				mutex_enter(&zio->io_vd->vdev_stat_lock);
5660 				zio->io_vd->vdev_stat.vs_checksum_errors++;
5661 				mutex_exit(&zio->io_vd->vdev_stat_lock);
5662 				(void) zfs_ereport_start_checksum(zio->io_spa,
5663 				    zio->io_vd, &zio->io_bookmark, zio,
5664 				    zio->io_offset, zio->io_size, &info);
5665 			}
5666 		}
5667 	}
5668 
5669 	return (zio);
5670 }
5671 
5672 static zio_t *
5673 zio_dio_checksum_verify(zio_t *zio)
5674 {
5675 	zio_t *pio = zio_unique_parent(zio);
5676 	int error;
5677 
5678 	ASSERT3P(zio->io_vd, !=, NULL);
5679 	ASSERT3P(zio->io_bp, !=, NULL);
5680 	ASSERT3U(zio->io_child_type, ==, ZIO_CHILD_VDEV);
5681 	ASSERT3U(zio->io_type, ==, ZIO_TYPE_WRITE);
5682 	ASSERT3B(pio->io_prop.zp_direct_write, ==, B_TRUE);
5683 	ASSERT3U(pio->io_child_type, ==, ZIO_CHILD_LOGICAL);
5684 
5685 	if (zfs_vdev_direct_write_verify == 0 || zio->io_error != 0)
5686 		goto out;
5687 
5688 	if ((error = zio_checksum_error(zio, NULL)) != 0) {
5689 		zio->io_error = error;
5690 		if (error == ECKSUM) {
5691 			zio->io_post |= ZIO_POST_DIO_CHKSUM_ERR;
5692 			zio_dio_chksum_verify_error_report(zio);
5693 		}
5694 	}
5695 
5696 out:
5697 	return (zio);
5698 }
5699 
5700 
5701 /*
5702  * Called by RAID-Z to ensure we don't compute the checksum twice.
5703  */
5704 void
5705 zio_checksum_verified(zio_t *zio)
5706 {
5707 	zio->io_pipeline &= ~ZIO_STAGE_CHECKSUM_VERIFY;
5708 }
5709 
5710 /*
5711  * Report Direct I/O checksum verify error and create ZED event.
5712  */
5713 void
5714 zio_dio_chksum_verify_error_report(zio_t *zio)
5715 {
5716 	ASSERT(zio->io_post & ZIO_POST_DIO_CHKSUM_ERR);
5717 
5718 	if (zio->io_child_type == ZIO_CHILD_LOGICAL)
5719 		return;
5720 
5721 	mutex_enter(&zio->io_vd->vdev_stat_lock);
5722 	zio->io_vd->vdev_stat.vs_dio_verify_errors++;
5723 	mutex_exit(&zio->io_vd->vdev_stat_lock);
5724 	if (zio->io_type == ZIO_TYPE_WRITE) {
5725 		/*
5726 		 * Convert checksum error for writes into EIO.
5727 		 */
5728 		zio->io_error = SET_ERROR(EIO);
5729 		/*
5730 		 * Report dio_verify_wr ZED event, rate limited.
5731 		 */
5732 		if (zfs_ratelimit(&zio->io_vd->vdev_dio_verify_rl))
5733 			(void) zfs_ereport_post(FM_EREPORT_ZFS_DIO_VERIFY_WR,
5734 			    zio->io_spa, zio->io_vd, &zio->io_bookmark, zio, 0);
5735 	} else {
5736 		/*
5737 		 * Report dio_verify_rd ZED event, rate limited.
5738 		 */
5739 		if (zfs_ratelimit(&zio->io_vd->vdev_dio_verify_rl))
5740 			(void) zfs_ereport_post(FM_EREPORT_ZFS_DIO_VERIFY_RD,
5741 			    zio->io_spa, zio->io_vd, &zio->io_bookmark, zio, 0);
5742 	}
5743 }
5744 
5745 /*
5746  * ==========================================================================
5747  * Error rank.  Error are ranked in the order 0, ENXIO, ECKSUM, EIO, other.
5748  * An error of 0 indicates success.  ENXIO indicates whole-device failure,
5749  * which may be transient (e.g. unplugged) or permanent.  ECKSUM and EIO
5750  * indicate errors that are specific to one I/O, and most likely permanent.
5751  * Any other error is presumed to be worse because we weren't expecting it.
5752  * ==========================================================================
5753  */
5754 int
5755 zio_worst_error(int e1, int e2)
5756 {
5757 	static int zio_error_rank[] = { 0, ENXIO, ECKSUM, EIO };
5758 	int r1, r2;
5759 
5760 	for (r1 = 0; r1 < sizeof (zio_error_rank) / sizeof (int); r1++)
5761 		if (e1 == zio_error_rank[r1])
5762 			break;
5763 
5764 	for (r2 = 0; r2 < sizeof (zio_error_rank) / sizeof (int); r2++)
5765 		if (e2 == zio_error_rank[r2])
5766 			break;
5767 
5768 	return (r1 > r2 ? e1 : e2);
5769 }
5770 
5771 /*
5772  * ==========================================================================
5773  * I/O completion
5774  * ==========================================================================
5775  */
5776 static zio_t *
5777 zio_ready(zio_t *zio)
5778 {
5779 	blkptr_t *bp = zio->io_bp;
5780 	zio_t *pio, *pio_next;
5781 	zio_link_t *zl = NULL;
5782 
5783 	if (zio_wait_for_children(zio, ZIO_CHILD_LOGICAL_BIT |
5784 	    ZIO_CHILD_GANG_BIT | ZIO_CHILD_DDT_BIT, ZIO_WAIT_READY)) {
5785 		return (NULL);
5786 	}
5787 
5788 	if (zio_injection_enabled) {
5789 		hrtime_t target = zio_handle_ready_delay(zio);
5790 		if (target != 0 && zio->io_target_timestamp == 0) {
5791 			zio->io_stage >>= 1;
5792 			zio->io_target_timestamp = target;
5793 			zio_delay_interrupt(zio);
5794 			return (NULL);
5795 		}
5796 	}
5797 
5798 	if (zio->io_ready) {
5799 		ASSERT(IO_IS_ALLOCATING(zio));
5800 		ASSERT(BP_GET_BIRTH(bp) == zio->io_txg ||
5801 		    BP_IS_HOLE(bp) || (zio->io_flags & ZIO_FLAG_NOPWRITE));
5802 		ASSERT0(zio->io_children[ZIO_CHILD_GANG][ZIO_WAIT_READY]);
5803 
5804 		zio->io_ready(zio);
5805 	}
5806 
5807 #ifdef ZFS_DEBUG
5808 	if (bp != NULL && bp != &zio->io_bp_copy)
5809 		zio->io_bp_copy = *bp;
5810 #endif
5811 
5812 	if (zio->io_error != 0) {
5813 		zio->io_pipeline = ZIO_INTERLOCK_PIPELINE;
5814 
5815 		if (zio->io_flags & ZIO_FLAG_ALLOC_THROTTLED) {
5816 			ASSERT(IO_IS_ALLOCATING(zio));
5817 			ASSERT(zio->io_priority == ZIO_PRIORITY_ASYNC_WRITE);
5818 			ASSERT(zio->io_metaslab_class != NULL);
5819 			ASSERT(ZIO_HAS_ALLOCATOR(zio));
5820 
5821 			/*
5822 			 * We were unable to allocate anything, unreserve and
5823 			 * issue the next I/O to allocate.
5824 			 */
5825 			if (metaslab_class_throttle_unreserve(
5826 			    zio->io_metaslab_class, zio->io_allocator,
5827 			    zio->io_prop.zp_copies, zio->io_size)) {
5828 				zio_allocate_dispatch(zio->io_metaslab_class,
5829 				    zio->io_allocator);
5830 			}
5831 		}
5832 	}
5833 
5834 	mutex_enter(&zio->io_lock);
5835 	zio->io_state[ZIO_WAIT_READY] = 1;
5836 	pio = zio_walk_parents(zio, &zl);
5837 	mutex_exit(&zio->io_lock);
5838 
5839 	/*
5840 	 * As we notify zio's parents, new parents could be added.
5841 	 * New parents go to the head of zio's io_parent_list, however,
5842 	 * so we will (correctly) not notify them.  The remainder of zio's
5843 	 * io_parent_list, from 'pio_next' onward, cannot change because
5844 	 * all parents must wait for us to be done before they can be done.
5845 	 */
5846 	zio_next_t next;
5847 	zio_next_init(&next);
5848 	for (; pio != NULL; pio = pio_next) {
5849 		pio_next = zio_walk_parents(zio, &zl);
5850 		zio_notify_parent(pio, zio, ZIO_WAIT_READY, &next);
5851 	}
5852 	ASSERT3P(zio->io_exec_next, ==, NULL);
5853 	zio->io_exec_next = next.zn_list;
5854 
5855 	if (zio->io_flags & ZIO_FLAG_NODATA) {
5856 		if (bp != NULL && BP_IS_GANG(bp)) {
5857 			zio->io_flags &= ~ZIO_FLAG_NODATA;
5858 		} else {
5859 			ASSERT((uintptr_t)zio->io_abd < SPA_MAXBLOCKSIZE);
5860 			zio->io_pipeline &= ~ZIO_VDEV_IO_STAGES;
5861 		}
5862 	}
5863 
5864 	if (zio_injection_enabled &&
5865 	    zio->io_spa->spa_syncing_txg == zio->io_txg)
5866 		zio_handle_ignored_writes(zio);
5867 
5868 	return (zio);
5869 }
5870 
5871 /*
5872  * Update the allocation throttle accounting.
5873  */
5874 static void
5875 zio_dva_throttle_done(zio_t *zio)
5876 {
5877 	zio_t *pio = zio_unique_parent(zio);
5878 	vdev_t *vd = zio->io_vd;
5879 	int flags = METASLAB_ASYNC_ALLOC;
5880 	const void *tag = pio;
5881 	uint64_t size = pio->io_size;
5882 
5883 	ASSERT3P(zio->io_bp, !=, NULL);
5884 	ASSERT3U(zio->io_type, ==, ZIO_TYPE_WRITE);
5885 	ASSERT3U(zio->io_priority, ==, ZIO_PRIORITY_ASYNC_WRITE);
5886 	ASSERT3U(zio->io_child_type, ==, ZIO_CHILD_VDEV);
5887 	ASSERT(vd != NULL);
5888 	ASSERT3P(vd, ==, vd->vdev_top);
5889 	ASSERT(zio_injection_enabled || !(zio->io_flags & ZIO_FLAG_IO_RETRY));
5890 	ASSERT(!(zio->io_flags & ZIO_FLAG_IO_REPAIR));
5891 	ASSERT(zio->io_flags & ZIO_FLAG_ALLOC_THROTTLED);
5892 
5893 	/*
5894 	 * Parents of gang children can have two flavors -- ones that allocated
5895 	 * the gang header (will have ZIO_FLAG_IO_REWRITE set) and ones that
5896 	 * allocated the constituent blocks.  The first use their parent as tag.
5897 	 * We set the size to match the original allocation call for that case.
5898 	 */
5899 	if (pio->io_child_type == ZIO_CHILD_GANG &&
5900 	    (pio->io_flags & ZIO_FLAG_IO_REWRITE)) {
5901 		tag = zio_unique_parent(pio);
5902 		size = SPA_OLD_GANGBLOCKSIZE;
5903 	}
5904 
5905 	ASSERT(IO_IS_ALLOCATING(pio) || (pio->io_child_type == ZIO_CHILD_GANG &&
5906 	    (pio->io_flags & ZIO_FLAG_IO_REWRITE)));
5907 	ASSERT(ZIO_HAS_ALLOCATOR(pio));
5908 	ASSERT3P(zio, !=, zio->io_logical);
5909 	ASSERT(zio->io_logical != NULL);
5910 	ASSERT(!(zio->io_flags & ZIO_FLAG_IO_REPAIR));
5911 	ASSERT0(zio->io_flags & ZIO_FLAG_NOPWRITE);
5912 	ASSERT(zio->io_metaslab_class != NULL);
5913 	ASSERT(zio->io_metaslab_class->mc_alloc_throttle_enabled);
5914 
5915 	metaslab_group_alloc_decrement(zio->io_spa, vd->vdev_id,
5916 	    pio->io_allocator, flags, size, tag);
5917 
5918 	if (metaslab_class_throttle_unreserve(pio->io_metaslab_class,
5919 	    pio->io_allocator, 1, pio->io_size)) {
5920 		zio_allocate_dispatch(zio->io_metaslab_class,
5921 		    pio->io_allocator);
5922 	}
5923 }
5924 
5925 static void
5926 zio_done_postread_done(zio_t *zio)
5927 {
5928 	abd_free(zio->io_abd);
5929 }
5930 
5931 static zio_t *
5932 zio_done(zio_t *zio)
5933 {
5934 	/*
5935 	 * Always attempt to keep stack usage minimal here since
5936 	 * we can be called recursively up to 19 levels deep.
5937 	 */
5938 	const uint64_t psize = zio->io_size;
5939 	zio_t *pio, *pio_next;
5940 	zio_link_t *zl = NULL;
5941 
5942 	/*
5943 	 * If our children haven't all completed,
5944 	 * wait for them and then repeat this pipeline stage.
5945 	 */
5946 	if (zio_wait_for_children(zio, ZIO_CHILD_ALL_BITS, ZIO_WAIT_DONE)) {
5947 		return (NULL);
5948 	}
5949 
5950 	/*
5951 	 * If the allocation throttle is enabled, then update the accounting.
5952 	 * We only track child I/Os that are part of an allocating async
5953 	 * write. We must do this since the allocation is performed
5954 	 * by the logical I/O but the actual write is done by child I/Os.
5955 	 */
5956 	if (zio->io_flags & ZIO_FLAG_ALLOC_THROTTLED &&
5957 	    zio->io_child_type == ZIO_CHILD_VDEV)
5958 		zio_dva_throttle_done(zio);
5959 
5960 	for (int c = 0; c < ZIO_CHILD_TYPES; c++)
5961 		for (int w = 0; w < ZIO_WAIT_TYPES; w++)
5962 			ASSERT0(zio->io_children[c][w]);
5963 
5964 	if (zio->io_bp != NULL && !BP_IS_EMBEDDED(zio->io_bp)) {
5965 		ASSERT(memcmp(zio->io_bp, &zio->io_bp_copy,
5966 		    sizeof (blkptr_t)) == 0 ||
5967 		    (zio->io_bp == zio_unique_parent(zio)->io_bp));
5968 		if (zio->io_type == ZIO_TYPE_WRITE && !BP_IS_HOLE(zio->io_bp) &&
5969 		    zio->io_bp_override == NULL &&
5970 		    !(zio->io_flags & ZIO_FLAG_IO_REPAIR)) {
5971 			ASSERT3U(zio->io_prop.zp_copies, <=,
5972 			    BP_GET_NDVAS(zio->io_bp));
5973 			ASSERT(BP_COUNT_GANG(zio->io_bp) == 0 ||
5974 			    (BP_COUNT_GANG(zio->io_bp) ==
5975 			    BP_GET_NDVAS(zio->io_bp)));
5976 		}
5977 		if (zio->io_flags & ZIO_FLAG_NOPWRITE)
5978 			VERIFY(BP_EQUAL(zio->io_bp, &zio->io_bp_orig));
5979 	}
5980 
5981 	/*
5982 	 * If there were child vdev/gang/ddt errors, they apply to us now.
5983 	 */
5984 	zio_inherit_child_errors(zio, ZIO_CHILD_VDEV);
5985 	zio_inherit_child_errors(zio, ZIO_CHILD_GANG);
5986 	zio_inherit_child_errors(zio, ZIO_CHILD_DDT);
5987 
5988 	/*
5989 	 * If the I/O on the transformed data was successful, generate any
5990 	 * checksum reports now while we still have the transformed data.
5991 	 */
5992 	if (zio->io_error == 0) {
5993 		while (zio->io_cksum_report != NULL) {
5994 			zio_cksum_report_t *zcr = zio->io_cksum_report;
5995 			uint64_t align = zcr->zcr_align;
5996 			uint64_t asize = P2ROUNDUP(psize, align);
5997 			abd_t *adata = zio->io_abd;
5998 
5999 			if (adata != NULL && asize != psize) {
6000 				adata = abd_alloc(asize, B_TRUE);
6001 				abd_copy(adata, zio->io_abd, psize);
6002 				abd_zero_off(adata, psize, asize - psize);
6003 			}
6004 
6005 			zio->io_cksum_report = zcr->zcr_next;
6006 			zcr->zcr_next = NULL;
6007 			zcr->zcr_finish(zcr, adata);
6008 			zfs_ereport_free_checksum(zcr);
6009 
6010 			if (adata != NULL && asize != psize)
6011 				abd_free(adata);
6012 		}
6013 	}
6014 
6015 	zio_pop_transforms(zio);	/* note: may set zio->io_error */
6016 
6017 	/*
6018 	 * During thorough scrub, if the dataset key is not loaded, decryption
6019 	 * or MAC verification fails with EACCES (spa_do_crypt_abd() and the
6020 	 * MAC helpers). Since the block's checksum was already successfully
6021 	 * verified by zio_checksum_verify() before we got here, treat it as
6022 	 * success and move on; this is as much as we can do without the keys
6023 	 * loaded.
6024 	 */
6025 	if (zio->io_error == EACCES && (zio->io_flags & ZIO_FLAG_SCRUB) &&
6026 	    !(zio->io_flags & ZIO_FLAG_RAW))
6027 		zio->io_error = 0;
6028 
6029 	vdev_stat_update(zio, psize);
6030 
6031 	/*
6032 	 * If this I/O is attached to a particular vdev is slow, exceeding
6033 	 * 30 seconds to complete, post an error described the I/O delay.
6034 	 * We ignore these errors if the device is currently unavailable.
6035 	 */
6036 	if (zio->io_delay >= MSEC2NSEC(zio_slow_io_ms)) {
6037 		if (zio->io_vd != NULL && !vdev_is_dead(zio->io_vd)) {
6038 			/*
6039 			 * We want to only increment our slow IO counters if
6040 			 * the IO is valid (i.e. not if the drive is removed).
6041 			 *
6042 			 * zfs_ereport_post() will also do these checks, but
6043 			 * it can also ratelimit and have other failures, so we
6044 			 * need to increment the slow_io counters independent
6045 			 * of it.
6046 			 */
6047 			if (zfs_ereport_is_valid(FM_EREPORT_ZFS_DELAY,
6048 			    zio->io_spa, zio->io_vd, zio)) {
6049 				mutex_enter(&zio->io_vd->vdev_stat_lock);
6050 				zio->io_vd->vdev_stat.vs_slow_ios++;
6051 				mutex_exit(&zio->io_vd->vdev_stat_lock);
6052 
6053 				if (zio->io_vd->vdev_slow_io_events) {
6054 					(void) zfs_ereport_post(
6055 					    FM_EREPORT_ZFS_DELAY,
6056 					    zio->io_spa, zio->io_vd,
6057 					    &zio->io_bookmark, zio, 0);
6058 				}
6059 			}
6060 		}
6061 	}
6062 
6063 	if (zio->io_error) {
6064 		/*
6065 		 * If this I/O is attached to a particular vdev,
6066 		 * generate an error message describing the I/O failure
6067 		 * at the block level.  We ignore these errors if the
6068 		 * device is currently unavailable.
6069 		 */
6070 		if (zio->io_error != ECKSUM && zio->io_vd != NULL &&
6071 		    !vdev_is_dead(zio->io_vd) &&
6072 		    !(zio->io_post & ZIO_POST_DIO_CHKSUM_ERR)) {
6073 			int ret = zfs_ereport_post(FM_EREPORT_ZFS_IO,
6074 			    zio->io_spa, zio->io_vd, &zio->io_bookmark, zio, 0);
6075 			if (ret != EALREADY) {
6076 				mutex_enter(&zio->io_vd->vdev_stat_lock);
6077 				if (zio->io_type == ZIO_TYPE_READ)
6078 					zio->io_vd->vdev_stat.vs_read_errors++;
6079 				else if (zio->io_type == ZIO_TYPE_WRITE)
6080 					zio->io_vd->vdev_stat.vs_write_errors++;
6081 				mutex_exit(&zio->io_vd->vdev_stat_lock);
6082 			}
6083 		}
6084 
6085 		if ((zio->io_error == EIO || !(zio->io_flags &
6086 		    (ZIO_FLAG_SPECULATIVE | ZIO_FLAG_DONT_PROPAGATE))) &&
6087 		    !(zio->io_post & ZIO_POST_DIO_CHKSUM_ERR) &&
6088 		    zio == zio->io_logical) {
6089 			/*
6090 			 * For logical I/O requests, tell the SPA to log the
6091 			 * error and generate a logical data ereport.
6092 			 */
6093 			spa_log_error(zio->io_spa, &zio->io_bookmark,
6094 			    BP_GET_PHYSICAL_BIRTH(zio->io_bp));
6095 			(void) zfs_ereport_post(FM_EREPORT_ZFS_DATA,
6096 			    zio->io_spa, NULL, &zio->io_bookmark, zio, 0);
6097 		}
6098 	}
6099 
6100 	if (zio->io_error && zio == zio->io_logical) {
6101 
6102 		/*
6103 		 * A DDT child tried to create a mixed gang/non-gang BP. We're
6104 		 * going to have to just retry as a non-dedup IO.
6105 		 */
6106 		if (zio->io_error == EAGAIN && IO_IS_ALLOCATING(zio) &&
6107 		    zio->io_prop.zp_dedup) {
6108 			zio->io_post |= ZIO_POST_REEXECUTE;
6109 			zio->io_prop.zp_dedup = B_FALSE;
6110 		}
6111 		/*
6112 		 * Determine whether zio should be reexecuted.  This will
6113 		 * propagate all the way to the root via zio_notify_parent().
6114 		 */
6115 		ASSERT(zio->io_vd == NULL && zio->io_bp != NULL);
6116 		ASSERT(zio->io_child_type == ZIO_CHILD_LOGICAL);
6117 
6118 		if (IO_IS_ALLOCATING(zio) &&
6119 		    !(zio->io_flags & ZIO_FLAG_CANFAIL) &&
6120 		    !(zio->io_post & ZIO_POST_DIO_CHKSUM_ERR)) {
6121 			if (zio->io_error != ENOSPC)
6122 				zio->io_post |= ZIO_POST_REEXECUTE;
6123 			else
6124 				zio->io_post |= ZIO_POST_SUSPEND;
6125 		}
6126 
6127 		if ((zio->io_type == ZIO_TYPE_READ ||
6128 		    zio->io_type == ZIO_TYPE_FREE) &&
6129 		    !(zio->io_flags & ZIO_FLAG_SCAN_THREAD) &&
6130 		    zio->io_error == ENXIO &&
6131 		    spa_load_state(zio->io_spa) == SPA_LOAD_NONE &&
6132 		    spa_get_failmode(zio->io_spa) != ZIO_FAILURE_MODE_CONTINUE)
6133 			zio->io_post |= ZIO_POST_SUSPEND;
6134 
6135 		if (!(zio->io_flags & ZIO_FLAG_CANFAIL) &&
6136 		    !(zio->io_post & (ZIO_POST_REEXECUTE|ZIO_POST_SUSPEND)))
6137 			zio->io_post |= ZIO_POST_SUSPEND;
6138 
6139 		/*
6140 		 * Here is a possibly good place to attempt to do
6141 		 * either combinatorial reconstruction or error correction
6142 		 * based on checksums.  It also might be a good place
6143 		 * to send out preliminary ereports before we suspend
6144 		 * processing.
6145 		 */
6146 	}
6147 
6148 	/*
6149 	 * If there were logical child errors, they apply to us now.
6150 	 * We defer this until now to avoid conflating logical child
6151 	 * errors with errors that happened to the zio itself when
6152 	 * updating vdev stats and reporting FMA events above.
6153 	 */
6154 	zio_inherit_child_errors(zio, ZIO_CHILD_LOGICAL);
6155 
6156 	if ((zio->io_error ||
6157 	    (zio->io_post & (ZIO_POST_REEXECUTE|ZIO_POST_SUSPEND))) &&
6158 	    IO_IS_ALLOCATING(zio) && zio->io_gang_leader == zio &&
6159 	    !(zio->io_flags & (ZIO_FLAG_IO_REWRITE | ZIO_FLAG_NOPWRITE)))
6160 		zio_dva_unallocate(zio, zio->io_gang_tree, zio->io_bp);
6161 
6162 	zio_gang_tree_free(&zio->io_gang_tree);
6163 
6164 	/*
6165 	 * Godfather I/Os should never suspend.
6166 	 */
6167 	if ((zio->io_flags & ZIO_FLAG_GODFATHER) &&
6168 	    (zio->io_post & ZIO_POST_SUSPEND))
6169 		zio->io_post &= ~ZIO_POST_SUSPEND;
6170 
6171 	if (zio->io_post & (ZIO_POST_REEXECUTE|ZIO_POST_SUSPEND)) {
6172 		/*
6173 		 * A Direct I/O operation that has a checksum verify error
6174 		 * should not attempt to reexecute. Instead, the error should
6175 		 * just be propagated back.
6176 		 */
6177 		ASSERT0(zio->io_post & ZIO_POST_DIO_CHKSUM_ERR);
6178 
6179 		/*
6180 		 * This is a logical I/O that wants to reexecute.
6181 		 *
6182 		 * Reexecute is top-down.  When an i/o fails, if it's not
6183 		 * the root, it simply notifies its parent and sticks around.
6184 		 * The parent, seeing that it still has children in zio_done(),
6185 		 * does the same.  This percolates all the way up to the root.
6186 		 * The root i/o will reexecute or suspend the entire tree.
6187 		 *
6188 		 * This approach ensures that zio_reexecute() honors
6189 		 * all the original i/o dependency relationships, e.g.
6190 		 * parents not executing until children are ready.
6191 		 */
6192 		ASSERT(zio->io_child_type == ZIO_CHILD_LOGICAL);
6193 
6194 		zio->io_gang_leader = NULL;
6195 
6196 		mutex_enter(&zio->io_lock);
6197 		zio->io_state[ZIO_WAIT_DONE] = 1;
6198 		mutex_exit(&zio->io_lock);
6199 
6200 		/*
6201 		 * "The Godfather" I/O monitors its children but is
6202 		 * not a true parent to them. It will track them through
6203 		 * the pipeline but severs its ties whenever they get into
6204 		 * trouble (e.g. suspended). This allows "The Godfather"
6205 		 * I/O to return status without blocking.
6206 		 */
6207 		zl = NULL;
6208 		for (pio = zio_walk_parents(zio, &zl); pio != NULL;
6209 		    pio = pio_next) {
6210 			zio_link_t *remove_zl = zl;
6211 			pio_next = zio_walk_parents(zio, &zl);
6212 
6213 			if ((pio->io_flags & ZIO_FLAG_GODFATHER) &&
6214 			    (zio->io_post & ZIO_POST_SUSPEND)) {
6215 				zio_remove_child(pio, zio, remove_zl);
6216 				/*
6217 				 * This is a rare code path, so we don't
6218 				 * bother with the "next" list.
6219 				 */
6220 				zio_notify_parent(pio, zio, ZIO_WAIT_DONE,
6221 				    NULL);
6222 			}
6223 		}
6224 
6225 		if ((pio = zio_unique_parent(zio)) != NULL) {
6226 			/*
6227 			 * We're not a root i/o, so there's nothing to do
6228 			 * but notify our parent.  Don't propagate errors
6229 			 * upward since we haven't permanently failed yet.
6230 			 */
6231 			ASSERT(!(zio->io_flags & ZIO_FLAG_GODFATHER));
6232 			zio->io_flags |= ZIO_FLAG_DONT_PROPAGATE;
6233 			/*
6234 			 * This is a rare code path, so we don't bother with
6235 			 * the "next" list.
6236 			 */
6237 			zio_notify_parent(pio, zio, ZIO_WAIT_DONE, NULL);
6238 		} else if (zio->io_post & ZIO_POST_SUSPEND) {
6239 			/*
6240 			 * We'd fail again if we reexecuted now, so suspend
6241 			 * until conditions improve (e.g. device comes online).
6242 			 */
6243 			zio_suspend(zio->io_spa, zio, ZIO_SUSPEND_IOERR);
6244 		} else {
6245 			ASSERT(zio->io_post & ZIO_POST_REEXECUTE);
6246 			/*
6247 			 * Reexecution is potentially a huge amount of work.
6248 			 * Hand it off to the otherwise-unused claim taskq.
6249 			 */
6250 			spa_taskq_dispatch(zio->io_spa,
6251 			    ZIO_TYPE_CLAIM, ZIO_TASKQ_ISSUE,
6252 			    zio_reexecute, zio, B_FALSE);
6253 		}
6254 		return (NULL);
6255 	}
6256 
6257 	ASSERT(list_is_empty(&zio->io_child_list));
6258 	ASSERT0(zio->io_post & ZIO_POST_REEXECUTE);
6259 	ASSERT0(zio->io_post & ZIO_POST_SUSPEND);
6260 	ASSERT(zio->io_error == 0 || (zio->io_flags & ZIO_FLAG_CANFAIL));
6261 
6262 	/*
6263 	 * Report any checksum errors, since the I/O is complete.
6264 	 */
6265 	while (zio->io_cksum_report != NULL) {
6266 		zio_cksum_report_t *zcr = zio->io_cksum_report;
6267 		zio->io_cksum_report = zcr->zcr_next;
6268 		zcr->zcr_next = NULL;
6269 		zcr->zcr_finish(zcr, NULL);
6270 		zfs_ereport_free_checksum(zcr);
6271 	}
6272 
6273 	if (zio->io_flags & ZIO_FLAG_POSTREAD) {
6274 		ASSERT3U(zio->io_type, ==, ZIO_TYPE_WRITE);
6275 		zl = NULL;
6276 		zio_t *pio = zio_walk_parents(zio, &zl);
6277 		blkptr_t *bp = zio->io_bp;
6278 		abd_t *abd = abd_alloc_for_io(BP_GET_PSIZE(bp), B_FALSE);
6279 		zio_priority_t prio = zio->io_priority ==
6280 		    ZIO_PRIORITY_SYNC_WRITE ? ZIO_PRIORITY_SYNC_READ :
6281 		    ZIO_PRIORITY_SCRUB;
6282 		zio_t *cio = zio_vdev_child_io(pio, zio->io_bp, zio->io_vd,
6283 		    zio->io_offset, abd, zio->io_size, ZIO_TYPE_READ, prio,
6284 		    ZIO_FLAG_SCRUB | ZIO_FLAG_RAW | ZIO_FLAG_CANFAIL |
6285 		    ZIO_FLAG_RESILVER | ZIO_FLAG_DONT_PROPAGATE,
6286 		    zio_done_postread_done, NULL);
6287 		cio->io_flags &= ~ZIO_FLAG_ALLOC_THROTTLED;
6288 		zio_nowait(cio);
6289 	}
6290 
6291 	/*
6292 	 * It is the responsibility of the done callback to ensure that this
6293 	 * particular zio is no longer discoverable for adoption, and as
6294 	 * such, cannot acquire any new parents.
6295 	 */
6296 	if (zio->io_done)
6297 		zio->io_done(zio);
6298 
6299 	mutex_enter(&zio->io_lock);
6300 	zio->io_state[ZIO_WAIT_DONE] = 1;
6301 	mutex_exit(&zio->io_lock);
6302 
6303 	/*
6304 	 * We are done executing this zio.  We may want to execute some of its
6305 	 * parents next.  See the comment in zio_notify_parent().
6306 	 */
6307 	zio_next_t next;
6308 	zio_next_init(&next);
6309 	zl = NULL;
6310 	for (pio = zio_walk_parents(zio, &zl); pio != NULL; pio = pio_next) {
6311 		zio_link_t *remove_zl = zl;
6312 		pio_next = zio_walk_parents(zio, &zl);
6313 		zio_remove_child(pio, zio, remove_zl);
6314 		zio_notify_parent(pio, zio, ZIO_WAIT_DONE, &next);
6315 	}
6316 
6317 	if (zio->io_waiter != NULL) {
6318 		mutex_enter(&zio->io_lock);
6319 		zio->io_executor = NULL;
6320 		cv_broadcast(&zio->io_cv);
6321 		mutex_exit(&zio->io_lock);
6322 	} else {
6323 		zio_destroy(zio);
6324 	}
6325 
6326 	return (next.zn_list);
6327 }
6328 
6329 /*
6330  * ==========================================================================
6331  * I/O pipeline definition
6332  * ==========================================================================
6333  */
6334 static zio_pipe_stage_t *zio_pipeline[] = {
6335 	NULL,
6336 	zio_read_bp_init,
6337 	zio_write_bp_init,
6338 	zio_free_bp_init,
6339 	zio_issue_async,
6340 	zio_write_compress,
6341 	zio_encrypt,
6342 	zio_checksum_generate,
6343 	zio_nop_write,
6344 	zio_ddt_read_start,
6345 	zio_ddt_read_done,
6346 	zio_ddt_write,
6347 	zio_ddt_free,
6348 	zio_brt_free,
6349 	zio_gang_assemble,
6350 	zio_gang_issue,
6351 	zio_dva_throttle,
6352 	zio_dva_allocate,
6353 	zio_dva_free,
6354 	zio_dva_claim,
6355 	zio_ready,
6356 	zio_vdev_io_start,
6357 	zio_vdev_io_done,
6358 	zio_vdev_io_assess,
6359 	zio_checksum_verify,
6360 	zio_dio_checksum_verify,
6361 	zio_done
6362 };
6363 
6364 
6365 
6366 
6367 /*
6368  * Compare two zbookmark_phys_t's to see which we would reach first in a
6369  * pre-order traversal of the object tree.
6370  *
6371  * This is simple in every case aside from the meta-dnode object. For all other
6372  * objects, we traverse them in order (object 1 before object 2, and so on).
6373  * However, all of these objects are traversed while traversing object 0, since
6374  * the data it points to is the list of objects.  Thus, we need to convert to a
6375  * canonical representation so we can compare meta-dnode bookmarks to
6376  * non-meta-dnode bookmarks.
6377  *
6378  * We do this by calculating "equivalents" for each field of the zbookmark.
6379  * zbookmarks outside of the meta-dnode use their own object and level, and
6380  * calculate the level 0 equivalent (the first L0 blkid that is contained in the
6381  * blocks this bookmark refers to) by multiplying their blkid by their span
6382  * (the number of L0 blocks contained within one block at their level).
6383  * zbookmarks inside the meta-dnode calculate their object equivalent
6384  * (which is L0equiv * dnodes per data block), use 0 for their L0equiv, and use
6385  * level + 1<<31 (any value larger than a level could ever be) for their level.
6386  * This causes them to always compare before a bookmark in their object
6387  * equivalent, compare appropriately to bookmarks in other objects, and to
6388  * compare appropriately to other bookmarks in the meta-dnode.
6389  */
6390 int
6391 zbookmark_compare(uint16_t dbss1, uint8_t ibs1, uint16_t dbss2, uint8_t ibs2,
6392     const zbookmark_phys_t *zb1, const zbookmark_phys_t *zb2)
6393 {
6394 	/*
6395 	 * These variables represent the "equivalent" values for the zbookmark,
6396 	 * after converting zbookmarks inside the meta dnode to their
6397 	 * normal-object equivalents.
6398 	 */
6399 	uint64_t zb1obj, zb2obj;
6400 	uint64_t zb1L0, zb2L0;
6401 	uint64_t zb1level, zb2level;
6402 
6403 	if (zb1->zb_object == zb2->zb_object &&
6404 	    zb1->zb_level == zb2->zb_level &&
6405 	    zb1->zb_blkid == zb2->zb_blkid)
6406 		return (0);
6407 
6408 	if (zb1->zb_level < 0 || zb2->zb_level < 0) {
6409 		/*
6410 		 * "Negative" levels are ZB_ROOT_LEVEL, ZB_ZIL_LEVEL or
6411 		 * ZB_DNODE_LEVEL, and represent some sort of auxiliary dataset
6412 		 * block or object. In this case, we're usually being called
6413 		 * from dsl_scan or dmu_traverse.
6414 		 *
6415 		 * These "levels" are more like a "type" signal, not directly
6416 		 * comparable, but we have to do something. So we order them in
6417 		 * the order we would see them during a typical scan or
6418 		 * traverse:
6419 		 *
6420 		 * - ZB_ROOT_LEVEL: the "top" block carrying the dataset head
6421 		 * - ZB_ZIL_LEVEL: the head ZIL block attached to the dataset
6422 		 * - ZB_DNODE_LEVEL: "virtual" position representing an
6423 		 *                   entire object. Sorts ahead of the true
6424 		 *                   data blocks for the object.
6425 		 * - level >= 0: data blocks
6426 		 *
6427 		 * We work through these cases from top to bottom, with
6428 		 * appropriate tiebreaks for each kind.
6429 		 */
6430 
6431 		/*
6432 		 * Root level wins. It shouldn't be possible for both to be the
6433 		 * root level in this per-dataset tree, and there's no obvious
6434 		 * tiebreaker, but we handle it as a defensive measure.
6435 		 */
6436 		if (zb1->zb_level == ZB_ROOT_LEVEL &&
6437 		    zb2->zb_level == ZB_ROOT_LEVEL)
6438 			return (TREE_PCMP(zb1, zb2));
6439 		if (zb1->zb_level == ZB_ROOT_LEVEL)
6440 			return (-1);
6441 		if (zb2->zb_level == ZB_ROOT_LEVEL)
6442 			return (1);
6443 
6444 		/* ZIL bookmarks have valid blkid, so the earlier one wins. */
6445 		if (zb1->zb_level == ZB_ZIL_LEVEL &&
6446 		    zb2->zb_level == ZB_ZIL_LEVEL)
6447 			return (TREE_CMP(zb1->zb_blkid, zb2->zb_blkid));
6448 		if (zb1->zb_level == ZB_ZIL_LEVEL)
6449 			return (-1);
6450 		if (zb2->zb_level == ZB_ZIL_LEVEL)
6451 			return (1);
6452 
6453 		/*
6454 		 * If we get this far, then at least one is ZB_DNODE_LEVEL, and
6455 		 * the other is either ZB_DNODE_LEVEL or a data block.
6456 		 * Regardless, the one with the lower-numbered object wins -
6457 		 * earler ZB_DNODE_LEVEL beats later, but data block on earlier
6458 		 * objects beats the virtual marker on later objects.
6459 		 */
6460 		int cmp = TREE_CMP(zb1->zb_object, zb2->zb_object);
6461 		if (cmp != 0)
6462 			return (cmp);
6463 
6464 		if (zb1->zb_level == ZB_DNODE_LEVEL)
6465 			return (-1);
6466 		return (1);
6467 	}
6468 
6469 	IMPLY(zb1->zb_level > 0, ibs1 >= SPA_MINBLOCKSHIFT);
6470 	IMPLY(zb2->zb_level > 0, ibs2 >= SPA_MINBLOCKSHIFT);
6471 
6472 	/*
6473 	 * BP_SPANB calculates the span in blocks.
6474 	 */
6475 	zb1L0 = (zb1->zb_blkid) * BP_SPANB(ibs1, zb1->zb_level);
6476 	zb2L0 = (zb2->zb_blkid) * BP_SPANB(ibs2, zb2->zb_level);
6477 
6478 	if (zb1->zb_object == DMU_META_DNODE_OBJECT) {
6479 		zb1obj = zb1L0 * (dbss1 << (SPA_MINBLOCKSHIFT - DNODE_SHIFT));
6480 		zb1L0 = 0;
6481 		zb1level = zb1->zb_level + COMPARE_META_LEVEL;
6482 	} else {
6483 		zb1obj = zb1->zb_object;
6484 		zb1level = zb1->zb_level;
6485 	}
6486 
6487 	if (zb2->zb_object == DMU_META_DNODE_OBJECT) {
6488 		zb2obj = zb2L0 * (dbss2 << (SPA_MINBLOCKSHIFT - DNODE_SHIFT));
6489 		zb2L0 = 0;
6490 		zb2level = zb2->zb_level + COMPARE_META_LEVEL;
6491 	} else {
6492 		zb2obj = zb2->zb_object;
6493 		zb2level = zb2->zb_level;
6494 	}
6495 
6496 	/* Now that we have a canonical representation, do the comparison. */
6497 	if (zb1obj != zb2obj)
6498 		return (zb1obj < zb2obj ? -1 : 1);
6499 	else if (zb1L0 != zb2L0)
6500 		return (zb1L0 < zb2L0 ? -1 : 1);
6501 	else if (zb1level != zb2level)
6502 		return (zb1level > zb2level ? -1 : 1);
6503 	/*
6504 	 * This can (theoretically) happen if the bookmarks have the same object
6505 	 * and level, but different blkids, if the block sizes are not the same.
6506 	 * There is presently no way to change the indirect block sizes
6507 	 */
6508 	return (0);
6509 }
6510 
6511 /*
6512  *  This function checks the following: given that last_block is the place that
6513  *  our traversal stopped last time, does that guarantee that we've visited
6514  *  every node under subtree_root?  Therefore, we can't just use the raw output
6515  *  of zbookmark_compare.  We have to pass in a modified version of
6516  *  subtree_root; by incrementing the block id, and then checking whether
6517  *  last_block is before or equal to that, we can tell whether or not having
6518  *  visited last_block implies that all of subtree_root's children have been
6519  *  visited.
6520  */
6521 boolean_t
6522 zbookmark_subtree_completed(const dnode_phys_t *dnp,
6523     const zbookmark_phys_t *subtree_root, const zbookmark_phys_t *last_block)
6524 {
6525 	zbookmark_phys_t mod_zb = *subtree_root;
6526 	mod_zb.zb_blkid++;
6527 	ASSERT0(last_block->zb_level);
6528 
6529 	/* The objset_phys_t isn't before anything. */
6530 	if (dnp == NULL)
6531 		return (B_FALSE);
6532 
6533 	/*
6534 	 * We pass in 1ULL << (DNODE_BLOCK_SHIFT - SPA_MINBLOCKSHIFT) for the
6535 	 * data block size in sectors, because that variable is only used if
6536 	 * the bookmark refers to a block in the meta-dnode.  Since we don't
6537 	 * know without examining it what object it refers to, and there's no
6538 	 * harm in passing in this value in other cases, we always pass it in.
6539 	 *
6540 	 * We pass in 0 for the indirect block size shift because zb2 must be
6541 	 * level 0.  The indirect block size is only used to calculate the span
6542 	 * of the bookmark, but since the bookmark must be level 0, the span is
6543 	 * always 1, so the math works out.
6544 	 *
6545 	 * If you make changes to how the zbookmark_compare code works, be sure
6546 	 * to make sure that this code still works afterwards.
6547 	 */
6548 	return (zbookmark_compare(dnp->dn_datablkszsec, dnp->dn_indblkshift,
6549 	    1ULL << (DNODE_BLOCK_SHIFT - SPA_MINBLOCKSHIFT), 0, &mod_zb,
6550 	    last_block) <= 0);
6551 }
6552 
6553 /*
6554  * This function is similar to zbookmark_subtree_completed(), but returns true
6555  * if subtree_root is equal or ahead of last_block, i.e. still to be done.
6556  */
6557 boolean_t
6558 zbookmark_subtree_tbd(const dnode_phys_t *dnp,
6559     const zbookmark_phys_t *subtree_root, const zbookmark_phys_t *last_block)
6560 {
6561 	ASSERT0(last_block->zb_level);
6562 	if (dnp == NULL)
6563 		return (B_FALSE);
6564 	return (zbookmark_compare(dnp->dn_datablkszsec, dnp->dn_indblkshift,
6565 	    1ULL << (DNODE_BLOCK_SHIFT - SPA_MINBLOCKSHIFT), 0, subtree_root,
6566 	    last_block) >= 0);
6567 }
6568 
6569 EXPORT_SYMBOL(zio_type_name);
6570 EXPORT_SYMBOL(zio_buf_alloc);
6571 EXPORT_SYMBOL(zio_data_buf_alloc);
6572 EXPORT_SYMBOL(zio_buf_free);
6573 EXPORT_SYMBOL(zio_data_buf_free);
6574 
6575 ZFS_MODULE_PARAM(zfs_zio, zio_, slow_io_ms, INT, ZMOD_RW,
6576 	"Max I/O completion time (milliseconds) before marking it as slow");
6577 
6578 ZFS_MODULE_PARAM(zfs_zio, zio_, requeue_io_start_cut_in_line, INT, ZMOD_RW,
6579 	"Prioritize requeued I/O");
6580 
6581 ZFS_MODULE_PARAM(zfs_zio, zio_, batch_enabled, INT, ZMOD_RW,
6582 	"Batch processing of vdev children I/O completions");
6583 
6584 ZFS_MODULE_PARAM(zfs, zfs_, sync_pass_deferred_free,  UINT, ZMOD_RW,
6585 	"Defer frees starting in this pass");
6586 
6587 ZFS_MODULE_PARAM(zfs, zfs_, sync_pass_dont_compress, UINT, ZMOD_RW,
6588 	"Don't compress starting in this pass");
6589 
6590 ZFS_MODULE_PARAM(zfs, zfs_, sync_pass_rewrite, UINT, ZMOD_RW,
6591 	"Rewrite new bps starting in this pass");
6592 
6593 ZFS_MODULE_PARAM(zfs_zio, zio_, dva_throttle_enabled, INT, ZMOD_RW,
6594 	"Throttle block allocations in the ZIO pipeline");
6595 
6596 ZFS_MODULE_PARAM(zfs_zio, zio_, deadman_log_all, INT, ZMOD_RW,
6597 	"Log all slow ZIOs, not just those with vdevs");
6598