1 // SPDX-License-Identifier: GPL-2.0
2
3 #include <linux/blkdev.h>
4 #include <linux/fscrypt.h>
5 #include <linux/iversion.h>
6 #include "ctree.h"
7 #include "fs.h"
8 #include "messages.h"
9 #include "compression.h"
10 #include "delalloc-space.h"
11 #include "disk-io.h"
12 #include "reflink.h"
13 #include "transaction.h"
14 #include "subpage.h"
15 #include "accessors.h"
16 #include "file-item.h"
17 #include "file.h"
18 #include "super.h"
19
20 #define BTRFS_MAX_DEDUPE_LEN SZ_16M
21
clone_finish_inode_update(struct btrfs_trans_handle * trans,struct btrfs_inode * inode,u64 endoff,const u64 destoff,const u64 olen,bool no_time_update)22 static int clone_finish_inode_update(struct btrfs_trans_handle *trans,
23 struct btrfs_inode *inode,
24 u64 endoff,
25 const u64 destoff,
26 const u64 olen,
27 bool no_time_update)
28 {
29 struct inode *vfs_inode = &inode->vfs_inode;
30 int ret;
31
32 inode_inc_iversion(vfs_inode);
33 if (!no_time_update)
34 inode_set_mtime_to_ts(vfs_inode, inode_set_ctime_current(vfs_inode));
35
36 /*
37 * We round up to the block size at eof when determining which
38 * extents to clone above, but shouldn't round up the file size.
39 */
40 if (endoff > destoff + olen)
41 endoff = destoff + olen;
42 if (endoff > vfs_inode->i_size) {
43 i_size_write(vfs_inode, endoff);
44 btrfs_inode_safe_disk_i_size_write(inode, 0);
45 }
46
47 ret = btrfs_update_inode(trans, inode);
48 if (unlikely(ret)) {
49 btrfs_abort_transaction(trans, ret);
50 btrfs_end_transaction(trans);
51 return ret;
52 }
53 return btrfs_end_transaction(trans);
54 }
55
copy_inline_to_page(struct btrfs_inode * inode,const u64 file_offset,char * inline_data,const u64 size,const u64 datal,const u8 comp_type)56 static int copy_inline_to_page(struct btrfs_inode *inode,
57 const u64 file_offset,
58 char *inline_data,
59 const u64 size,
60 const u64 datal,
61 const u8 comp_type)
62 {
63 struct btrfs_fs_info *fs_info = inode->root->fs_info;
64 const u32 block_size = fs_info->sectorsize;
65 const u64 range_end = file_offset + block_size - 1;
66 const size_t inline_size = size - btrfs_file_extent_calc_inline_size(0);
67 char *data_start = inline_data + btrfs_file_extent_calc_inline_size(0);
68 struct extent_changeset *data_reserved = NULL;
69 struct folio *folio = NULL;
70 struct address_space *mapping = inode->vfs_inode.i_mapping;
71 int ret;
72
73 ASSERT(IS_ALIGNED(file_offset, block_size), "file_offset=%llu block_size=%u",
74 file_offset, block_size);
75
76 /*
77 * We have flushed and locked the ranges of the source and destination
78 * inodes, we also have locked the inodes, so we are safe to do a
79 * reservation here. Also we must not do the reservation while holding
80 * a transaction open, otherwise we would deadlock.
81 */
82 ret = btrfs_delalloc_reserve_space(inode, &data_reserved, file_offset,
83 block_size);
84 if (ret)
85 goto out;
86
87 folio = __filemap_get_folio(mapping, file_offset >> PAGE_SHIFT,
88 FGP_LOCK | FGP_ACCESSED | FGP_CREAT,
89 btrfs_alloc_write_mask(mapping));
90 if (IS_ERR(folio)) {
91 ret = PTR_ERR(folio);
92 goto out_unlock;
93 }
94
95 ret = set_folio_extent_mapped(folio);
96 if (ret < 0)
97 goto out_unlock;
98
99 ret = btrfs_reset_extent_delalloc(inode, file_offset, range_end, 0, NULL);
100 if (ret)
101 goto out_unlock;
102
103 /*
104 * After dirtying the page our caller will need to start a transaction,
105 * and if we are low on metadata free space, that can cause flushing of
106 * delalloc for all inodes in order to get metadata space released.
107 * However we are holding the range locked for the whole duration of
108 * the clone/dedupe operation, so we may deadlock if that happens and no
109 * other task releases enough space. So mark this inode as not being
110 * possible to flush to avoid such deadlock. We will clear that flag
111 * when we finish cloning all extents, since a transaction is started
112 * after finding each extent to clone.
113 */
114 set_bit(BTRFS_INODE_NO_DELALLOC_FLUSH, &inode->runtime_flags);
115
116 if (comp_type == BTRFS_COMPRESS_NONE) {
117 memcpy_to_folio(folio, offset_in_folio(folio, file_offset), data_start,
118 datal);
119 } else {
120 ret = btrfs_decompress(comp_type, data_start, folio,
121 offset_in_folio(folio, file_offset),
122 inline_size, datal);
123 if (ret)
124 goto out_unlock;
125 flush_dcache_folio(folio);
126 }
127
128 /*
129 * If our inline data is smaller then the block/page size, then the
130 * remaining of the block/page is equivalent to zeroes. We had something
131 * like the following done:
132 *
133 * $ xfs_io -f -c "pwrite -S 0xab 0 500" file
134 * $ sync # (or fsync)
135 * $ xfs_io -c "falloc 0 4K" file
136 * $ xfs_io -c "pwrite -S 0xcd 4K 4K"
137 *
138 * So what's in the range [500, 4095] corresponds to zeroes.
139 */
140 if (datal < block_size)
141 folio_zero_range(folio, datal, block_size - datal);
142
143 btrfs_folio_set_uptodate(fs_info, folio, file_offset, block_size);
144 btrfs_folio_set_dirty(fs_info, folio, file_offset, block_size);
145 out_unlock:
146 if (!IS_ERR(folio)) {
147 folio_unlock(folio);
148 folio_put(folio);
149 }
150 if (ret)
151 btrfs_delalloc_release_space(inode, data_reserved, file_offset,
152 block_size, true);
153 btrfs_delalloc_release_extents(inode, block_size);
154 out:
155 extent_changeset_free(data_reserved);
156
157 return ret;
158 }
159
160 /*
161 * Deal with cloning of inline extents. We try to copy the inline extent from
162 * the source inode to destination inode when possible. When not possible we
163 * copy the inline extent's data into the respective page of the inode.
164 */
clone_copy_inline_extent(struct btrfs_inode * inode,struct btrfs_path * path,struct btrfs_key * new_key,const u64 drop_start,const u64 datal,const u64 size,const u8 comp_type,char * inline_data,struct btrfs_trans_handle ** trans_out)165 static int clone_copy_inline_extent(struct btrfs_inode *inode,
166 struct btrfs_path *path,
167 struct btrfs_key *new_key,
168 const u64 drop_start,
169 const u64 datal,
170 const u64 size,
171 const u8 comp_type,
172 char *inline_data,
173 struct btrfs_trans_handle **trans_out)
174 {
175 struct btrfs_root *root = inode->root;
176 struct btrfs_fs_info *fs_info = root->fs_info;
177 const u64 aligned_end = ALIGN(new_key->offset + datal,
178 fs_info->sectorsize);
179 struct btrfs_trans_handle *trans = NULL;
180 struct btrfs_drop_extents_args drop_args = { 0 };
181 int ret;
182 struct btrfs_key key;
183 bool copied_inline_to_page = false;
184
185 if (new_key->offset > 0) {
186 ret = copy_inline_to_page(inode, new_key->offset,
187 inline_data, size, datal, comp_type);
188 copied_inline_to_page = (ret == 0);
189 goto out;
190 }
191
192 key.objectid = btrfs_ino(inode);
193 key.type = BTRFS_EXTENT_DATA_KEY;
194 key.offset = 0;
195 ret = btrfs_search_slot(NULL, root, &key, path, 0, 0);
196 if (ret < 0) {
197 return ret;
198 } else if (ret > 0) {
199 if (path->slots[0] >= btrfs_header_nritems(path->nodes[0])) {
200 ret = btrfs_next_leaf(root, path);
201 if (ret < 0)
202 return ret;
203 else if (ret > 0)
204 goto copy_inline_extent;
205 }
206 btrfs_item_key_to_cpu(path->nodes[0], &key, path->slots[0]);
207 if (key.objectid == btrfs_ino(inode) &&
208 key.type == BTRFS_EXTENT_DATA_KEY) {
209 /*
210 * There's an implicit hole at file offset 0, copy the
211 * inline extent's data to the page.
212 */
213 ASSERT(key.offset > 0);
214 goto copy_to_page;
215 }
216 } else if (i_size_read(&inode->vfs_inode) <= datal) {
217 struct btrfs_file_extent_item *ei;
218
219 ei = btrfs_item_ptr(path->nodes[0], path->slots[0],
220 struct btrfs_file_extent_item);
221 /*
222 * If it's an inline extent replace it with the source inline
223 * extent, otherwise copy the source inline extent data into
224 * the respective page at the destination inode.
225 */
226 if (btrfs_file_extent_type(path->nodes[0], ei) ==
227 BTRFS_FILE_EXTENT_INLINE)
228 goto copy_inline_extent;
229
230 goto copy_to_page;
231 }
232
233 copy_inline_extent:
234 /*
235 * We have no extent items, or we have an extent at offset 0 which may
236 * or may not be inlined. All these cases are dealt the same way.
237 */
238 if (i_size_read(&inode->vfs_inode) > datal) {
239 /*
240 * At the destination offset 0 we have either a hole, a regular
241 * extent or an inline extent larger then the one we want to
242 * clone. Deal with all these cases by copying the inline extent
243 * data into the respective page at the destination inode.
244 */
245 goto copy_to_page;
246 }
247
248 /*
249 * Release path before starting a new transaction so we don't hold locks
250 * that would confuse lockdep.
251 */
252 btrfs_release_path(path);
253 /*
254 * If we end up here it means were copy the inline extent into a leaf
255 * of the destination inode. We know we will drop or adjust at most one
256 * extent item in the destination root.
257 *
258 * 1 unit - adjusting old extent (we may have to split it)
259 * 1 unit - add new extent
260 * 1 unit - inode update
261 */
262 trans = btrfs_start_transaction(root, 3);
263 if (IS_ERR(trans)) {
264 ret = PTR_ERR(trans);
265 trans = NULL;
266 goto out;
267 }
268 drop_args.path = path;
269 drop_args.start = drop_start;
270 drop_args.end = aligned_end;
271 drop_args.drop_cache = true;
272 ret = btrfs_drop_extents(trans, root, inode, &drop_args);
273 if (unlikely(ret)) {
274 btrfs_abort_transaction(trans, ret);
275 goto out;
276 }
277 ret = btrfs_insert_empty_item(trans, root, path, new_key, size);
278 if (unlikely(ret)) {
279 btrfs_abort_transaction(trans, ret);
280 goto out;
281 }
282
283 write_extent_buffer(path->nodes[0], inline_data,
284 btrfs_item_ptr_offset(path->nodes[0],
285 path->slots[0]),
286 size);
287 btrfs_update_inode_bytes(inode, datal, drop_args.bytes_found);
288 btrfs_set_inode_full_sync(inode);
289 ret = btrfs_inode_set_file_extent_range(inode, 0, aligned_end);
290 if (unlikely(ret))
291 btrfs_abort_transaction(trans, ret);
292 out:
293 if (!ret && !trans) {
294 if (copied_inline_to_page &&
295 new_key->offset + datal > i_size_read(&inode->vfs_inode)) {
296 /*
297 * If we copied the inline extent data to a page/folio
298 * beyond the i_size of the destination inode, then we
299 * need to increase the i_size before we start a
300 * transaction to update the inode item. This is to
301 * prevent a deadlock when the flushoncommit mount
302 * option is used, which happens like this:
303 *
304 * 1) Task A clones an inline extent from inode X to an
305 * offset of inode Y that is beyond Y's current
306 * i_size. This means we copied the inline extent's
307 * data to a folio of inode Y that is beyond its EOF,
308 * using the call above to copy_inline_to_page();
309 *
310 * 2) Task B starts a transaction commit and calls
311 * btrfs_start_delalloc_flush() to flush delalloc;
312 *
313 * 3) The delalloc flushing sees the new dirty folio of
314 * inode Y and when it attempts to flush it, it ends
315 * up at extent_writepage() and sees that the offset
316 * of the folio is beyond the i_size of inode Y, so
317 * it attempts to invalidate the folio by calling
318 * folio_invalidate(), which ends up at btrfs' folio
319 * invalidate callback - btrfs_invalidate_folio().
320 * There it tries to lock the folio's range in inode
321 * Y's extent io tree, but it blocks since it's
322 * currently locked by task A - during reflink we
323 * lock the inodes and the source and destination
324 * ranges after flushing all delalloc and waiting for
325 * ordered extent completion - after that we don't
326 * expect to have dirty folios in the ranges, the
327 * exception is if we have to copy an inline extent's
328 * data (because the destination offset is not zero);
329 *
330 * 4) Task A then does the 'goto out' below and attempts
331 * to start a transaction to update the inode item,
332 * and then it's blocked since the current
333 * transaction is in the TRANS_STATE_COMMIT_START
334 * state. Therefore task A has to wait for the
335 * current transaction to become unblocked (its
336 * state >= TRANS_STATE_UNBLOCKED).
337 *
338 * This leads to a deadlock - the task committing the
339 * transaction waiting for the delalloc flushing which
340 * is blocked during folio invalidation on the inode's
341 * extent lock and the reflink task waiting for the
342 * current transaction to be unblocked so that it can
343 * start a new one to update the inode item (while
344 * holding the extent lock).
345 */
346 i_size_write(&inode->vfs_inode, new_key->offset + datal);
347 }
348 /*
349 * No transaction here means we copied the inline extent into a
350 * page of the destination inode.
351 *
352 * 1 unit to update inode item
353 */
354 trans = btrfs_start_transaction(root, 1);
355 if (IS_ERR(trans)) {
356 ret = PTR_ERR(trans);
357 trans = NULL;
358 }
359 }
360 if (ret && trans)
361 btrfs_end_transaction(trans);
362 if (!ret)
363 *trans_out = trans;
364
365 return ret;
366
367 copy_to_page:
368 /*
369 * Release our path because we don't need it anymore and also because
370 * copy_inline_to_page() needs to reserve data and metadata, which may
371 * need to flush delalloc when we are low on available space and
372 * therefore cause a deadlock if writeback of an inline extent needs to
373 * write to the same leaf or an ordered extent completion needs to write
374 * to the same leaf.
375 */
376 btrfs_release_path(path);
377
378 ret = copy_inline_to_page(inode, new_key->offset,
379 inline_data, size, datal, comp_type);
380 copied_inline_to_page = (ret == 0);
381
382 goto out;
383 }
384
385 /*
386 * Clone a range from inode file to another.
387 *
388 * @src: Inode to clone from
389 * @inode: Inode to clone to
390 * @off: Offset within source to start clone from
391 * @olen: Original length, passed by user, of range to clone
392 * @olen_aligned: Block-aligned value of olen
393 * @destoff: Offset within @inode to start clone
394 * @no_time_update: Whether to update mtime/ctime on the target inode
395 */
btrfs_clone(struct btrfs_inode * src,struct btrfs_inode * inode,const u64 off,const u64 olen,const u64 olen_aligned,const u64 destoff,bool no_time_update)396 static int btrfs_clone(struct btrfs_inode *src, struct btrfs_inode *inode,
397 const u64 off, const u64 olen, const u64 olen_aligned,
398 const u64 destoff, bool no_time_update)
399 {
400 struct btrfs_fs_info *fs_info = inode->root->fs_info;
401 BTRFS_PATH_AUTO_FREE(path);
402 struct extent_buffer *leaf;
403 struct btrfs_trans_handle *trans;
404 char AUTO_KVFREE(buf);
405 struct btrfs_key key;
406 u32 nritems;
407 int slot;
408 int ret;
409 const u64 len = olen_aligned;
410 u64 last_dest_end = destoff;
411 u64 prev_extent_end = off;
412
413 ret = -ENOMEM;
414 buf = kvmalloc(fs_info->nodesize, GFP_KERNEL);
415 if (!buf)
416 return ret;
417
418 path = btrfs_alloc_path();
419 if (!path)
420 return ret;
421
422 path->reada = READA_FORWARD;
423 /* Clone data */
424 key.objectid = btrfs_ino(src);
425 key.type = BTRFS_EXTENT_DATA_KEY;
426 key.offset = off;
427
428 while (1) {
429 struct btrfs_file_extent_item *extent;
430 u64 extent_gen;
431 int type;
432 u32 size;
433 struct btrfs_key new_key;
434 u64 disko = 0, diskl = 0;
435 u64 datao = 0, datal = 0;
436 u8 comp;
437 u64 drop_start;
438
439 /* Note the key will change type as we walk through the tree */
440 ret = btrfs_search_slot(NULL, src->root, &key, path, 0, 0);
441 if (ret < 0)
442 goto out;
443 /*
444 * First search, if no extent item that starts at offset off was
445 * found but the previous item is an extent item, it's possible
446 * it might overlap our target range, therefore process it.
447 */
448 if (key.offset == off && ret > 0 && path->slots[0] > 0) {
449 btrfs_item_key_to_cpu(path->nodes[0], &key,
450 path->slots[0] - 1);
451 if (key.type == BTRFS_EXTENT_DATA_KEY)
452 path->slots[0]--;
453 }
454
455 nritems = btrfs_header_nritems(path->nodes[0]);
456 process_slot:
457 if (path->slots[0] >= nritems) {
458 ret = btrfs_next_leaf(src->root, path);
459 if (ret < 0)
460 goto out;
461 if (ret > 0)
462 break;
463 nritems = btrfs_header_nritems(path->nodes[0]);
464 }
465 leaf = path->nodes[0];
466 slot = path->slots[0];
467
468 btrfs_item_key_to_cpu(leaf, &key, slot);
469 if (key.type > BTRFS_EXTENT_DATA_KEY || key.objectid != btrfs_ino(src))
470 break;
471
472 ASSERT(key.type == BTRFS_EXTENT_DATA_KEY, "key.type=%u", key.type);
473
474 extent = btrfs_item_ptr(leaf, slot,
475 struct btrfs_file_extent_item);
476 extent_gen = btrfs_file_extent_generation(leaf, extent);
477 comp = btrfs_file_extent_compression(leaf, extent);
478 type = btrfs_file_extent_type(leaf, extent);
479 if (type == BTRFS_FILE_EXTENT_REG ||
480 type == BTRFS_FILE_EXTENT_PREALLOC) {
481 disko = btrfs_file_extent_disk_bytenr(leaf, extent);
482 diskl = btrfs_file_extent_disk_num_bytes(leaf, extent);
483 datao = btrfs_file_extent_offset(leaf, extent);
484 datal = btrfs_file_extent_num_bytes(leaf, extent);
485 } else if (type == BTRFS_FILE_EXTENT_INLINE) {
486 /* Take upper bound, may be compressed */
487 datal = btrfs_file_extent_ram_bytes(leaf, extent);
488 }
489
490 /*
491 * The first search might have left us at an extent item that
492 * ends before our target range's start, can happen if we have
493 * holes and NO_HOLES feature enabled.
494 *
495 * Subsequent searches may leave us on a file range we have
496 * processed before - this happens due to a race with ordered
497 * extent completion for a file range that is outside our source
498 * range, but that range was part of a file extent item that
499 * also covered a leading part of our source range.
500 */
501 if (key.offset + datal <= prev_extent_end) {
502 path->slots[0]++;
503 goto process_slot;
504 } else if (key.offset >= off + len) {
505 break;
506 }
507
508 prev_extent_end = key.offset + datal;
509 size = btrfs_item_size(leaf, slot);
510 read_extent_buffer(leaf, buf, btrfs_item_ptr_offset(leaf, slot),
511 size);
512
513 btrfs_release_path(path);
514
515 memcpy(&new_key, &key, sizeof(new_key));
516 new_key.objectid = btrfs_ino(inode);
517 if (off <= key.offset)
518 new_key.offset = key.offset + destoff - off;
519 else
520 new_key.offset = destoff;
521
522 /*
523 * Deal with a hole that doesn't have an extent item that
524 * represents it (NO_HOLES feature enabled).
525 * This hole is either in the middle of the cloning range or at
526 * the beginning (fully overlaps it or partially overlaps it).
527 */
528 if (new_key.offset != last_dest_end)
529 drop_start = last_dest_end;
530 else
531 drop_start = new_key.offset;
532
533 if (type == BTRFS_FILE_EXTENT_REG ||
534 type == BTRFS_FILE_EXTENT_PREALLOC) {
535 struct btrfs_replace_extent_info clone_info;
536
537 /*
538 * a | --- range to clone ---| b
539 * | ------------- extent ------------- |
540 */
541
542 /* Subtract range b */
543 if (key.offset + datal > off + len)
544 datal = off + len - key.offset;
545
546 /* Subtract range a */
547 if (off > key.offset) {
548 datao += off - key.offset;
549 datal -= off - key.offset;
550 }
551
552 clone_info.disk_offset = disko;
553 clone_info.disk_len = diskl;
554 clone_info.data_offset = datao;
555 clone_info.data_len = datal;
556 clone_info.file_offset = new_key.offset;
557 clone_info.extent_buf = buf;
558 clone_info.is_new_extent = false;
559 clone_info.update_times = !no_time_update;
560 ret = btrfs_replace_file_extents(inode, path,
561 drop_start, new_key.offset + datal - 1,
562 &clone_info, &trans);
563 if (ret)
564 goto out;
565 } else {
566 ASSERT(type == BTRFS_FILE_EXTENT_INLINE);
567 /*
568 * Inline extents always have to start at file offset 0
569 * and can never be bigger then the sector size. We can
570 * never clone only parts of an inline extent, since all
571 * reflink operations must start at a sector size aligned
572 * offset, and the length must be aligned too or end at
573 * the i_size (which implies the whole inlined data).
574 */
575 ASSERT(key.offset == 0);
576 ASSERT(datal <= fs_info->sectorsize);
577 if (WARN_ON(type != BTRFS_FILE_EXTENT_INLINE) ||
578 WARN_ON(key.offset != 0) ||
579 WARN_ON(datal > fs_info->sectorsize)) {
580 ret = -EUCLEAN;
581 goto out;
582 }
583
584 ret = clone_copy_inline_extent(inode, path, &new_key,
585 drop_start, datal, size,
586 comp, buf, &trans);
587 if (ret)
588 goto out;
589 }
590
591 btrfs_release_path(path);
592
593 /*
594 * Whenever we share an extent we update the last_reflink_trans
595 * of each inode to the current transaction. This is needed to
596 * make sure fsync does not log multiple checksum items with
597 * overlapping ranges (because some extent items might refer
598 * only to sections of the original extent). For the destination
599 * inode we do this regardless of the generation of the extents
600 * or even if they are inline extents or explicit holes, to make
601 * sure a full fsync does not skip them. For the source inode,
602 * we only need to update last_reflink_trans in case it's a new
603 * extent that is not a hole or an inline extent, to deal with
604 * the checksums problem on fsync.
605 */
606 if (extent_gen == trans->transid && disko > 0)
607 src->last_reflink_trans = trans->transid;
608
609 inode->last_reflink_trans = trans->transid;
610
611 last_dest_end = ALIGN(new_key.offset + datal,
612 fs_info->sectorsize);
613 ret = clone_finish_inode_update(trans, inode, last_dest_end,
614 destoff, olen, no_time_update);
615 if (ret)
616 goto out;
617 if (new_key.offset + datal >= destoff + len)
618 break;
619
620 btrfs_release_path(path);
621 key.offset = prev_extent_end;
622
623 if (fatal_signal_pending(current)) {
624 ret = -EINTR;
625 goto out;
626 }
627
628 cond_resched();
629 }
630 ret = 0;
631
632 if (last_dest_end < destoff + len) {
633 /*
634 * We have an implicit hole that fully or partially overlaps our
635 * cloning range at its end. This means that we either have the
636 * NO_HOLES feature enabled or the implicit hole happened due to
637 * mixing buffered and direct IO writes against this file.
638 */
639 btrfs_release_path(path);
640
641 /*
642 * When using NO_HOLES and we are cloning a range that covers
643 * only a hole (no extents) into a range beyond the current
644 * i_size, punching a hole in the target range will not create
645 * an extent map defining a hole, because the range starts at or
646 * beyond current i_size. If the file previously had an i_size
647 * greater than the new i_size set by this clone operation, we
648 * need to make sure the next fsync is a full fsync, so that it
649 * detects and logs a hole covering a range from the current
650 * i_size to the new i_size. If the clone range covers extents,
651 * besides a hole, then we know the full sync flag was already
652 * set by previous calls to btrfs_replace_file_extents() that
653 * replaced file extent items.
654 */
655 if (last_dest_end >= i_size_read(&inode->vfs_inode))
656 btrfs_set_inode_full_sync(inode);
657
658 ret = btrfs_replace_file_extents(inode, path,
659 last_dest_end, destoff + len - 1, NULL, &trans);
660 if (ret)
661 goto out;
662
663 ret = clone_finish_inode_update(trans, inode, destoff + len,
664 destoff, olen, no_time_update);
665 }
666
667 out:
668 clear_bit(BTRFS_INODE_NO_DELALLOC_FLUSH, &inode->runtime_flags);
669
670 return ret;
671 }
672
btrfs_double_mmap_lock(struct btrfs_inode * inode1,struct btrfs_inode * inode2)673 static void btrfs_double_mmap_lock(struct btrfs_inode *inode1, struct btrfs_inode *inode2)
674 {
675 if (inode1 < inode2)
676 swap(inode1, inode2);
677 down_write(&inode1->i_mmap_lock);
678 down_write_nested(&inode2->i_mmap_lock, SINGLE_DEPTH_NESTING);
679 }
680
btrfs_double_mmap_unlock(struct btrfs_inode * inode1,struct btrfs_inode * inode2)681 static void btrfs_double_mmap_unlock(struct btrfs_inode *inode1, struct btrfs_inode *inode2)
682 {
683 up_write(&inode1->i_mmap_lock);
684 up_write(&inode2->i_mmap_lock);
685 }
686
btrfs_extent_same_range(struct btrfs_inode * src,u64 loff,u64 len,struct btrfs_inode * dst,u64 dst_loff)687 static int btrfs_extent_same_range(struct btrfs_inode *src, u64 loff, u64 len,
688 struct btrfs_inode *dst, u64 dst_loff)
689 {
690 struct extent_state *cached_state = NULL;
691 struct btrfs_fs_info *fs_info = src->root->fs_info;
692 const u32 bs = fs_info->sectorsize;
693 const u64 end = round_up(dst_loff + len, bs) - 1;
694 int ret;
695
696 /*
697 * Lock destination range to serialize with concurrent readahead(), and
698 * we are safe from concurrency with relocation of source extents
699 * because we have already locked the inode's i_mmap_lock in exclusive
700 * mode.
701 */
702 btrfs_lock_extent(&dst->io_tree, dst_loff, end, &cached_state);
703 ret = btrfs_clone(src, dst, loff, len, ALIGN(len, bs), dst_loff, true);
704 btrfs_unlock_extent(&dst->io_tree, dst_loff, end, &cached_state);
705
706 btrfs_btree_balance_dirty(fs_info);
707
708 return ret;
709 }
710
btrfs_extent_same(struct btrfs_inode * src,u64 loff,u64 olen,struct btrfs_inode * dst,u64 dst_loff)711 static int btrfs_extent_same(struct btrfs_inode *src, u64 loff, u64 olen,
712 struct btrfs_inode *dst, u64 dst_loff)
713 {
714 int ret = 0;
715 u64 i, tail_len, chunk_count;
716 struct btrfs_root *root_dst = dst->root;
717
718 spin_lock(&root_dst->root_item_lock);
719 if (root_dst->send_in_progress) {
720 btrfs_warn_rl(root_dst->fs_info,
721 "cannot deduplicate to root %llu while send operations are using it (%d in progress)",
722 btrfs_root_id(root_dst),
723 root_dst->send_in_progress);
724 spin_unlock(&root_dst->root_item_lock);
725 return -EAGAIN;
726 }
727 root_dst->dedupe_in_progress++;
728 spin_unlock(&root_dst->root_item_lock);
729
730 tail_len = olen % BTRFS_MAX_DEDUPE_LEN;
731 chunk_count = div_u64(olen, BTRFS_MAX_DEDUPE_LEN);
732
733 for (i = 0; i < chunk_count; i++) {
734 ret = btrfs_extent_same_range(src, loff, BTRFS_MAX_DEDUPE_LEN,
735 dst, dst_loff);
736 if (ret)
737 goto out;
738
739 loff += BTRFS_MAX_DEDUPE_LEN;
740 dst_loff += BTRFS_MAX_DEDUPE_LEN;
741 }
742
743 if (tail_len > 0)
744 ret = btrfs_extent_same_range(src, loff, tail_len, dst, dst_loff);
745 out:
746 spin_lock(&root_dst->root_item_lock);
747 root_dst->dedupe_in_progress--;
748 spin_unlock(&root_dst->root_item_lock);
749
750 return ret;
751 }
752
btrfs_clone_files(struct file * file,struct file * file_src,u64 off,u64 olen,u64 destoff)753 static noinline int btrfs_clone_files(struct file *file, struct file *file_src,
754 u64 off, u64 olen, u64 destoff)
755 {
756 struct extent_state *cached_state = NULL;
757 struct btrfs_inode *inode = BTRFS_I(file_inode(file));
758 struct btrfs_inode *src = BTRFS_I(file_inode(file_src));
759 struct btrfs_fs_info *fs_info = inode->root->fs_info;
760 const u64 src_isize = src->vfs_inode.i_size;
761 const u64 inode_isize = inode->vfs_inode.i_size;
762 int ret;
763 u64 len = olen;
764 const u32 bs = fs_info->sectorsize;
765 u64 end;
766
767 /*
768 * VFS's generic_remap_file_range_prep() protects us from cloning the
769 * eof block into the middle of a file, which would result in corruption
770 * if the file size is not blocksize aligned. So we don't need to check
771 * for that case here.
772 */
773 if (off + len == src_isize)
774 len = ALIGN(src_isize, bs) - off;
775
776 if (destoff > inode_isize) {
777 const u64 wb_start = ALIGN_DOWN(inode_isize, bs);
778
779 ret = btrfs_cont_expand(inode, inode_isize, destoff);
780 if (ret)
781 return ret;
782 /*
783 * We may have truncated the last block if the inode's size is
784 * not sector size aligned, so we need to wait for writeback to
785 * complete before proceeding further, otherwise we can race
786 * with cloning and attempt to increment a reference to an
787 * extent that no longer exists (writeback completed right after
788 * we found the previous extent covering eof and before we
789 * attempted to increment its reference count).
790 */
791 ret = btrfs_wait_ordered_range(inode, wb_start, destoff - wb_start);
792 if (ret)
793 return ret;
794 }
795
796 /*
797 * Lock destination range to serialize with concurrent readahead(), and
798 * we are safe from concurrency with relocation of source extents
799 * because we have already locked the inode's i_mmap_lock in exclusive
800 * mode.
801 */
802 end = round_up(destoff + len, bs) - 1;
803 btrfs_lock_extent(&inode->io_tree, destoff, end, &cached_state);
804 ret = btrfs_clone(src, inode, off, olen, len, destoff, false);
805 btrfs_unlock_extent(&inode->io_tree, destoff, end, &cached_state);
806 if (ret < 0)
807 return ret;
808
809 /*
810 * We may have copied an inline extent into a page of the destination
811 * range. So flush delalloc and wait for ordered extent completion.
812 * This is to ensure the invalidation below does not fail, as if for
813 * example it finds a dirty folio, our folio release callback
814 * (btrfs_release_folio()) returns false, which makes the invalidation
815 * return an -EBUSY error. We can't ignore such failures since they
816 * could come from some range other than the copied inline extent's
817 * destination range and we have no way to know that.
818 */
819 ret = btrfs_wait_ordered_range(inode, destoff, len);
820 if (ret < 0)
821 return ret;
822
823 /*
824 * Invalidate page cache so that future reads will see the cloned data
825 * immediately and not the previous data.
826 */
827 ret = filemap_invalidate_inode(&inode->vfs_inode, false, destoff, end);
828 if (ret < 0)
829 return ret;
830
831 btrfs_btree_balance_dirty(fs_info);
832
833 return 0;
834 }
835
btrfs_remap_file_range_prep(struct file * file_in,loff_t pos_in,struct file * file_out,loff_t pos_out,loff_t * len,unsigned int remap_flags)836 static int btrfs_remap_file_range_prep(struct file *file_in, loff_t pos_in,
837 struct file *file_out, loff_t pos_out,
838 loff_t *len, unsigned int remap_flags)
839 {
840 struct btrfs_inode *inode_in = BTRFS_I(file_inode(file_in));
841 struct btrfs_inode *inode_out = BTRFS_I(file_inode(file_out));
842 const u32 bs = inode_out->root->fs_info->sectorsize;
843 u64 wb_len;
844 int ret;
845
846 if (!(remap_flags & REMAP_FILE_DEDUP)) {
847 struct btrfs_root *root_out = inode_out->root;
848
849 if (btrfs_root_readonly(root_out))
850 return -EROFS;
851
852 ASSERT(inode_in->vfs_inode.i_sb == inode_out->vfs_inode.i_sb);
853 }
854
855 /* Can only reflink encrypted files if both files are encrypted. */
856 if (IS_ENCRYPTED(&inode_in->vfs_inode) != IS_ENCRYPTED(&inode_out->vfs_inode))
857 return -EINVAL;
858
859 /* Don't make the dst file partly checksummed */
860 if ((inode_in->flags & BTRFS_INODE_NODATASUM) !=
861 (inode_out->flags & BTRFS_INODE_NODATASUM)) {
862 return -EINVAL;
863 }
864
865 /*
866 * Now that the inodes are locked, we need to start writeback ourselves
867 * and can not rely on the writeback from the VFS's generic helper
868 * generic_remap_file_range_prep() because:
869 *
870 * 1) For compression we must call filemap_fdatawrite_range() range
871 * twice (btrfs_fdatawrite_range() does it for us), and the generic
872 * helper only calls it once;
873 *
874 * 2) filemap_fdatawrite_range(), called by the generic helper only
875 * waits for the writeback to complete, i.e. for IO to be done, and
876 * not for the ordered extents to complete. We need to wait for them
877 * to complete so that new file extent items are in the fs tree.
878 */
879 if (*len == 0 && !(remap_flags & REMAP_FILE_DEDUP))
880 wb_len = ALIGN(inode_in->vfs_inode.i_size, bs) - ALIGN_DOWN(pos_in, bs);
881 else
882 wb_len = ALIGN(*len, bs);
883
884 /*
885 * Workaround to make sure NOCOW buffered write reach disk as NOCOW.
886 *
887 * Btrfs' back references do not have a block level granularity, they
888 * work at the whole extent level.
889 * NOCOW buffered write without data space reserved may not be able
890 * to fall back to CoW due to lack of data space, thus could cause
891 * data loss.
892 *
893 * Here we take a shortcut by flushing the whole inode, so that all
894 * nocow write should reach disk as nocow before we increase the
895 * reference of the extent. We could do better by only flushing NOCOW
896 * data, but that needs extra accounting.
897 *
898 * Also we don't need to check ASYNC_EXTENT, as async extent will be
899 * CoWed anyway, not affecting nocow part.
900 */
901 ret = filemap_flush(inode_in->vfs_inode.i_mapping);
902 if (ret < 0)
903 return ret;
904
905 ret = btrfs_wait_ordered_range(inode_in, ALIGN_DOWN(pos_in, bs), wb_len);
906 if (ret < 0)
907 return ret;
908 ret = btrfs_wait_ordered_range(inode_out, ALIGN_DOWN(pos_out, bs), wb_len);
909 if (ret < 0)
910 return ret;
911
912 return generic_remap_file_range_prep(file_in, pos_in, file_out, pos_out,
913 len, remap_flags);
914 }
915
file_sync_write(const struct file * file)916 static bool file_sync_write(const struct file *file)
917 {
918 if (file->f_flags & (__O_SYNC | O_DSYNC))
919 return true;
920 if (IS_SYNC(file_inode(file)))
921 return true;
922
923 return false;
924 }
925
btrfs_remap_file_range(struct file * src_file,loff_t off,struct file * dst_file,loff_t destoff,loff_t len,unsigned int remap_flags)926 loff_t btrfs_remap_file_range(struct file *src_file, loff_t off,
927 struct file *dst_file, loff_t destoff, loff_t len,
928 unsigned int remap_flags)
929 {
930 struct btrfs_inode *src_inode = BTRFS_I(file_inode(src_file));
931 struct btrfs_inode *dst_inode = BTRFS_I(file_inode(dst_file));
932 bool same_inode = dst_inode == src_inode;
933 int ret;
934
935 if (btrfs_is_shutdown(src_inode->root->fs_info))
936 return -EIO;
937
938 if (remap_flags & ~(REMAP_FILE_DEDUP | REMAP_FILE_ADVISORY))
939 return -EINVAL;
940
941 if (same_inode) {
942 btrfs_inode_lock(src_inode, BTRFS_ILOCK_MMAP);
943 } else {
944 lock_two_nondirectories(&src_inode->vfs_inode, &dst_inode->vfs_inode);
945 btrfs_double_mmap_lock(src_inode, dst_inode);
946 }
947
948 ret = btrfs_remap_file_range_prep(src_file, off, dst_file, destoff,
949 &len, remap_flags);
950 if (ret < 0 || len == 0)
951 goto out_unlock;
952
953 if (remap_flags & REMAP_FILE_DEDUP)
954 ret = btrfs_extent_same(src_inode, off, len, dst_inode, destoff);
955 else
956 ret = btrfs_clone_files(dst_file, src_file, off, len, destoff);
957
958 out_unlock:
959 if (same_inode) {
960 btrfs_inode_unlock(src_inode, BTRFS_ILOCK_MMAP);
961 } else {
962 btrfs_double_mmap_unlock(src_inode, dst_inode);
963 unlock_two_nondirectories(&src_inode->vfs_inode,
964 &dst_inode->vfs_inode);
965 }
966
967 /*
968 * If either the source or the destination file was opened with O_SYNC,
969 * O_DSYNC or has the S_SYNC attribute, fsync both the destination and
970 * source files/ranges, so that after a successful return (0) followed
971 * by a power failure results in the reflinked data to be readable from
972 * both files/ranges.
973 */
974 if (ret == 0 && len > 0 &&
975 (file_sync_write(src_file) || file_sync_write(dst_file))) {
976 ret = btrfs_sync_file(src_file, off, off + len - 1, 0);
977 if (ret == 0)
978 ret = btrfs_sync_file(dst_file, destoff,
979 destoff + len - 1, 0);
980 }
981
982 return ret < 0 ? ret : len;
983 }
984