1 // SPDX-License-Identifier: GPL-2.0
2 /*
3 * fs/f2fs/data.c
4 *
5 * Copyright (c) 2012 Samsung Electronics Co., Ltd.
6 * http://www.samsung.com/
7 */
8 #include <linux/fs.h>
9 #include <linux/f2fs_fs.h>
10 #include <linux/sched/mm.h>
11 #include <linux/mpage.h>
12 #include <linux/writeback.h>
13 #include <linux/folio_batch.h>
14 #include <linux/blkdev.h>
15 #include <linux/bio.h>
16 #include <linux/blk-crypto.h>
17 #include <linux/swap.h>
18 #include <linux/prefetch.h>
19 #include <linux/uio.h>
20 #include <linux/sched/signal.h>
21 #include <linux/fiemap.h>
22 #include <linux/iomap.h>
23 #include <linux/fserror.h>
24
25 #include "f2fs.h"
26 #include "node.h"
27 #include "segment.h"
28 #include "iostat.h"
29 #include <trace/events/f2fs.h>
30
31 #define NUM_PREALLOC_POST_READ_CTXS 128
32
33 static struct kmem_cache *bio_post_read_ctx_cache;
34 static struct kmem_cache *bio_entry_slab;
35 static struct kmem_cache *ffs_entry_slab;
36 static mempool_t *bio_post_read_ctx_pool;
37 static struct bio_set f2fs_bioset;
38
39 struct f2fs_folio_state {
40 spinlock_t state_lock;
41 unsigned int read_pages_pending;
42 };
43
44 struct f2fs_bio {
45 struct work_struct work;
46 struct bio bio;
47 };
48
49 #define F2FS_BIO_POOL_SIZE NR_CURSEG_TYPE
50
f2fs_init_bioset(void)51 int __init f2fs_init_bioset(void)
52 {
53 return bioset_init(&f2fs_bioset, F2FS_BIO_POOL_SIZE,
54 offsetof(struct f2fs_bio, bio), BIOSET_NEED_BVECS);
55 }
56
f2fs_destroy_bioset(void)57 void f2fs_destroy_bioset(void)
58 {
59 bioset_exit(&f2fs_bioset);
60 }
61
f2fs_is_cp_guaranteed(const struct folio * folio)62 bool f2fs_is_cp_guaranteed(const struct folio *folio)
63 {
64 struct address_space *mapping = folio->mapping;
65 struct inode *inode;
66 struct f2fs_sb_info *sbi;
67
68 inode = mapping->host;
69 sbi = F2FS_I_SB(inode);
70
71 if (inode->i_ino == F2FS_META_INO(sbi) ||
72 inode->i_ino == F2FS_NODE_INO(sbi) ||
73 S_ISDIR(inode->i_mode))
74 return true;
75
76 if ((S_ISREG(inode->i_mode) && IS_NOQUOTA(inode)) ||
77 folio_test_f2fs_gcing(folio))
78 return true;
79 return false;
80 }
81
__read_io_type(struct folio * folio)82 static enum count_type __read_io_type(struct folio *folio)
83 {
84 struct address_space *mapping = folio->mapping;
85
86 if (mapping) {
87 struct inode *inode = mapping->host;
88 struct f2fs_sb_info *sbi = F2FS_I_SB(inode);
89
90 if (inode->i_ino == F2FS_META_INO(sbi))
91 return F2FS_RD_META;
92
93 if (inode->i_ino == F2FS_NODE_INO(sbi))
94 return F2FS_RD_NODE;
95 }
96 return F2FS_RD_DATA;
97 }
98
99 /* postprocessing steps for read bios */
100 enum bio_post_read_step {
101 #ifdef CONFIG_F2FS_FS_COMPRESSION
102 STEP_DECOMPRESS = BIT(1),
103 #else
104 STEP_DECOMPRESS = 0, /* compile out the decompression-related code */
105 #endif
106 #ifdef CONFIG_FS_VERITY
107 STEP_VERITY = BIT(2),
108 #else
109 STEP_VERITY = 0, /* compile out the verity-related code */
110 #endif
111 };
112
113 struct bio_post_read_ctx {
114 struct bio *bio;
115 struct f2fs_sb_info *sbi;
116 struct fsverity_info *vi;
117 struct work_struct work;
118 unsigned int enabled_steps;
119 /*
120 * decompression_attempted keeps track of whether
121 * f2fs_end_read_compressed_page() has been called on the pages in the
122 * bio that belong to a compressed cluster yet.
123 */
124 bool decompression_attempted;
125 block_t fs_blkaddr;
126 };
127
128 /*
129 * Update and unlock a bio's pages, and free the bio.
130 *
131 * This marks pages up-to-date only if there was no error in the bio (I/O error,
132 * decryption error, or verity error), as indicated by bio->bi_status.
133 *
134 * "Compressed pages" (pagecache pages backed by a compressed cluster on-disk)
135 * aren't marked up-to-date here, as decompression is done on a per-compression-
136 * cluster basis rather than a per-bio basis. Instead, we only must do two
137 * things for each compressed page here: call f2fs_end_read_compressed_page()
138 * with failed=true if an error occurred before it would have normally gotten
139 * called (i.e., I/O error or decryption error, but *not* verity error), and
140 * release the bio's reference to the decompress_io_ctx of the page's cluster.
141 */
f2fs_finish_read_bio(struct bio * bio,bool in_task)142 static void f2fs_finish_read_bio(struct bio *bio, bool in_task)
143 {
144 struct folio_iter fi;
145 struct bio_post_read_ctx *ctx = bio->bi_private;
146 unsigned long flags;
147
148 bio_for_each_folio_all(fi, bio) {
149 struct folio *folio = fi.folio;
150 unsigned nr_pages = fi.length >> PAGE_SHIFT;
151 bool finished = true;
152
153 if (!folio_test_large(folio) &&
154 f2fs_is_compressed_page(folio)) {
155 if (ctx && !ctx->decompression_attempted)
156 f2fs_end_read_compressed_page(folio, true, 0,
157 in_task);
158 f2fs_put_folio_dic(folio, in_task);
159 continue;
160 }
161
162 if (folio_test_large(folio)) {
163 struct f2fs_folio_state *ffs = folio->private;
164
165 spin_lock_irqsave(&ffs->state_lock, flags);
166 ffs->read_pages_pending -= nr_pages;
167 finished = !ffs->read_pages_pending;
168 spin_unlock_irqrestore(&ffs->state_lock, flags);
169 }
170
171 while (nr_pages--)
172 dec_page_count(F2FS_F_SB(folio), __read_io_type(folio));
173
174 if (bio->bi_status == BLK_STS_OK &&
175 F2FS_F_SB(folio)->node_inode && is_node_folio(folio) &&
176 f2fs_sanity_check_node_footer(F2FS_F_SB(folio),
177 folio, folio->index, NODE_TYPE_REGULAR, true))
178 bio->bi_status = BLK_STS_IOERR;
179
180 if (finished)
181 folio_end_read(folio, bio->bi_status == BLK_STS_OK);
182 }
183
184 if (ctx)
185 mempool_free(ctx, bio_post_read_ctx_pool);
186 bio_put(bio);
187 }
188
f2fs_verify_bio(struct work_struct * work)189 static void f2fs_verify_bio(struct work_struct *work)
190 {
191 struct bio_post_read_ctx *ctx =
192 container_of(work, struct bio_post_read_ctx, work);
193 struct bio *bio = ctx->bio;
194 bool may_have_compressed_pages = (ctx->enabled_steps & STEP_DECOMPRESS);
195 struct fsverity_info *vi = ctx->vi;
196
197 /*
198 * fsverity_verify_bio() may call readahead() again, and while verity
199 * will be disabled for this, decryption and/or decompression may still
200 * be needed, resulting in another bio_post_read_ctx being allocated.
201 * So to prevent deadlocks we need to release the current ctx to the
202 * mempool first. This assumes that verity is the last post-read step.
203 */
204 mempool_free(ctx, bio_post_read_ctx_pool);
205 bio->bi_private = NULL;
206
207 /*
208 * Verify the bio's pages with fs-verity. Exclude compressed pages,
209 * as those were handled separately by f2fs_end_read_compressed_page().
210 */
211 if (may_have_compressed_pages) {
212 struct folio_iter fi;
213
214 bio_for_each_folio_all(fi, bio) {
215 struct folio *folio = fi.folio;
216
217 if (!f2fs_is_compressed_page(folio) &&
218 !fsverity_verify_folio(vi, folio)) {
219 bio->bi_status = BLK_STS_IOERR;
220 break;
221 }
222 }
223 } else {
224 fsverity_verify_bio(vi, bio);
225 }
226
227 f2fs_finish_read_bio(bio, true);
228 }
229
230 /*
231 * If the bio's data needs to be verified with fs-verity, then enqueue the
232 * verity work for the bio. Otherwise finish the bio now.
233 *
234 * Note that to avoid deadlocks, the verity work can't be done on the
235 * decryption/decompression workqueue. This is because verifying the data pages
236 * can involve reading verity metadata pages from the file, and these verity
237 * metadata pages may be encrypted and/or compressed.
238 */
f2fs_verify_and_finish_bio(struct bio * bio,bool in_task)239 static void f2fs_verify_and_finish_bio(struct bio *bio, bool in_task)
240 {
241 struct bio_post_read_ctx *ctx = bio->bi_private;
242
243 if (ctx && (ctx->enabled_steps & STEP_VERITY)) {
244 INIT_WORK(&ctx->work, f2fs_verify_bio);
245 fsverity_enqueue_verify_work(&ctx->work);
246 } else {
247 f2fs_finish_read_bio(bio, in_task);
248 }
249 }
250
251 /*
252 * Handle STEP_DECOMPRESS by decompressing any compressed clusters whose last
253 * remaining page was read by @ctx->bio.
254 *
255 * Note that a bio may span clusters (even a mix of compressed and uncompressed
256 * clusters) or be for just part of a cluster. STEP_DECOMPRESS just indicates
257 * that the bio includes at least one compressed page. The actual decompression
258 * is done on a per-cluster basis, not a per-bio basis.
259 */
f2fs_handle_step_decompress(struct bio_post_read_ctx * ctx,bool in_task)260 static void f2fs_handle_step_decompress(struct bio_post_read_ctx *ctx,
261 bool in_task)
262 {
263 struct folio_iter fi;
264 bool all_compressed = true;
265 block_t blkaddr = ctx->fs_blkaddr;
266
267 bio_for_each_folio_all(fi, ctx->bio) {
268 struct folio *folio = fi.folio;
269
270 if (f2fs_is_compressed_page(folio))
271 f2fs_end_read_compressed_page(folio, false, blkaddr,
272 in_task);
273 else
274 all_compressed = false;
275
276 blkaddr++;
277 }
278
279 ctx->decompression_attempted = true;
280
281 /*
282 * Optimization: if all the bio's pages are compressed, then scheduling
283 * the per-bio verity work is unnecessary, as verity will be fully
284 * handled at the compression cluster level.
285 */
286 if (all_compressed)
287 ctx->enabled_steps &= ~STEP_VERITY;
288 }
289
f2fs_post_read_work(struct work_struct * work)290 static void f2fs_post_read_work(struct work_struct *work)
291 {
292 struct bio_post_read_ctx *ctx =
293 container_of(work, struct bio_post_read_ctx, work);
294 struct bio *bio = ctx->bio;
295
296 if (ctx->enabled_steps & STEP_DECOMPRESS)
297 f2fs_handle_step_decompress(ctx, true);
298
299 f2fs_verify_and_finish_bio(bio, true);
300 }
301
f2fs_read_end_io(struct bio * bio)302 static void f2fs_read_end_io(struct bio *bio)
303 {
304 struct f2fs_sb_info *sbi = F2FS_F_SB(bio_first_folio_all(bio));
305 struct bio_post_read_ctx *ctx;
306 bool intask = in_task() && !irqs_disabled();
307
308 iostat_update_and_unbind_ctx(bio);
309 ctx = bio->bi_private;
310
311 if (time_to_inject(sbi, FAULT_READ_IO))
312 bio->bi_status = BLK_STS_IOERR;
313
314 if (bio->bi_status != BLK_STS_OK) {
315 f2fs_finish_read_bio(bio, intask);
316 return;
317 }
318
319 if (ctx && (ctx->enabled_steps & STEP_DECOMPRESS)) {
320 if (!f2fs_low_mem_mode(sbi)) {
321 /* Decompress inline. */
322 f2fs_handle_step_decompress(ctx, intask);
323 } else {
324 INIT_WORK(&ctx->work, f2fs_post_read_work);
325 queue_work(ctx->sbi->wq, &ctx->work);
326 return;
327 }
328 }
329
330 f2fs_verify_and_finish_bio(bio, intask);
331 }
332
f2fs_write_end_bio(struct bio * bio)333 static void f2fs_write_end_bio(struct bio *bio)
334 {
335 struct f2fs_sb_info *sbi = bio->bi_private;
336 struct folio_iter fi;
337
338 if (time_to_inject(sbi, FAULT_WRITE_IO))
339 bio->bi_status = BLK_STS_IOERR;
340
341 bio_for_each_folio_all(fi, bio) {
342 struct folio *folio = fi.folio;
343 enum count_type type;
344
345 #ifdef CONFIG_F2FS_FS_COMPRESSION
346 if (f2fs_is_compressed_page(folio)) {
347 f2fs_compress_write_end_io(bio, folio);
348 continue;
349 }
350 #endif
351
352 type = WB_DATA_TYPE(folio, false);
353
354 if (unlikely(bio->bi_status != BLK_STS_OK)) {
355 mapping_set_error(folio->mapping, -EIO);
356 if (type == F2FS_WB_CP_DATA) {
357 f2fs_stop_checkpoint(sbi, true,
358 STOP_CP_REASON_WRITE_FAIL);
359 }
360 }
361
362 if (is_node_folio(folio)) {
363 f2fs_sanity_check_node_footer(sbi, folio,
364 folio->index, NODE_TYPE_REGULAR, true);
365 f2fs_bug_on(sbi, folio->index != nid_of_node(folio));
366 }
367 if (f2fs_in_warm_node_list(folio))
368 f2fs_del_fsync_node_entry(sbi, folio);
369
370 dec_page_count(sbi, type);
371
372 /*
373 * we should access sbi before folio_end_writeback() to
374 * avoid racing w/ kill_f2fs_super()
375 */
376 if (type == F2FS_WB_CP_DATA && !get_pages(sbi, type) &&
377 wq_has_sleeper(&sbi->cp_wait))
378 wake_up(&sbi->cp_wait);
379
380 folio_clear_f2fs_gcing(folio);
381 folio_end_writeback(folio);
382 }
383
384 bio_put(bio);
385 }
386
f2fs_write_end_io_work(struct work_struct * work)387 static void f2fs_write_end_io_work(struct work_struct *work)
388 {
389 struct bio *bio = &container_of(work, struct f2fs_bio, work)->bio;
390
391 f2fs_write_end_bio(bio);
392 }
393
f2fs_write_end_io(struct bio * bio)394 static void f2fs_write_end_io(struct bio *bio)
395 {
396 struct f2fs_sb_info *sbi;
397
398 iostat_update_and_unbind_ctx(bio);
399
400 sbi = bio->bi_private;
401
402 if (in_atomic() && bio->bi_iter.bi_size > sbi->max_atc_write_bio_size) {
403 struct work_struct *w;
404
405 w = &container_of(bio, struct f2fs_bio, bio)->work;
406 INIT_WORK(w, f2fs_write_end_io_work);
407 queue_work(sbi->wq, w);
408 } else {
409 f2fs_write_end_bio(bio);
410 }
411 }
412
413 #ifdef CONFIG_BLK_DEV_ZONED
f2fs_zone_write_end_io(struct bio * bio)414 static void f2fs_zone_write_end_io(struct bio *bio)
415 {
416 struct f2fs_bio_info *io = (struct f2fs_bio_info *)bio->bi_private;
417
418 bio->bi_private = io->bi_private;
419 complete(&io->zone_wait);
420 f2fs_write_end_io(bio);
421 }
422 #endif
423
f2fs_target_device(struct f2fs_sb_info * sbi,block_t blk_addr,sector_t * sector)424 struct block_device *f2fs_target_device(struct f2fs_sb_info *sbi,
425 block_t blk_addr, sector_t *sector)
426 {
427 struct block_device *bdev = sbi->sb->s_bdev;
428 int i;
429
430 if (f2fs_is_multi_device(sbi)) {
431 for (i = 0; i < sbi->s_ndevs; i++) {
432 if (FDEV(i).start_blk <= blk_addr &&
433 FDEV(i).end_blk >= blk_addr) {
434 blk_addr -= FDEV(i).start_blk;
435 bdev = FDEV(i).bdev;
436 break;
437 }
438 }
439 }
440
441 if (sector)
442 *sector = SECTOR_FROM_BLOCK(blk_addr);
443 return bdev;
444 }
445
f2fs_target_device_index(struct f2fs_sb_info * sbi,block_t blkaddr)446 int f2fs_target_device_index(struct f2fs_sb_info *sbi, block_t blkaddr)
447 {
448 int i;
449
450 if (!f2fs_is_multi_device(sbi))
451 return 0;
452
453 for (i = 0; i < sbi->s_ndevs; i++)
454 if (FDEV(i).start_blk <= blkaddr && FDEV(i).end_blk >= blkaddr)
455 return i;
456 return 0;
457 }
458
f2fs_io_flags(struct f2fs_io_info * fio)459 static blk_opf_t f2fs_io_flags(struct f2fs_io_info *fio)
460 {
461 unsigned int temp_mask = GENMASK(NR_TEMP_TYPE - 1, 0);
462 unsigned int fua_flag, meta_flag, io_flag;
463 blk_opf_t op_flags = 0;
464
465 if (fio->op != REQ_OP_WRITE)
466 return 0;
467 if (fio->type == DATA)
468 io_flag = fio->sbi->data_io_flag;
469 else if (fio->type == NODE)
470 io_flag = fio->sbi->node_io_flag;
471 else
472 return 0;
473
474 fua_flag = io_flag & temp_mask;
475 meta_flag = (io_flag >> NR_TEMP_TYPE) & temp_mask;
476
477 /*
478 * data/node io flag bits per temp:
479 * REQ_META | REQ_FUA |
480 * 5 | 4 | 3 | 2 | 1 | 0 |
481 * Cold | Warm | Hot | Cold | Warm | Hot |
482 */
483 if (BIT(fio->temp) & meta_flag)
484 op_flags |= REQ_META;
485 if (BIT(fio->temp) & fua_flag)
486 op_flags |= REQ_FUA;
487
488 if (fio->type == DATA &&
489 F2FS_I(fio->folio->mapping->host)->ioprio_hint == F2FS_IOPRIO_WRITE)
490 op_flags |= REQ_PRIO;
491
492 return op_flags;
493 }
494
__bio_alloc(struct f2fs_io_info * fio,int npages)495 static struct bio *__bio_alloc(struct f2fs_io_info *fio, int npages)
496 {
497 struct f2fs_sb_info *sbi = fio->sbi;
498 struct block_device *bdev;
499 sector_t sector;
500 struct bio *bio;
501
502 bdev = f2fs_target_device(sbi, fio->new_blkaddr, §or);
503 bio = bio_alloc_bioset(bdev, npages,
504 fio->op | fio->op_flags | f2fs_io_flags(fio),
505 GFP_NOIO, &f2fs_bioset);
506 bio->bi_iter.bi_sector = sector;
507 if (is_read_io(fio->op)) {
508 bio->bi_end_io = f2fs_read_end_io;
509 bio->bi_private = NULL;
510 } else {
511 bio->bi_end_io = f2fs_write_end_io;
512 bio->bi_private = sbi;
513 bio->bi_write_hint = f2fs_io_type_to_rw_hint(sbi,
514 fio->type, fio->temp);
515 bio->bi_write_stream = f2fs_io_type_to_write_stream(bdev, fio->type,
516 fio->temp);
517 }
518 iostat_alloc_and_bind_ctx(sbi, bio, NULL);
519
520 if (fio->io_wbc)
521 wbc_init_bio(fio->io_wbc, bio);
522
523 return bio;
524 }
525
f2fs_set_bio_crypt_ctx(struct bio * bio,const struct inode * inode,pgoff_t first_idx,const struct f2fs_io_info * fio,gfp_t gfp_mask)526 static void f2fs_set_bio_crypt_ctx(struct bio *bio, const struct inode *inode,
527 pgoff_t first_idx,
528 const struct f2fs_io_info *fio,
529 gfp_t gfp_mask)
530 {
531 /*
532 * The f2fs garbage collector sets ->encrypted_page when it wants to
533 * read/write raw data without encryption.
534 */
535 if (!fio || !fio->encrypted_page)
536 fscrypt_set_bio_crypt_ctx(bio, inode,
537 (loff_t)first_idx << inode->i_blkbits,
538 gfp_mask);
539 }
540
f2fs_crypt_mergeable_bio(struct bio * bio,const struct inode * inode,pgoff_t next_idx,const struct f2fs_io_info * fio)541 static bool f2fs_crypt_mergeable_bio(struct bio *bio, const struct inode *inode,
542 pgoff_t next_idx,
543 const struct f2fs_io_info *fio)
544 {
545 /*
546 * The f2fs garbage collector sets ->encrypted_page when it wants to
547 * read/write raw data without encryption.
548 */
549 if (fio && fio->encrypted_page)
550 return !bio_has_crypt_ctx(bio);
551
552 return fscrypt_mergeable_bio(bio, inode,
553 (loff_t)next_idx << inode->i_blkbits);
554 }
555
f2fs_submit_read_bio(struct f2fs_sb_info * sbi,struct bio * bio,enum page_type type)556 void f2fs_submit_read_bio(struct f2fs_sb_info *sbi, struct bio *bio,
557 enum page_type type)
558 {
559 if (!bio)
560 return;
561
562 WARN_ON_ONCE(!is_read_io(bio_op(bio)));
563 trace_f2fs_submit_read_bio(sbi->sb, type, bio);
564
565 iostat_update_submit_ctx(bio, type);
566 blk_crypto_submit_bio(bio);
567 }
568
f2fs_submit_write_bio(struct f2fs_sb_info * sbi,struct bio * bio,enum page_type type)569 static void f2fs_submit_write_bio(struct f2fs_sb_info *sbi, struct bio *bio,
570 enum page_type type)
571 {
572 WARN_ON_ONCE(is_read_io(bio_op(bio)));
573 trace_f2fs_submit_write_bio(sbi->sb, type, bio);
574 iostat_update_submit_ctx(bio, type);
575 blk_crypto_submit_bio(bio);
576 }
577
__submit_merged_bio(struct f2fs_bio_info * io)578 static void __submit_merged_bio(struct f2fs_bio_info *io)
579 {
580 struct f2fs_io_info *fio = &io->fio;
581
582 if (!io->bio)
583 return;
584
585 if (is_read_io(fio->op)) {
586 trace_f2fs_prepare_read_bio(io->sbi->sb, fio->type, io->bio);
587 f2fs_submit_read_bio(io->sbi, io->bio, fio->type);
588 } else {
589 trace_f2fs_prepare_write_bio(io->sbi->sb, fio->type, io->bio);
590 f2fs_submit_write_bio(io->sbi, io->bio, fio->type);
591 }
592 io->bio = NULL;
593 }
594
__has_merged_page(struct bio * bio,struct inode * inode,struct folio * folio,nid_t ino)595 static bool __has_merged_page(struct bio *bio, struct inode *inode,
596 struct folio *folio, nid_t ino)
597 {
598 struct folio_iter fi;
599
600 if (!bio)
601 return false;
602
603 if (!inode && !folio && !ino)
604 return true;
605
606 bio_for_each_folio_all(fi, bio) {
607 struct folio *target = fi.folio;
608
609 if (f2fs_is_compressed_page(target)) {
610 target = f2fs_compress_control_folio(target);
611 if (IS_ERR(target))
612 continue;
613 }
614
615 if (inode && inode == target->mapping->host)
616 return true;
617 if (folio && folio == target)
618 return true;
619 if (ino && ino == ino_of_node(target))
620 return true;
621 }
622
623 return false;
624 }
625
f2fs_init_write_merge_io(struct f2fs_sb_info * sbi)626 int f2fs_init_write_merge_io(struct f2fs_sb_info *sbi)
627 {
628 int i;
629
630 for (i = 0; i < NR_PAGE_TYPE; i++) {
631 int n = (i == META) ? 1 : NR_TEMP_TYPE;
632 int j;
633
634 sbi->write_io[i] = f2fs_kmalloc(sbi,
635 array_size(n, sizeof(struct f2fs_bio_info)),
636 GFP_KERNEL);
637 if (!sbi->write_io[i])
638 return -ENOMEM;
639
640 for (j = HOT; j < n; j++) {
641 struct f2fs_bio_info *io = &sbi->write_io[i][j];
642
643 init_f2fs_rwsem_trace(&io->io_rwsem, sbi,
644 LOCK_NAME_IO_RWSEM);
645 io->sbi = sbi;
646 io->bio = NULL;
647 io->last_block_in_bio = 0;
648 spin_lock_init(&io->io_lock);
649 INIT_LIST_HEAD(&io->io_list);
650 INIT_LIST_HEAD(&io->bio_list);
651 init_f2fs_rwsem(&io->bio_list_lock);
652 #ifdef CONFIG_BLK_DEV_ZONED
653 init_completion(&io->zone_wait);
654 io->zone_pending_bio = NULL;
655 io->bi_private = NULL;
656 #endif
657 }
658 }
659
660 return 0;
661 }
662
__f2fs_submit_merged_write(struct f2fs_sb_info * sbi,enum page_type type,enum temp_type temp)663 static void __f2fs_submit_merged_write(struct f2fs_sb_info *sbi,
664 enum page_type type, enum temp_type temp)
665 {
666 enum page_type btype = PAGE_TYPE_OF_BIO(type);
667 struct f2fs_bio_info *io = sbi->write_io[btype] + temp;
668 struct f2fs_lock_context lc;
669
670 f2fs_down_write_trace(&io->io_rwsem, &lc);
671
672 if (!io->bio)
673 goto unlock_out;
674
675 /* change META to META_FLUSH in the checkpoint procedure */
676 if (type >= META_FLUSH) {
677 io->fio.type = META_FLUSH;
678 io->bio->bi_opf |= REQ_META | REQ_PRIO | REQ_SYNC;
679 if (!test_opt(sbi, NOBARRIER))
680 io->bio->bi_opf |= REQ_PREFLUSH | REQ_FUA;
681 }
682 __submit_merged_bio(io);
683 unlock_out:
684 f2fs_up_write_trace(&io->io_rwsem, &lc);
685 }
686
__submit_merged_write_cond(struct f2fs_sb_info * sbi,struct inode * inode,struct folio * folio,nid_t ino,enum page_type type,bool writeback)687 static void __submit_merged_write_cond(struct f2fs_sb_info *sbi,
688 struct inode *inode, struct folio *folio,
689 nid_t ino, enum page_type type, bool writeback)
690 {
691 enum temp_type temp;
692 bool ret = true;
693 bool force = !inode && !folio && !ino;
694
695 for (temp = HOT; temp < NR_TEMP_TYPE; temp++) {
696 if (!force) {
697 enum page_type btype = PAGE_TYPE_OF_BIO(type);
698 struct f2fs_bio_info *io = sbi->write_io[btype] + temp;
699 struct f2fs_lock_context lc;
700
701 f2fs_down_read_trace(&io->io_rwsem, &lc);
702 ret = __has_merged_page(io->bio, inode, folio, ino);
703 f2fs_up_read_trace(&io->io_rwsem, &lc);
704 }
705 if (ret) {
706 __f2fs_submit_merged_write(sbi, type, temp);
707 /*
708 * For waitting writebck case, if the bio owned by the
709 * folio is already submitted, we do not need to submit
710 * other types of bios.
711 */
712 if (writeback)
713 break;
714 }
715
716 /* TODO: use HOT temp only for meta pages now. */
717 if (type >= META)
718 break;
719 }
720 }
721
f2fs_submit_merged_write(struct f2fs_sb_info * sbi,enum page_type type)722 void f2fs_submit_merged_write(struct f2fs_sb_info *sbi, enum page_type type)
723 {
724 __submit_merged_write_cond(sbi, NULL, NULL, 0, type, false);
725 }
726
f2fs_submit_merged_write_cond(struct f2fs_sb_info * sbi,struct inode * inode,struct folio * folio,nid_t ino,enum page_type type)727 void f2fs_submit_merged_write_cond(struct f2fs_sb_info *sbi,
728 struct inode *inode, struct folio *folio,
729 nid_t ino, enum page_type type)
730 {
731 __submit_merged_write_cond(sbi, inode, folio, ino, type, false);
732 }
733
f2fs_submit_merged_write_folio(struct f2fs_sb_info * sbi,struct folio * folio,enum page_type type)734 void f2fs_submit_merged_write_folio(struct f2fs_sb_info *sbi,
735 struct folio *folio, enum page_type type)
736 {
737 __submit_merged_write_cond(sbi, NULL, folio, 0, type, true);
738 }
739
f2fs_flush_merged_writes(struct f2fs_sb_info * sbi)740 void f2fs_flush_merged_writes(struct f2fs_sb_info *sbi)
741 {
742 f2fs_submit_merged_write(sbi, DATA);
743 f2fs_submit_merged_write(sbi, NODE);
744 f2fs_submit_merged_write(sbi, META);
745 }
746
747 /*
748 * Fill the locked page with data located in the block address.
749 * A caller needs to unlock the page on failure.
750 */
f2fs_submit_page_bio(struct f2fs_io_info * fio)751 int f2fs_submit_page_bio(struct f2fs_io_info *fio)
752 {
753 struct bio *bio;
754 struct folio *fio_folio = fio->folio;
755 struct folio *data_folio = fio->encrypted_page ?
756 page_folio(fio->encrypted_page) : fio_folio;
757
758 if (!f2fs_is_valid_blkaddr(fio->sbi, fio->new_blkaddr,
759 fio->is_por ? META_POR : (__is_meta_io(fio) ?
760 META_GENERIC : DATA_GENERIC_ENHANCE)))
761 return -EFSCORRUPTED;
762
763 trace_f2fs_submit_folio_bio(data_folio, fio);
764
765 /* Allocate a new bio */
766 bio = __bio_alloc(fio, 1);
767
768 f2fs_set_bio_crypt_ctx(bio, fio_folio->mapping->host,
769 fio_folio->index, fio, GFP_NOIO);
770 bio_add_folio_nofail(bio, data_folio, folio_size(data_folio), 0);
771
772 if (fio->io_wbc && !is_read_io(fio->op))
773 wbc_account_cgroup_owner(fio->io_wbc, fio_folio, PAGE_SIZE);
774
775 inc_page_count(fio->sbi, is_read_io(fio->op) ?
776 __read_io_type(data_folio) : WB_DATA_TYPE(fio->folio, false));
777
778 if (is_read_io(bio_op(bio)))
779 f2fs_submit_read_bio(fio->sbi, bio, fio->type);
780 else
781 f2fs_submit_write_bio(fio->sbi, bio, fio->type);
782 return 0;
783 }
784
page_is_mergeable(struct f2fs_sb_info * sbi,struct bio * bio,block_t last_blkaddr,block_t cur_blkaddr)785 static bool page_is_mergeable(struct f2fs_sb_info *sbi, struct bio *bio,
786 block_t last_blkaddr, block_t cur_blkaddr)
787 {
788 if (unlikely(sbi->max_io_bytes &&
789 bio->bi_iter.bi_size >= sbi->max_io_bytes))
790 return false;
791 if (last_blkaddr + 1 != cur_blkaddr)
792 return false;
793 return bio->bi_bdev == f2fs_target_device(sbi, cur_blkaddr, NULL);
794 }
795
io_type_is_mergeable(struct f2fs_bio_info * io,struct f2fs_io_info * fio)796 static bool io_type_is_mergeable(struct f2fs_bio_info *io,
797 struct f2fs_io_info *fio)
798 {
799 blk_opf_t mask = ~(REQ_PREFLUSH | REQ_FUA);
800
801 if (io->fio.op != fio->op)
802 return false;
803 return (io->fio.op_flags & mask) == (fio->op_flags & mask);
804 }
805
io_is_mergeable(struct f2fs_sb_info * sbi,struct bio * bio,struct f2fs_bio_info * io,struct f2fs_io_info * fio,block_t last_blkaddr,block_t cur_blkaddr)806 static bool io_is_mergeable(struct f2fs_sb_info *sbi, struct bio *bio,
807 struct f2fs_bio_info *io,
808 struct f2fs_io_info *fio,
809 block_t last_blkaddr,
810 block_t cur_blkaddr)
811 {
812 if (!page_is_mergeable(sbi, bio, last_blkaddr, cur_blkaddr))
813 return false;
814 return io_type_is_mergeable(io, fio);
815 }
816
add_bio_entry(struct f2fs_sb_info * sbi,struct bio * bio,struct folio * folio,enum temp_type temp)817 static void add_bio_entry(struct f2fs_sb_info *sbi, struct bio *bio,
818 struct folio *folio, enum temp_type temp)
819 {
820 struct f2fs_bio_info *io = sbi->write_io[DATA] + temp;
821 struct bio_entry *be;
822
823 be = f2fs_kmem_cache_alloc(bio_entry_slab, GFP_NOFS, true, NULL);
824 be->bio = bio;
825 bio_get(bio);
826
827 bio_add_folio_nofail(bio, folio, folio_size(folio), 0);
828
829 f2fs_down_write(&io->bio_list_lock);
830 list_add_tail(&be->list, &io->bio_list);
831 f2fs_up_write(&io->bio_list_lock);
832 }
833
del_bio_entry(struct bio_entry * be)834 static void del_bio_entry(struct bio_entry *be)
835 {
836 list_del(&be->list);
837 kmem_cache_free(bio_entry_slab, be);
838 }
839
add_ipu_page(struct f2fs_io_info * fio,struct bio ** bio,struct folio * folio)840 static int add_ipu_page(struct f2fs_io_info *fio, struct bio **bio,
841 struct folio *folio)
842 {
843 struct folio *fio_folio = fio->folio;
844 struct f2fs_sb_info *sbi = fio->sbi;
845 enum temp_type temp;
846 bool found = false;
847 int ret = -EAGAIN;
848
849 for (temp = HOT; temp < NR_TEMP_TYPE && !found; temp++) {
850 struct f2fs_bio_info *io = sbi->write_io[DATA] + temp;
851 struct list_head *head = &io->bio_list;
852 struct bio_entry *be;
853
854 f2fs_down_write(&io->bio_list_lock);
855 list_for_each_entry(be, head, list) {
856 if (be->bio != *bio)
857 continue;
858
859 found = true;
860
861 f2fs_bug_on(sbi, !page_is_mergeable(sbi, *bio,
862 *fio->last_block,
863 fio->new_blkaddr));
864 if (f2fs_crypt_mergeable_bio(*bio,
865 fio_folio->mapping->host,
866 fio_folio->index, fio) &&
867 bio_add_folio(*bio, folio, folio_size(folio), 0)) {
868 ret = 0;
869 break;
870 }
871
872 /* page can't be merged into bio; submit the bio */
873 del_bio_entry(be);
874 f2fs_submit_write_bio(sbi, *bio, DATA);
875 break;
876 }
877 f2fs_up_write(&io->bio_list_lock);
878 }
879
880 if (ret) {
881 bio_put(*bio);
882 *bio = NULL;
883 }
884
885 return ret;
886 }
887
f2fs_submit_merged_ipu_write(struct f2fs_sb_info * sbi,struct bio ** bio,struct folio * folio)888 void f2fs_submit_merged_ipu_write(struct f2fs_sb_info *sbi,
889 struct bio **bio, struct folio *folio)
890 {
891 enum temp_type temp;
892 bool found = false;
893 struct bio *target = bio ? *bio : NULL;
894
895 f2fs_bug_on(sbi, !target && !folio);
896
897 for (temp = HOT; temp < NR_TEMP_TYPE && !found; temp++) {
898 struct f2fs_bio_info *io = sbi->write_io[DATA] + temp;
899 struct list_head *head = &io->bio_list;
900 struct bio_entry *be;
901
902 if (list_empty(head))
903 continue;
904
905 f2fs_down_read(&io->bio_list_lock);
906 list_for_each_entry(be, head, list) {
907 if (target)
908 found = (target == be->bio);
909 else
910 found = __has_merged_page(be->bio, NULL,
911 folio, 0);
912 if (found)
913 break;
914 }
915 f2fs_up_read(&io->bio_list_lock);
916
917 if (!found)
918 continue;
919
920 found = false;
921
922 f2fs_down_write(&io->bio_list_lock);
923 list_for_each_entry(be, head, list) {
924 if (target)
925 found = (target == be->bio);
926 else
927 found = __has_merged_page(be->bio, NULL,
928 folio, 0);
929 if (found) {
930 target = be->bio;
931 del_bio_entry(be);
932 break;
933 }
934 }
935 f2fs_up_write(&io->bio_list_lock);
936 }
937
938 if (found)
939 f2fs_submit_write_bio(sbi, target, DATA);
940 if (bio && *bio) {
941 bio_put(*bio);
942 *bio = NULL;
943 }
944 }
945
f2fs_submit_all_merged_ipu_writes(struct f2fs_sb_info * sbi)946 void f2fs_submit_all_merged_ipu_writes(struct f2fs_sb_info *sbi)
947 {
948 struct bio_entry *be, *tmp;
949 struct f2fs_bio_info *io;
950 enum temp_type temp;
951
952 for (temp = HOT; temp < NR_TEMP_TYPE; temp++) {
953 LIST_HEAD(list);
954
955 io = sbi->write_io[DATA] + temp;
956
957 /* A lockless list_empty() check is safe here: any bios from
958 * other kworkers that we miss will be submitted by those
959 * kworkers accordingly.
960 */
961 if (list_empty(&io->bio_list))
962 continue;
963
964 f2fs_down_write(&io->bio_list_lock);
965 list_splice_init(&io->bio_list, &list);
966 f2fs_up_write(&io->bio_list_lock);
967
968 list_for_each_entry_safe(be, tmp, &list, list) {
969 f2fs_submit_write_bio(sbi, be->bio, DATA);
970 del_bio_entry(be);
971 }
972 }
973 }
974
f2fs_merge_page_bio(struct f2fs_io_info * fio)975 int f2fs_merge_page_bio(struct f2fs_io_info *fio)
976 {
977 struct bio *bio = *fio->bio;
978 struct folio *data_folio = fio->encrypted_page ?
979 page_folio(fio->encrypted_page) : fio->folio;
980 struct folio *folio = fio->folio;
981
982 if (!f2fs_is_valid_blkaddr(fio->sbi, fio->new_blkaddr,
983 __is_meta_io(fio) ? META_GENERIC : DATA_GENERIC))
984 return -EFSCORRUPTED;
985
986 trace_f2fs_submit_folio_bio(data_folio, fio);
987
988 if (bio && !page_is_mergeable(fio->sbi, bio, *fio->last_block,
989 fio->new_blkaddr))
990 f2fs_submit_merged_ipu_write(fio->sbi, &bio, NULL);
991 alloc_new:
992 if (!bio) {
993 bio = __bio_alloc(fio, BIO_MAX_VECS);
994 f2fs_set_bio_crypt_ctx(bio, folio->mapping->host,
995 folio->index, fio, GFP_NOIO);
996
997 add_bio_entry(fio->sbi, bio, data_folio, fio->temp);
998 } else {
999 if (add_ipu_page(fio, &bio, data_folio))
1000 goto alloc_new;
1001 }
1002
1003 if (fio->io_wbc)
1004 wbc_account_cgroup_owner(fio->io_wbc, folio, folio_size(folio));
1005
1006 inc_page_count(fio->sbi, WB_DATA_TYPE(folio, false));
1007
1008 *fio->last_block = fio->new_blkaddr;
1009 *fio->bio = bio;
1010
1011 return 0;
1012 }
1013
1014 #ifdef CONFIG_BLK_DEV_ZONED
is_end_zone_blkaddr(struct f2fs_sb_info * sbi,block_t blkaddr)1015 static bool is_end_zone_blkaddr(struct f2fs_sb_info *sbi, block_t blkaddr)
1016 {
1017 struct block_device *bdev = sbi->sb->s_bdev;
1018 int devi = 0;
1019
1020 if (f2fs_is_multi_device(sbi)) {
1021 devi = f2fs_target_device_index(sbi, blkaddr);
1022 if (blkaddr < FDEV(devi).start_blk ||
1023 blkaddr > FDEV(devi).end_blk) {
1024 f2fs_err(sbi, "Invalid block %x", blkaddr);
1025 return false;
1026 }
1027 blkaddr -= FDEV(devi).start_blk;
1028 bdev = FDEV(devi).bdev;
1029 }
1030 return bdev_is_zoned(bdev) &&
1031 f2fs_blkz_is_seq(sbi, devi, blkaddr) &&
1032 (blkaddr % sbi->blocks_per_blkz == sbi->blocks_per_blkz - 1);
1033 }
1034 #endif
1035
f2fs_submit_page_write(struct f2fs_io_info * fio)1036 void f2fs_submit_page_write(struct f2fs_io_info *fio)
1037 {
1038 struct f2fs_sb_info *sbi = fio->sbi;
1039 enum page_type btype = PAGE_TYPE_OF_BIO(fio->type);
1040 struct f2fs_bio_info *io = sbi->write_io[btype] + fio->temp;
1041 struct folio *bio_folio;
1042 struct f2fs_lock_context lc;
1043 enum count_type type;
1044
1045 f2fs_bug_on(sbi, is_read_io(fio->op));
1046
1047 f2fs_down_write_trace(&io->io_rwsem, &lc);
1048 next:
1049 #ifdef CONFIG_BLK_DEV_ZONED
1050 if (f2fs_sb_has_blkzoned(sbi) && btype < META && io->zone_pending_bio) {
1051 wait_for_completion_io(&io->zone_wait);
1052 bio_put(io->zone_pending_bio);
1053 io->zone_pending_bio = NULL;
1054 io->bi_private = NULL;
1055 }
1056 #endif
1057
1058 if (fio->in_list) {
1059 spin_lock(&io->io_lock);
1060 if (list_empty(&io->io_list)) {
1061 spin_unlock(&io->io_lock);
1062 goto out;
1063 }
1064 fio = list_first_entry(&io->io_list,
1065 struct f2fs_io_info, list);
1066 list_del(&fio->list);
1067 spin_unlock(&io->io_lock);
1068 }
1069
1070 verify_fio_blkaddr(fio);
1071
1072 if (fio->encrypted_page)
1073 bio_folio = page_folio(fio->encrypted_page);
1074 else if (fio->compressed_page)
1075 bio_folio = page_folio(fio->compressed_page);
1076 else
1077 bio_folio = fio->folio;
1078
1079 /* set submitted = true as a return value */
1080 fio->submitted = 1;
1081
1082 type = WB_DATA_TYPE(bio_folio, fio->compressed_page);
1083 inc_page_count(sbi, type);
1084
1085 if (io->bio &&
1086 (!io_is_mergeable(sbi, io->bio, io, fio, io->last_block_in_bio,
1087 fio->new_blkaddr) ||
1088 !f2fs_crypt_mergeable_bio(io->bio, fio_inode(fio),
1089 bio_folio->index, fio)))
1090 __submit_merged_bio(io);
1091 alloc_new:
1092 if (io->bio == NULL) {
1093 io->bio = __bio_alloc(fio, BIO_MAX_VECS);
1094 f2fs_set_bio_crypt_ctx(io->bio, fio_inode(fio),
1095 bio_folio->index, fio, GFP_NOIO);
1096 io->fio = *fio;
1097 }
1098
1099 if (!bio_add_folio(io->bio, bio_folio, folio_size(bio_folio), 0)) {
1100 __submit_merged_bio(io);
1101 goto alloc_new;
1102 }
1103
1104 if (fio->io_wbc)
1105 wbc_account_cgroup_owner(fio->io_wbc, fio->folio,
1106 folio_size(fio->folio));
1107
1108 io->last_block_in_bio = fio->new_blkaddr;
1109
1110 trace_f2fs_submit_folio_write(fio->folio, fio);
1111 #ifdef CONFIG_BLK_DEV_ZONED
1112 if (f2fs_sb_has_blkzoned(sbi) && btype < META &&
1113 is_end_zone_blkaddr(sbi, fio->new_blkaddr)) {
1114 bio_get(io->bio);
1115 reinit_completion(&io->zone_wait);
1116 io->bi_private = io->bio->bi_private;
1117 io->bio->bi_private = io;
1118 io->bio->bi_end_io = f2fs_zone_write_end_io;
1119 io->zone_pending_bio = io->bio;
1120 __submit_merged_bio(io);
1121 }
1122 #endif
1123 if (fio->in_list)
1124 goto next;
1125 out:
1126 if (is_sbi_flag_set(sbi, SBI_IS_SHUTDOWN) ||
1127 !f2fs_is_checkpoint_ready(sbi))
1128 __submit_merged_bio(io);
1129 f2fs_up_write_trace(&io->io_rwsem, &lc);
1130 }
1131
f2fs_grab_read_bio(struct inode * inode,struct fsverity_info * vi,block_t blkaddr,unsigned nr_pages,blk_opf_t op_flag,pgoff_t first_idx,bool for_write)1132 static struct bio *f2fs_grab_read_bio(struct inode *inode,
1133 struct fsverity_info *vi, block_t blkaddr,
1134 unsigned nr_pages, blk_opf_t op_flag,
1135 pgoff_t first_idx, bool for_write)
1136 {
1137 struct f2fs_sb_info *sbi = F2FS_I_SB(inode);
1138 struct bio *bio;
1139 struct bio_post_read_ctx *ctx = NULL;
1140 unsigned int post_read_steps = 0;
1141 sector_t sector;
1142 struct block_device *bdev = f2fs_target_device(sbi, blkaddr, §or);
1143
1144 bio = bio_alloc_bioset(bdev, bio_max_segs(nr_pages),
1145 REQ_OP_READ | op_flag,
1146 for_write ? GFP_NOIO : GFP_KERNEL, &f2fs_bioset);
1147 bio->bi_iter.bi_sector = sector;
1148 f2fs_set_bio_crypt_ctx(bio, inode, first_idx, NULL, GFP_NOFS);
1149 bio->bi_end_io = f2fs_read_end_io;
1150
1151 if (vi)
1152 post_read_steps |= STEP_VERITY;
1153
1154 /*
1155 * STEP_DECOMPRESS is handled specially, since a compressed file might
1156 * contain both compressed and uncompressed clusters. We'll allocate a
1157 * bio_post_read_ctx if the file is compressed, but the caller is
1158 * responsible for enabling STEP_DECOMPRESS if it's actually needed.
1159 */
1160
1161 if (post_read_steps || f2fs_compressed_file(inode)) {
1162 /* Due to the mempool, this never fails. */
1163 ctx = mempool_alloc(bio_post_read_ctx_pool, GFP_NOFS);
1164 ctx->bio = bio;
1165 ctx->sbi = sbi;
1166 ctx->vi = vi;
1167 ctx->enabled_steps = post_read_steps;
1168 ctx->fs_blkaddr = blkaddr;
1169 ctx->decompression_attempted = false;
1170 bio->bi_private = ctx;
1171 }
1172 iostat_alloc_and_bind_ctx(sbi, bio, ctx);
1173
1174 return bio;
1175 }
1176
1177 /* This can handle encryption stuffs */
f2fs_submit_page_read(struct inode * inode,struct fsverity_info * vi,struct folio * folio,block_t blkaddr,blk_opf_t op_flags,bool for_write)1178 static void f2fs_submit_page_read(struct inode *inode, struct fsverity_info *vi,
1179 struct folio *folio, block_t blkaddr,
1180 blk_opf_t op_flags, bool for_write)
1181 {
1182 struct f2fs_sb_info *sbi = F2FS_I_SB(inode);
1183 struct bio *bio;
1184
1185 bio = f2fs_grab_read_bio(inode, vi, blkaddr, 1, op_flags, folio->index,
1186 for_write);
1187
1188 /* wait for GCed page writeback via META_MAPPING */
1189 f2fs_wait_on_block_writeback(inode, blkaddr);
1190
1191 if (!bio_add_folio(bio, folio, PAGE_SIZE, 0))
1192 f2fs_bug_on(sbi, 1);
1193
1194 inc_page_count(sbi, F2FS_RD_DATA);
1195 f2fs_update_iostat(sbi, NULL, FS_DATA_READ_IO, F2FS_BLKSIZE);
1196 f2fs_submit_read_bio(sbi, bio, DATA);
1197 }
1198
__set_data_blkaddr(struct dnode_of_data * dn,block_t blkaddr)1199 static void __set_data_blkaddr(struct dnode_of_data *dn, block_t blkaddr)
1200 {
1201 __le32 *addr = get_dnode_addr(dn->inode, dn->node_folio);
1202
1203 dn->data_blkaddr = blkaddr;
1204 addr[dn->ofs_in_node] = cpu_to_le32(dn->data_blkaddr);
1205 }
1206
1207 /*
1208 * Lock ordering for the change of data block address:
1209 * ->data_page
1210 * ->node_folio
1211 * update block addresses in the node page
1212 */
f2fs_set_data_blkaddr(struct dnode_of_data * dn,block_t blkaddr)1213 void f2fs_set_data_blkaddr(struct dnode_of_data *dn, block_t blkaddr)
1214 {
1215 f2fs_folio_wait_writeback(dn->node_folio, NODE, true, true);
1216 __set_data_blkaddr(dn, blkaddr);
1217 if (folio_mark_dirty(dn->node_folio))
1218 dn->node_changed = true;
1219 }
1220
f2fs_update_data_blkaddr(struct dnode_of_data * dn,block_t blkaddr)1221 void f2fs_update_data_blkaddr(struct dnode_of_data *dn, block_t blkaddr)
1222 {
1223 f2fs_set_data_blkaddr(dn, blkaddr);
1224 f2fs_update_read_extent_cache(dn);
1225 }
1226
1227 /* dn->ofs_in_node will be returned with up-to-date last block pointer */
f2fs_reserve_new_blocks(struct dnode_of_data * dn,blkcnt_t count)1228 int f2fs_reserve_new_blocks(struct dnode_of_data *dn, blkcnt_t count)
1229 {
1230 struct f2fs_sb_info *sbi = F2FS_I_SB(dn->inode);
1231 int err;
1232
1233 if (!count)
1234 return 0;
1235
1236 if (unlikely(is_inode_flag_set(dn->inode, FI_NO_ALLOC)))
1237 return -EPERM;
1238 err = inc_valid_block_count(sbi, dn->inode, &count, true, false);
1239 if (unlikely(err))
1240 return err;
1241
1242 trace_f2fs_reserve_new_blocks(dn->inode, dn->nid,
1243 dn->ofs_in_node, count);
1244
1245 f2fs_folio_wait_writeback(dn->node_folio, NODE, true, true);
1246
1247 for (; count > 0; dn->ofs_in_node++) {
1248 block_t blkaddr = f2fs_data_blkaddr(dn);
1249
1250 if (blkaddr == NULL_ADDR) {
1251 __set_data_blkaddr(dn, NEW_ADDR);
1252 count--;
1253 }
1254 }
1255
1256 if (folio_mark_dirty(dn->node_folio))
1257 dn->node_changed = true;
1258 return 0;
1259 }
1260
1261 /* Should keep dn->ofs_in_node unchanged */
f2fs_reserve_new_block(struct dnode_of_data * dn)1262 int f2fs_reserve_new_block(struct dnode_of_data *dn)
1263 {
1264 unsigned int ofs_in_node = dn->ofs_in_node;
1265 int ret;
1266
1267 ret = f2fs_reserve_new_blocks(dn, 1);
1268 dn->ofs_in_node = ofs_in_node;
1269 return ret;
1270 }
1271
f2fs_reserve_block(struct dnode_of_data * dn,pgoff_t index)1272 int f2fs_reserve_block(struct dnode_of_data *dn, pgoff_t index)
1273 {
1274 bool need_put = dn->inode_folio ? false : true;
1275 int err;
1276
1277 err = f2fs_get_dnode_of_data(dn, index, ALLOC_NODE);
1278 if (err)
1279 return err;
1280
1281 if (dn->data_blkaddr == NULL_ADDR)
1282 err = f2fs_reserve_new_block(dn);
1283 if (err || need_put)
1284 f2fs_put_dnode(dn);
1285 return err;
1286 }
1287
f2fs_need_verity(const struct inode * inode,pgoff_t idx)1288 static inline struct fsverity_info *f2fs_need_verity(const struct inode *inode,
1289 pgoff_t idx)
1290 {
1291 if (idx < DIV_ROUND_UP(inode->i_size, PAGE_SIZE))
1292 return fsverity_get_info(inode);
1293 return NULL;
1294 }
1295
f2fs_get_read_data_folio(struct inode * inode,pgoff_t index,blk_opf_t op_flags,bool for_write,pgoff_t * next_pgofs)1296 struct folio *f2fs_get_read_data_folio(struct inode *inode, pgoff_t index,
1297 blk_opf_t op_flags, bool for_write, pgoff_t *next_pgofs)
1298 {
1299 struct address_space *mapping = inode->i_mapping;
1300 struct dnode_of_data dn;
1301 struct folio *folio;
1302 int err;
1303 retry:
1304 folio = f2fs_grab_cache_folio(mapping, index, for_write);
1305 if (IS_ERR(folio))
1306 return folio;
1307
1308 if (folio_test_large(folio)) {
1309 pgoff_t folio_index = mapping_align_index(mapping, index);
1310 unsigned long nr_pages = folio_nr_pages(folio);
1311
1312 f2fs_folio_put(folio, true);
1313 invalidate_inode_pages2_range(mapping, folio_index,
1314 folio_index + nr_pages - 1);
1315 f2fs_schedule_timeout(DEFAULT_SCHEDULE_TIMEOUT);
1316 goto retry;
1317 }
1318
1319 if (f2fs_lookup_read_extent_cache_block(inode, index,
1320 &dn.data_blkaddr)) {
1321 if (!f2fs_is_valid_blkaddr(F2FS_I_SB(inode), dn.data_blkaddr,
1322 DATA_GENERIC_ENHANCE_READ)) {
1323 err = -EFSCORRUPTED;
1324 goto put_err;
1325 }
1326 goto got_it;
1327 }
1328
1329 set_new_dnode(&dn, inode, NULL, NULL, 0);
1330 err = f2fs_get_dnode_of_data(&dn, index, LOOKUP_NODE);
1331 if (err) {
1332 if (err == -ENOENT && next_pgofs)
1333 *next_pgofs = f2fs_get_next_page_offset(&dn, index);
1334 goto put_err;
1335 }
1336 f2fs_put_dnode(&dn);
1337
1338 if (unlikely(dn.data_blkaddr == NULL_ADDR)) {
1339 err = -ENOENT;
1340 if (next_pgofs)
1341 *next_pgofs = index + 1;
1342 goto put_err;
1343 }
1344 if (dn.data_blkaddr != NEW_ADDR &&
1345 !f2fs_is_valid_blkaddr(F2FS_I_SB(inode),
1346 dn.data_blkaddr,
1347 DATA_GENERIC_ENHANCE)) {
1348 err = -EFSCORRUPTED;
1349 goto put_err;
1350 }
1351 got_it:
1352 if (folio_test_uptodate(folio)) {
1353 folio_unlock(folio);
1354 return folio;
1355 }
1356
1357 /*
1358 * A new dentry page is allocated but not able to be written, since its
1359 * new inode page couldn't be allocated due to -ENOSPC.
1360 * In such the case, its blkaddr can be remained as NEW_ADDR.
1361 * see, f2fs_add_link -> f2fs_get_new_data_folio ->
1362 * f2fs_init_inode_metadata.
1363 */
1364 if (dn.data_blkaddr == NEW_ADDR) {
1365 folio_zero_segment(folio, 0, folio_size(folio));
1366 if (!folio_test_uptodate(folio))
1367 folio_mark_uptodate(folio);
1368 folio_unlock(folio);
1369 return folio;
1370 }
1371
1372 f2fs_submit_page_read(inode, f2fs_need_verity(inode, folio->index),
1373 folio, dn.data_blkaddr, op_flags, for_write);
1374 return folio;
1375
1376 put_err:
1377 f2fs_folio_put(folio, true);
1378 return ERR_PTR(err);
1379 }
1380
f2fs_find_data_folio(struct inode * inode,pgoff_t index,pgoff_t * next_pgofs)1381 struct folio *f2fs_find_data_folio(struct inode *inode, pgoff_t index,
1382 pgoff_t *next_pgofs)
1383 {
1384 struct address_space *mapping = inode->i_mapping;
1385 struct folio *folio;
1386
1387 folio = f2fs_filemap_get_folio(mapping, index, FGP_ACCESSED, 0);
1388 if (IS_ERR(folio))
1389 goto read;
1390 if (folio_test_uptodate(folio))
1391 return folio;
1392 f2fs_folio_put(folio, false);
1393
1394 read:
1395 folio = f2fs_get_read_data_folio(inode, index, 0, false, next_pgofs);
1396 if (IS_ERR(folio))
1397 return folio;
1398
1399 if (folio_test_uptodate(folio))
1400 return folio;
1401
1402 folio_wait_locked(folio);
1403 if (unlikely(!folio_test_uptodate(folio))) {
1404 f2fs_folio_put(folio, false);
1405 return ERR_PTR(-EIO);
1406 }
1407 return folio;
1408 }
1409
1410 /*
1411 * If it tries to access a hole, return an error.
1412 * Because, the callers, functions in dir.c and GC, should be able to know
1413 * whether this page exists or not.
1414 */
f2fs_get_lock_data_folio(struct inode * inode,pgoff_t index,bool for_write)1415 struct folio *f2fs_get_lock_data_folio(struct inode *inode, pgoff_t index,
1416 bool for_write)
1417 {
1418 struct address_space *mapping = inode->i_mapping;
1419 struct folio *folio;
1420
1421 folio = f2fs_get_read_data_folio(inode, index, 0, for_write, NULL);
1422 if (IS_ERR(folio))
1423 return folio;
1424
1425 /* wait for read completion */
1426 folio_lock(folio);
1427 if (unlikely(folio->mapping != mapping || !folio_test_uptodate(folio))) {
1428 f2fs_folio_put(folio, true);
1429 return ERR_PTR(-EIO);
1430 }
1431 return folio;
1432 }
1433
1434 /*
1435 * Caller ensures that this data page is never allocated.
1436 * A new zero-filled data page is allocated in the page cache.
1437 *
1438 * Also, caller should grab and release a rwsem by calling f2fs_lock_op() and
1439 * f2fs_unlock_op().
1440 * Note that, ifolio is set only by make_empty_dir, and if any error occur,
1441 * ifolio should be released by this function.
1442 */
f2fs_get_new_data_folio(struct inode * inode,struct folio * ifolio,pgoff_t index,bool new_i_size)1443 struct folio *f2fs_get_new_data_folio(struct inode *inode,
1444 struct folio *ifolio, pgoff_t index, bool new_i_size)
1445 {
1446 struct address_space *mapping = inode->i_mapping;
1447 struct folio *folio;
1448 struct dnode_of_data dn;
1449 int err;
1450
1451 folio = f2fs_grab_cache_folio(mapping, index, true);
1452 if (IS_ERR(folio)) {
1453 /*
1454 * before exiting, we should make sure ifolio will be released
1455 * if any error occur.
1456 */
1457 f2fs_folio_put(ifolio, true);
1458 return ERR_PTR(-ENOMEM);
1459 }
1460
1461 set_new_dnode(&dn, inode, ifolio, NULL, 0);
1462 err = f2fs_reserve_block(&dn, index);
1463 if (err) {
1464 f2fs_folio_put(folio, true);
1465 return ERR_PTR(err);
1466 }
1467 if (!ifolio)
1468 f2fs_put_dnode(&dn);
1469
1470 if (folio_test_uptodate(folio))
1471 goto got_it;
1472
1473 if (dn.data_blkaddr == NEW_ADDR) {
1474 folio_zero_segment(folio, 0, folio_size(folio));
1475 if (!folio_test_uptodate(folio))
1476 folio_mark_uptodate(folio);
1477 } else {
1478 f2fs_folio_put(folio, true);
1479
1480 /* if ifolio exists, blkaddr should be NEW_ADDR */
1481 f2fs_bug_on(F2FS_I_SB(inode), ifolio);
1482 folio = f2fs_get_lock_data_folio(inode, index, true);
1483 if (IS_ERR(folio))
1484 return folio;
1485 }
1486 got_it:
1487 if (new_i_size && i_size_read(inode) <
1488 ((loff_t)(index + 1) << PAGE_SHIFT))
1489 f2fs_i_size_write(inode, ((loff_t)(index + 1) << PAGE_SHIFT));
1490 return folio;
1491 }
1492
__allocate_data_block(struct dnode_of_data * dn,int seg_type)1493 static int __allocate_data_block(struct dnode_of_data *dn, int seg_type)
1494 {
1495 struct f2fs_sb_info *sbi = F2FS_I_SB(dn->inode);
1496 struct f2fs_summary sum;
1497 struct node_info ni;
1498 block_t old_blkaddr;
1499 blkcnt_t count = 1;
1500 int err;
1501
1502 if (unlikely(is_inode_flag_set(dn->inode, FI_NO_ALLOC)))
1503 return -EPERM;
1504
1505 err = f2fs_get_node_info(sbi, dn->nid, &ni, false);
1506 if (err)
1507 return err;
1508
1509 dn->data_blkaddr = f2fs_data_blkaddr(dn);
1510 if (dn->data_blkaddr == NULL_ADDR) {
1511 err = inc_valid_block_count(sbi, dn->inode, &count, true, false);
1512 if (unlikely(err))
1513 return err;
1514 }
1515
1516 set_summary(&sum, dn->nid, dn->ofs_in_node, ni.version);
1517 old_blkaddr = dn->data_blkaddr;
1518 err = f2fs_allocate_data_block(sbi, NULL, old_blkaddr,
1519 &dn->data_blkaddr, &sum, seg_type, NULL);
1520 if (err) {
1521 if (old_blkaddr == NULL_ADDR)
1522 dec_valid_block_count(sbi, dn->inode, count);
1523 return err;
1524 }
1525
1526 if (GET_SEGNO(sbi, old_blkaddr) != NULL_SEGNO)
1527 f2fs_invalidate_internal_cache(sbi, old_blkaddr, 1);
1528
1529 f2fs_update_data_blkaddr(dn, dn->data_blkaddr);
1530 return 0;
1531 }
1532
f2fs_map_lock(struct f2fs_sb_info * sbi,struct f2fs_lock_context * lc,int flag)1533 static void f2fs_map_lock(struct f2fs_sb_info *sbi,
1534 struct f2fs_lock_context *lc,
1535 int flag)
1536 {
1537 if (flag == F2FS_GET_BLOCK_PRE_AIO)
1538 f2fs_down_read_trace(&sbi->node_change, lc);
1539 else
1540 f2fs_lock_op(sbi, lc);
1541 }
1542
f2fs_map_unlock(struct f2fs_sb_info * sbi,struct f2fs_lock_context * lc,int flag)1543 static void f2fs_map_unlock(struct f2fs_sb_info *sbi,
1544 struct f2fs_lock_context *lc,
1545 int flag)
1546 {
1547 if (flag == F2FS_GET_BLOCK_PRE_AIO)
1548 f2fs_up_read_trace(&sbi->node_change, lc);
1549 else
1550 f2fs_unlock_op(sbi, lc);
1551 }
1552
f2fs_get_block_locked(struct dnode_of_data * dn,pgoff_t index)1553 int f2fs_get_block_locked(struct dnode_of_data *dn, pgoff_t index)
1554 {
1555 struct f2fs_sb_info *sbi = F2FS_I_SB(dn->inode);
1556 struct f2fs_lock_context lc;
1557 int err = 0;
1558
1559 f2fs_map_lock(sbi, &lc, F2FS_GET_BLOCK_PRE_AIO);
1560 if (!f2fs_lookup_read_extent_cache_block(dn->inode, index,
1561 &dn->data_blkaddr))
1562 err = f2fs_reserve_block(dn, index);
1563 f2fs_map_unlock(sbi, &lc, F2FS_GET_BLOCK_PRE_AIO);
1564
1565 return err;
1566 }
1567
f2fs_map_no_dnode(struct inode * inode,struct f2fs_map_blocks * map,struct dnode_of_data * dn,pgoff_t pgoff)1568 static int f2fs_map_no_dnode(struct inode *inode,
1569 struct f2fs_map_blocks *map, struct dnode_of_data *dn,
1570 pgoff_t pgoff)
1571 {
1572 struct f2fs_sb_info *sbi = F2FS_I_SB(inode);
1573
1574 /*
1575 * There is one exceptional case that read_node_page() may return
1576 * -ENOENT due to filesystem has been shutdown or cp_error, return
1577 * -EIO in that case.
1578 */
1579 if (map->m_may_create &&
1580 (is_sbi_flag_set(sbi, SBI_IS_SHUTDOWN) || f2fs_cp_error(sbi)))
1581 return -EIO;
1582
1583 if (map->m_next_pgofs)
1584 *map->m_next_pgofs = f2fs_get_next_page_offset(dn, pgoff);
1585 if (map->m_next_extent)
1586 *map->m_next_extent = f2fs_get_next_page_offset(dn, pgoff);
1587 return 0;
1588 }
1589
f2fs_map_blocks_cached(struct inode * inode,struct f2fs_map_blocks * map,int flag)1590 static bool f2fs_map_blocks_cached(struct inode *inode,
1591 struct f2fs_map_blocks *map, int flag)
1592 {
1593 struct f2fs_sb_info *sbi = F2FS_I_SB(inode);
1594 unsigned int maxblocks = map->m_len;
1595 pgoff_t pgoff = (pgoff_t)map->m_lblk;
1596 struct extent_info ei = {};
1597
1598 if (!f2fs_lookup_read_extent_cache(inode, pgoff, &ei))
1599 return false;
1600
1601 map->m_pblk = ei.blk + pgoff - ei.fofs;
1602 map->m_len = min((pgoff_t)maxblocks, ei.fofs + ei.len - pgoff);
1603 map->m_flags = F2FS_MAP_MAPPED;
1604 if (map->m_next_extent)
1605 *map->m_next_extent = pgoff + map->m_len;
1606
1607 /* for hardware encryption, but to avoid potential issue in future */
1608 if (flag == F2FS_GET_BLOCK_DIO)
1609 f2fs_wait_on_block_writeback_range(inode,
1610 map->m_pblk, map->m_len);
1611
1612 map->m_multidev_dio = f2fs_allow_multi_device_dio(sbi, flag);
1613 if (map->m_multidev_dio) {
1614 int bidx = f2fs_target_device_index(sbi, map->m_pblk);
1615 struct f2fs_dev_info *dev = &sbi->devs[bidx];
1616
1617 map->m_bdev = dev->bdev;
1618 map->m_len = min(map->m_len, dev->end_blk + 1 - map->m_pblk);
1619 map->m_pblk -= dev->start_blk;
1620 } else {
1621 map->m_bdev = inode->i_sb->s_bdev;
1622 }
1623 return true;
1624 }
1625
map_is_mergeable(struct f2fs_sb_info * sbi,struct f2fs_map_blocks * map,block_t blkaddr,int flag,int bidx,int ofs)1626 static bool map_is_mergeable(struct f2fs_sb_info *sbi,
1627 struct f2fs_map_blocks *map,
1628 block_t blkaddr, int flag, int bidx,
1629 int ofs)
1630 {
1631 if (map->m_multidev_dio && map->m_bdev != FDEV(bidx).bdev)
1632 return false;
1633 if (map->m_pblk != NEW_ADDR && blkaddr == (map->m_pblk + ofs))
1634 return true;
1635 if (map->m_pblk == NEW_ADDR && blkaddr == NEW_ADDR)
1636 return true;
1637 if (flag == F2FS_GET_BLOCK_PRE_DIO)
1638 return true;
1639 if (flag == F2FS_GET_BLOCK_DIO &&
1640 map->m_pblk == NULL_ADDR && blkaddr == NULL_ADDR)
1641 return true;
1642 return false;
1643 }
1644
1645 /*
1646 * f2fs_map_blocks() tries to find or build mapping relationship which
1647 * maps continuous logical blocks to physical blocks, and return such
1648 * info via f2fs_map_blocks structure.
1649 */
f2fs_map_blocks(struct inode * inode,struct f2fs_map_blocks * map,int flag)1650 int f2fs_map_blocks(struct inode *inode, struct f2fs_map_blocks *map, int flag)
1651 {
1652 unsigned int maxblocks = map->m_len;
1653 struct dnode_of_data dn;
1654 struct f2fs_sb_info *sbi = F2FS_I_SB(inode);
1655 struct f2fs_lock_context lc;
1656 int mode = map->m_may_create ? ALLOC_NODE : LOOKUP_NODE;
1657 pgoff_t pgofs, end_offset, end;
1658 int err = 0, ofs = 1;
1659 unsigned int ofs_in_node, last_ofs_in_node;
1660 blkcnt_t prealloc;
1661 block_t blkaddr;
1662 unsigned int start_pgofs;
1663 int bidx = 0;
1664 bool is_hole;
1665 bool lfs_dio_write;
1666
1667 if (!maxblocks)
1668 return 0;
1669
1670 lfs_dio_write = (flag == F2FS_GET_BLOCK_DIO && f2fs_lfs_mode(sbi) &&
1671 map->m_may_create);
1672
1673 if (!map->m_may_create && f2fs_map_blocks_cached(inode, map, flag)) {
1674 struct extent_info ei;
1675
1676 /*
1677 * 1. If map->m_multidev_dio is true, map->m_pblk cannot be
1678 * waitted by f2fs_wait_on_block_writeback_range() and are not
1679 * mergeable.
1680 * 2. If pgofs hits the read extent cache, it means the mapping
1681 * is already cached in the extent cache, but it is not
1682 * mergeable, and there is no need to query the mapping again
1683 * via f2fs_get_dnode_of_data().
1684 */
1685 pgofs = (pgoff_t)map->m_lblk + map->m_len;
1686 if (map->m_len == maxblocks ||
1687 map->m_multidev_dio ||
1688 f2fs_lookup_read_extent_cache(inode, pgofs, &ei))
1689 goto out;
1690 ofs = map->m_len;
1691 goto map_more;
1692 }
1693
1694 map->m_bdev = inode->i_sb->s_bdev;
1695 map->m_multidev_dio =
1696 f2fs_allow_multi_device_dio(F2FS_I_SB(inode), flag);
1697
1698 map->m_len = 0;
1699 map->m_flags = 0;
1700
1701 /* it only supports block size == page size */
1702 pgofs = (pgoff_t)map->m_lblk;
1703 map_more:
1704 end = (pgoff_t)map->m_lblk + maxblocks;
1705
1706 if (flag == F2FS_GET_BLOCK_PRECACHE)
1707 mode = LOOKUP_NODE_RA;
1708
1709 next_dnode:
1710 if (map->m_may_create) {
1711 if (f2fs_lfs_mode(sbi))
1712 f2fs_balance_fs(sbi, true);
1713 f2fs_map_lock(sbi, &lc, flag);
1714 }
1715
1716 /* When reading holes, we need its node page */
1717 set_new_dnode(&dn, inode, NULL, NULL, 0);
1718 err = f2fs_get_dnode_of_data(&dn, pgofs, mode);
1719 if (err) {
1720 if (flag == F2FS_GET_BLOCK_BMAP)
1721 map->m_pblk = 0;
1722 if (err == -ENOENT)
1723 err = f2fs_map_no_dnode(inode, map, &dn, pgofs);
1724 goto unlock_out;
1725 }
1726
1727 start_pgofs = pgofs;
1728 prealloc = 0;
1729 last_ofs_in_node = ofs_in_node = dn.ofs_in_node;
1730 end_offset = ADDRS_PER_PAGE(dn.node_folio, inode);
1731
1732 next_block:
1733 blkaddr = f2fs_data_blkaddr(&dn);
1734 is_hole = !__is_valid_data_blkaddr(blkaddr);
1735 if (!is_hole &&
1736 !f2fs_is_valid_blkaddr(sbi, blkaddr, DATA_GENERIC_ENHANCE)) {
1737 err = -EFSCORRUPTED;
1738 goto sync_out;
1739 }
1740
1741 /* use out-place-update for direct IO under LFS mode */
1742 if (map->m_may_create && (is_hole ||
1743 (flag == F2FS_GET_BLOCK_DIO && f2fs_lfs_mode(sbi) &&
1744 !f2fs_is_pinned_file(inode) && map->m_last_pblk != blkaddr))) {
1745 if (unlikely(f2fs_cp_error(sbi))) {
1746 err = -EIO;
1747 goto sync_out;
1748 }
1749
1750 switch (flag) {
1751 case F2FS_GET_BLOCK_PRE_AIO:
1752 if (blkaddr == NULL_ADDR) {
1753 prealloc++;
1754 last_ofs_in_node = dn.ofs_in_node;
1755 }
1756 break;
1757 case F2FS_GET_BLOCK_PRE_DIO:
1758 case F2FS_GET_BLOCK_DIO:
1759 err = __allocate_data_block(&dn, map->m_seg_type);
1760 if (err)
1761 goto sync_out;
1762 if (flag == F2FS_GET_BLOCK_PRE_DIO)
1763 file_need_truncate(inode);
1764 set_inode_flag(inode, FI_APPEND_WRITE);
1765 break;
1766 default:
1767 WARN_ON_ONCE(1);
1768 err = -EIO;
1769 goto sync_out;
1770 }
1771
1772 blkaddr = dn.data_blkaddr;
1773 if (is_hole)
1774 map->m_flags |= F2FS_MAP_NEW;
1775 } else if (is_hole) {
1776 if (f2fs_compressed_file(inode) &&
1777 f2fs_sanity_check_cluster(&dn)) {
1778 err = -EFSCORRUPTED;
1779 f2fs_handle_error(sbi,
1780 ERROR_CORRUPTED_CLUSTER);
1781 fserror_report_file_metadata(inode, err, GFP_NOFS);
1782 goto sync_out;
1783 }
1784
1785 switch (flag) {
1786 case F2FS_GET_BLOCK_PRECACHE:
1787 goto sync_out;
1788 case F2FS_GET_BLOCK_BMAP:
1789 map->m_pblk = 0;
1790 goto sync_out;
1791 case F2FS_GET_BLOCK_FIEMAP:
1792 if (blkaddr == NULL_ADDR) {
1793 if (map->m_next_pgofs)
1794 *map->m_next_pgofs = pgofs + 1;
1795 goto sync_out;
1796 }
1797 break;
1798 case F2FS_GET_BLOCK_DIO:
1799 if (map->m_next_pgofs)
1800 *map->m_next_pgofs = pgofs + 1;
1801 break;
1802 default:
1803 /* for defragment case */
1804 if (map->m_next_pgofs)
1805 *map->m_next_pgofs = pgofs + 1;
1806 goto sync_out;
1807 }
1808 }
1809
1810 if (flag == F2FS_GET_BLOCK_PRE_AIO)
1811 goto skip;
1812
1813 if (map->m_multidev_dio)
1814 bidx = f2fs_target_device_index(sbi, blkaddr);
1815
1816 if (map->m_len == 0) {
1817 /* reserved delalloc block should be mapped for fiemap. */
1818 if (blkaddr == NEW_ADDR)
1819 map->m_flags |= F2FS_MAP_DELALLOC;
1820 /* DIO READ and hole case, should not map the blocks. */
1821 if (!(flag == F2FS_GET_BLOCK_DIO && is_hole && !map->m_may_create))
1822 map->m_flags |= F2FS_MAP_MAPPED;
1823
1824 map->m_pblk = blkaddr;
1825 map->m_len = 1;
1826
1827 if (map->m_multidev_dio)
1828 map->m_bdev = FDEV(bidx).bdev;
1829
1830 if (lfs_dio_write)
1831 map->m_last_pblk = NULL_ADDR;
1832 } else if (map_is_mergeable(sbi, map, blkaddr, flag, bidx, ofs)) {
1833 ofs++;
1834 map->m_len++;
1835 } else {
1836 if (lfs_dio_write && !f2fs_is_pinned_file(inode))
1837 map->m_last_pblk = blkaddr;
1838 goto sync_out;
1839 }
1840
1841 skip:
1842 dn.ofs_in_node++;
1843 pgofs++;
1844
1845 /* preallocate blocks in batch for one dnode page */
1846 if (flag == F2FS_GET_BLOCK_PRE_AIO &&
1847 (pgofs == end || dn.ofs_in_node == end_offset)) {
1848
1849 dn.ofs_in_node = ofs_in_node;
1850 err = f2fs_reserve_new_blocks(&dn, prealloc);
1851 if (err)
1852 goto sync_out;
1853
1854 map->m_len += dn.ofs_in_node - ofs_in_node;
1855 if (prealloc && dn.ofs_in_node != last_ofs_in_node + 1) {
1856 err = -ENOSPC;
1857 goto sync_out;
1858 }
1859 dn.ofs_in_node = end_offset;
1860 }
1861
1862 if (pgofs >= end)
1863 goto sync_out;
1864 else if (dn.ofs_in_node < end_offset)
1865 goto next_block;
1866
1867 if (flag == F2FS_GET_BLOCK_PRECACHE) {
1868 if (map->m_flags & F2FS_MAP_MAPPED) {
1869 unsigned int ofs = start_pgofs - map->m_lblk;
1870
1871 f2fs_update_read_extent_cache_range(&dn,
1872 start_pgofs, map->m_pblk + ofs,
1873 map->m_len - ofs);
1874 }
1875 }
1876
1877 f2fs_put_dnode(&dn);
1878
1879 if (map->m_may_create) {
1880 f2fs_map_unlock(sbi, &lc, flag);
1881 f2fs_balance_fs(sbi, dn.node_changed);
1882 }
1883 goto next_dnode;
1884
1885 sync_out:
1886
1887 if (flag == F2FS_GET_BLOCK_DIO && map->m_flags & F2FS_MAP_MAPPED) {
1888 /*
1889 * for hardware encryption, but to avoid potential issue
1890 * in future
1891 */
1892 f2fs_wait_on_block_writeback_range(inode,
1893 map->m_pblk, map->m_len);
1894
1895 if (map->m_multidev_dio) {
1896 block_t blk_addr = map->m_pblk;
1897
1898 bidx = f2fs_target_device_index(sbi, map->m_pblk);
1899
1900 map->m_bdev = FDEV(bidx).bdev;
1901 map->m_pblk -= FDEV(bidx).start_blk;
1902
1903 if (map->m_may_create)
1904 f2fs_update_device_state(sbi, inode->i_ino,
1905 blk_addr, map->m_len);
1906
1907 f2fs_bug_on(sbi, blk_addr + map->m_len >
1908 FDEV(bidx).end_blk + 1);
1909 }
1910 }
1911
1912 if (flag == F2FS_GET_BLOCK_PRECACHE) {
1913 if (map->m_flags & F2FS_MAP_MAPPED) {
1914 unsigned int ofs = start_pgofs - map->m_lblk;
1915
1916 if (map->m_len > ofs)
1917 f2fs_update_read_extent_cache_range(&dn,
1918 start_pgofs, map->m_pblk + ofs,
1919 map->m_len - ofs);
1920 }
1921 if (map->m_next_extent)
1922 *map->m_next_extent = is_hole ? pgofs + 1 : pgofs;
1923 }
1924 f2fs_put_dnode(&dn);
1925 unlock_out:
1926 if (map->m_may_create) {
1927 f2fs_map_unlock(sbi, &lc, flag);
1928 f2fs_balance_fs(sbi, dn.node_changed);
1929 }
1930 out:
1931 trace_f2fs_map_blocks(inode, map, flag, err);
1932 return err;
1933 }
1934
__f2fs_overwrite_io(struct inode * inode,loff_t pos,size_t len,bool check_first)1935 static bool __f2fs_overwrite_io(struct inode *inode, loff_t pos, size_t len,
1936 bool check_first)
1937 {
1938 struct f2fs_map_blocks map;
1939 block_t last_lblk;
1940 int err;
1941
1942 if (pos + len > i_size_read(inode))
1943 return false;
1944
1945 map.m_lblk = F2FS_BYTES_TO_BLK(pos);
1946 map.m_next_pgofs = NULL;
1947 map.m_next_extent = NULL;
1948 map.m_seg_type = NO_CHECK_TYPE;
1949 map.m_may_create = false;
1950 last_lblk = F2FS_BLK_ALIGN(pos + len);
1951
1952 while (map.m_lblk < last_lblk) {
1953 map.m_len = last_lblk - map.m_lblk;
1954 err = f2fs_map_blocks(inode, &map, F2FS_GET_BLOCK_DEFAULT);
1955 if (err || map.m_len == 0)
1956 return false;
1957 map.m_lblk += map.m_len;
1958 if (check_first)
1959 break;
1960 }
1961 return true;
1962 }
1963
f2fs_overwrite_io(struct inode * inode,loff_t pos,size_t len)1964 bool f2fs_overwrite_io(struct inode *inode, loff_t pos, size_t len)
1965 {
1966 return __f2fs_overwrite_io(inode, pos, len, false);
1967 }
1968
f2fs_xattr_fiemap(struct inode * inode,struct fiemap_extent_info * fieinfo)1969 static int f2fs_xattr_fiemap(struct inode *inode,
1970 struct fiemap_extent_info *fieinfo)
1971 {
1972 struct f2fs_sb_info *sbi = F2FS_I_SB(inode);
1973 struct node_info ni;
1974 __u64 phys = 0, len;
1975 __u32 flags;
1976 nid_t xnid = F2FS_I(inode)->i_xattr_nid;
1977 int err = 0;
1978
1979 if (f2fs_has_inline_xattr(inode)) {
1980 int offset;
1981 struct folio *folio = f2fs_grab_cache_folio(NODE_MAPPING(sbi),
1982 inode->i_ino, false);
1983
1984 if (IS_ERR(folio))
1985 return PTR_ERR(folio);
1986
1987 err = f2fs_get_node_info(sbi, inode->i_ino, &ni, false);
1988 if (err) {
1989 f2fs_folio_put(folio, true);
1990 return err;
1991 }
1992
1993 phys = F2FS_BLK_TO_BYTES(ni.blk_addr);
1994 offset = offsetof(struct f2fs_inode, i_addr) +
1995 sizeof(__le32) * (DEF_ADDRS_PER_INODE -
1996 get_inline_xattr_addrs(inode));
1997
1998 phys += offset;
1999 len = inline_xattr_size(inode);
2000
2001 f2fs_folio_put(folio, true);
2002
2003 flags = FIEMAP_EXTENT_DATA_INLINE | FIEMAP_EXTENT_NOT_ALIGNED;
2004
2005 if (!xnid)
2006 flags |= FIEMAP_EXTENT_LAST;
2007
2008 err = fiemap_fill_next_extent(fieinfo, 0, phys, len, flags);
2009 trace_f2fs_fiemap(inode, 0, phys, len, flags, err);
2010 if (err)
2011 return err;
2012 }
2013
2014 if (xnid) {
2015 struct folio *folio = f2fs_grab_cache_folio(NODE_MAPPING(sbi),
2016 xnid, false);
2017
2018 if (IS_ERR(folio))
2019 return PTR_ERR(folio);
2020
2021 err = f2fs_get_node_info(sbi, xnid, &ni, false);
2022 if (err) {
2023 f2fs_folio_put(folio, true);
2024 return err;
2025 }
2026
2027 phys = F2FS_BLK_TO_BYTES(ni.blk_addr);
2028 len = inode->i_sb->s_blocksize;
2029
2030 f2fs_folio_put(folio, true);
2031
2032 flags = FIEMAP_EXTENT_LAST;
2033 }
2034
2035 if (phys) {
2036 err = fiemap_fill_next_extent(fieinfo, 0, phys, len, flags);
2037 trace_f2fs_fiemap(inode, 0, phys, len, flags, err);
2038 }
2039
2040 return (err < 0 ? err : 0);
2041 }
2042
f2fs_fiemap(struct inode * inode,struct fiemap_extent_info * fieinfo,u64 start,u64 len)2043 int f2fs_fiemap(struct inode *inode, struct fiemap_extent_info *fieinfo,
2044 u64 start, u64 len)
2045 {
2046 struct f2fs_map_blocks map;
2047 sector_t start_blk, last_blk, blk_len, max_len;
2048 pgoff_t next_pgofs;
2049 u64 logical = 0, phys = 0, size = 0;
2050 u32 flags = 0;
2051 int ret = 0;
2052 bool compr_cluster = false, compr_appended;
2053 unsigned int cluster_size = F2FS_I(inode)->i_cluster_size;
2054 unsigned int count_in_cluster = 0;
2055 loff_t maxbytes;
2056
2057 if (fieinfo->fi_flags & FIEMAP_FLAG_CACHE) {
2058 ret = f2fs_precache_extents(inode);
2059 if (ret)
2060 return ret;
2061 }
2062
2063 ret = fiemap_prep(inode, fieinfo, start, &len, FIEMAP_FLAG_XATTR);
2064 if (ret)
2065 return ret;
2066
2067 inode_lock_shared(inode);
2068
2069 maxbytes = F2FS_BLK_TO_BYTES(max_file_blocks(inode));
2070 if (start > maxbytes) {
2071 ret = -EFBIG;
2072 goto out;
2073 }
2074
2075 if (len > maxbytes || (maxbytes - len) < start)
2076 len = maxbytes - start;
2077
2078 if (fieinfo->fi_flags & FIEMAP_FLAG_XATTR) {
2079 ret = f2fs_xattr_fiemap(inode, fieinfo);
2080 goto out;
2081 }
2082
2083 if (f2fs_has_inline_data(inode) || f2fs_has_inline_dentry(inode)) {
2084 ret = f2fs_inline_data_fiemap(inode, fieinfo, start, len);
2085 if (ret != -EAGAIN)
2086 goto out;
2087 }
2088
2089 start_blk = F2FS_BYTES_TO_BLK(start);
2090 last_blk = F2FS_BYTES_TO_BLK(start + len - 1);
2091 blk_len = last_blk - start_blk + 1;
2092 max_len = F2FS_BYTES_TO_BLK(maxbytes) - start_blk;
2093
2094 next:
2095 memset(&map, 0, sizeof(map));
2096 map.m_lblk = start_blk;
2097 map.m_len = blk_len;
2098 map.m_next_pgofs = &next_pgofs;
2099 map.m_seg_type = NO_CHECK_TYPE;
2100
2101 if (compr_cluster) {
2102 map.m_lblk += 1;
2103 map.m_len = cluster_size - count_in_cluster;
2104 }
2105
2106 ret = f2fs_map_blocks(inode, &map, F2FS_GET_BLOCK_FIEMAP);
2107 if (ret)
2108 goto out;
2109
2110 /* HOLE */
2111 if (!compr_cluster && !(map.m_flags & F2FS_MAP_FLAGS)) {
2112 start_blk = next_pgofs;
2113
2114 if (F2FS_BLK_TO_BYTES(start_blk) < maxbytes)
2115 goto prep_next;
2116
2117 flags |= FIEMAP_EXTENT_LAST;
2118 }
2119
2120 /*
2121 * current extent may cross boundary of inquiry, increase len to
2122 * requery.
2123 */
2124 if (!compr_cluster && (map.m_flags & F2FS_MAP_MAPPED) &&
2125 map.m_lblk + map.m_len - 1 == last_blk &&
2126 blk_len != max_len) {
2127 blk_len = max_len;
2128 goto next;
2129 }
2130
2131 compr_appended = false;
2132 /* In a case of compressed cluster, append this to the last extent */
2133 if (compr_cluster && ((map.m_flags & F2FS_MAP_DELALLOC) ||
2134 !(map.m_flags & F2FS_MAP_FLAGS))) {
2135 compr_appended = true;
2136 goto skip_fill;
2137 }
2138
2139 if (size) {
2140 flags |= FIEMAP_EXTENT_MERGED;
2141 if (IS_ENCRYPTED(inode))
2142 flags |= FIEMAP_EXTENT_DATA_ENCRYPTED;
2143
2144 ret = fiemap_fill_next_extent(fieinfo, logical,
2145 phys, size, flags);
2146 trace_f2fs_fiemap(inode, logical, phys, size, flags, ret);
2147 if (ret)
2148 goto out;
2149 size = 0;
2150 }
2151
2152 if (start_blk > last_blk)
2153 goto out;
2154
2155 skip_fill:
2156 if (map.m_pblk == COMPRESS_ADDR) {
2157 compr_cluster = true;
2158 count_in_cluster = 1;
2159 } else if (compr_appended) {
2160 unsigned int appended_blks = cluster_size -
2161 count_in_cluster + 1;
2162 size += F2FS_BLK_TO_BYTES(appended_blks);
2163 start_blk += appended_blks;
2164 compr_cluster = false;
2165 } else {
2166 logical = F2FS_BLK_TO_BYTES(start_blk);
2167 phys = __is_valid_data_blkaddr(map.m_pblk) ?
2168 F2FS_BLK_TO_BYTES(map.m_pblk) : 0;
2169 size = F2FS_BLK_TO_BYTES(map.m_len);
2170 flags = 0;
2171
2172 if (compr_cluster) {
2173 flags = FIEMAP_EXTENT_ENCODED;
2174 count_in_cluster += map.m_len;
2175 if (count_in_cluster == cluster_size) {
2176 compr_cluster = false;
2177 size += F2FS_BLKSIZE;
2178 }
2179 } else if (map.m_flags & F2FS_MAP_DELALLOC) {
2180 flags = FIEMAP_EXTENT_UNWRITTEN;
2181 }
2182
2183 start_blk += F2FS_BYTES_TO_BLK(size);
2184 }
2185
2186 prep_next:
2187 cond_resched();
2188 if (fatal_signal_pending(current))
2189 ret = -EINTR;
2190 else
2191 goto next;
2192 out:
2193 if (ret == 1)
2194 ret = 0;
2195
2196 inode_unlock_shared(inode);
2197 return ret;
2198 }
2199
f2fs_readpage_limit(struct inode * inode)2200 static inline loff_t f2fs_readpage_limit(struct inode *inode)
2201 {
2202 if (IS_ENABLED(CONFIG_FS_VERITY) && IS_VERITY(inode))
2203 return F2FS_BLK_TO_BYTES(max_file_blocks(inode));
2204
2205 return i_size_read(inode);
2206 }
2207
f2fs_ra_op_flags(struct readahead_control * rac)2208 static inline blk_opf_t f2fs_ra_op_flags(struct readahead_control *rac)
2209 {
2210 return rac ? REQ_RAHEAD : 0;
2211 }
2212
f2fs_read_single_page(struct inode * inode,struct fsverity_info * vi,struct folio * folio,unsigned int nr_pages,struct f2fs_map_blocks * map,struct bio ** bio_ret,sector_t * last_block_in_bio,struct readahead_control * rac)2213 static int f2fs_read_single_page(struct inode *inode, struct fsverity_info *vi,
2214 struct folio *folio, unsigned int nr_pages,
2215 struct f2fs_map_blocks *map,
2216 struct bio **bio_ret,
2217 sector_t *last_block_in_bio,
2218 struct readahead_control *rac)
2219 {
2220 struct bio *bio = *bio_ret;
2221 const unsigned int blocksize = F2FS_BLKSIZE;
2222 sector_t block_in_file;
2223 sector_t last_block;
2224 sector_t last_block_in_file;
2225 sector_t block_nr;
2226 pgoff_t index = folio->index;
2227 int ret = 0;
2228
2229 block_in_file = (sector_t)index;
2230 last_block = block_in_file + nr_pages;
2231 last_block_in_file = F2FS_BYTES_TO_BLK(f2fs_readpage_limit(inode) +
2232 blocksize - 1);
2233 if (last_block > last_block_in_file)
2234 last_block = last_block_in_file;
2235
2236 /* just zeroing out page which is beyond EOF */
2237 if (block_in_file >= last_block)
2238 goto zero_out;
2239 /*
2240 * Map blocks using the previous result first.
2241 */
2242 if (map->m_flags & F2FS_MAP_MAPPED) {
2243 if (block_in_file > map->m_lblk &&
2244 block_in_file < (map->m_lblk + map->m_len))
2245 goto got_it;
2246 } else if (block_in_file < *map->m_next_pgofs) {
2247 goto got_it;
2248 }
2249
2250 /*
2251 * Then do more f2fs_map_blocks() calls until we are
2252 * done with this page.
2253 */
2254 map->m_lblk = block_in_file;
2255 map->m_len = last_block - block_in_file;
2256
2257 ret = f2fs_map_blocks(inode, map, F2FS_GET_BLOCK_DEFAULT);
2258 if (ret)
2259 goto out;
2260 got_it:
2261 if ((map->m_flags & F2FS_MAP_MAPPED)) {
2262 block_nr = map->m_pblk + block_in_file - map->m_lblk;
2263 folio_set_mappedtodisk(folio);
2264
2265 if (!f2fs_is_valid_blkaddr(F2FS_I_SB(inode), block_nr,
2266 DATA_GENERIC_ENHANCE_READ)) {
2267 ret = -EFSCORRUPTED;
2268 goto out;
2269 }
2270 } else {
2271 zero_out:
2272 folio_zero_segment(folio, 0, folio_size(folio));
2273 if (vi && !fsverity_verify_folio(vi, folio)) {
2274 ret = -EIO;
2275 goto out;
2276 }
2277 if (!folio_test_uptodate(folio))
2278 folio_mark_uptodate(folio);
2279 folio_unlock(folio);
2280 goto out;
2281 }
2282
2283 /*
2284 * This page will go to BIO. Do we need to send this
2285 * BIO off first?
2286 */
2287 if (bio && (!page_is_mergeable(F2FS_I_SB(inode), bio,
2288 *last_block_in_bio, block_nr) ||
2289 !f2fs_crypt_mergeable_bio(bio, inode, index, NULL))) {
2290 submit_and_realloc:
2291 f2fs_submit_read_bio(F2FS_I_SB(inode), bio, DATA);
2292 bio = NULL;
2293 }
2294 if (bio == NULL)
2295 bio = f2fs_grab_read_bio(inode, vi, block_nr, nr_pages,
2296 f2fs_ra_op_flags(rac), index, false);
2297
2298 /*
2299 * If the page is under writeback, we need to wait for
2300 * its completion to see the correct decrypted data.
2301 */
2302 f2fs_wait_on_block_writeback(inode, block_nr);
2303
2304 if (!bio_add_folio(bio, folio, blocksize, 0))
2305 goto submit_and_realloc;
2306
2307 inc_page_count(F2FS_I_SB(inode), F2FS_RD_DATA);
2308 f2fs_update_iostat(F2FS_I_SB(inode), NULL, FS_DATA_READ_IO,
2309 F2FS_BLKSIZE);
2310 *last_block_in_bio = block_nr;
2311 out:
2312 *bio_ret = bio;
2313 return ret;
2314 }
2315
2316 #ifdef CONFIG_F2FS_FS_COMPRESSION
f2fs_read_multi_pages(struct compress_ctx * cc,struct bio ** bio_ret,unsigned nr_pages,sector_t * last_block_in_bio,struct readahead_control * rac,bool for_write)2317 int f2fs_read_multi_pages(struct compress_ctx *cc, struct bio **bio_ret,
2318 unsigned nr_pages, sector_t *last_block_in_bio,
2319 struct readahead_control *rac, bool for_write)
2320 {
2321 struct dnode_of_data dn;
2322 struct inode *inode = cc->inode;
2323 struct f2fs_sb_info *sbi = F2FS_I_SB(inode);
2324 struct bio *bio = *bio_ret;
2325 unsigned int start_idx = cc->cluster_idx << cc->log_cluster_size;
2326 sector_t last_block_in_file;
2327 const unsigned int blocksize = F2FS_BLKSIZE;
2328 struct decompress_io_ctx *dic = NULL;
2329 struct extent_info ei = {};
2330 bool from_dnode = true;
2331 int i;
2332 int ret = 0;
2333
2334 if (unlikely(f2fs_cp_error(sbi))) {
2335 ret = -EIO;
2336 from_dnode = false;
2337 goto out_put_dnode;
2338 }
2339
2340 f2fs_bug_on(sbi, f2fs_cluster_is_empty(cc));
2341
2342 last_block_in_file = F2FS_BYTES_TO_BLK(f2fs_readpage_limit(inode) +
2343 blocksize - 1);
2344
2345 /* get rid of pages beyond EOF */
2346 for (i = 0; i < cc->cluster_size; i++) {
2347 struct page *page = cc->rpages[i];
2348 struct folio *folio;
2349
2350 if (!page)
2351 continue;
2352
2353 folio = page_folio(page);
2354 if ((sector_t)folio->index >= last_block_in_file) {
2355 folio_zero_segment(folio, 0, folio_size(folio));
2356 if (!folio_test_uptodate(folio))
2357 folio_mark_uptodate(folio);
2358 } else if (!folio_test_uptodate(folio)) {
2359 continue;
2360 }
2361 folio_unlock(folio);
2362 if (for_write)
2363 folio_put(folio);
2364 cc->rpages[i] = NULL;
2365 cc->nr_rpages--;
2366 }
2367
2368 /* we are done since all pages are beyond EOF */
2369 if (f2fs_cluster_is_empty(cc))
2370 goto out;
2371
2372 if (f2fs_lookup_read_extent_cache(inode, start_idx, &ei))
2373 from_dnode = false;
2374
2375 if (!from_dnode)
2376 goto skip_reading_dnode;
2377
2378 set_new_dnode(&dn, inode, NULL, NULL, 0);
2379 ret = f2fs_get_dnode_of_data(&dn, start_idx, LOOKUP_NODE);
2380 if (ret)
2381 goto out;
2382
2383 f2fs_bug_on(sbi, dn.data_blkaddr != COMPRESS_ADDR);
2384
2385 skip_reading_dnode:
2386 for (i = 1; i < cc->cluster_size; i++) {
2387 block_t blkaddr;
2388
2389 blkaddr = from_dnode ? data_blkaddr(dn.inode, dn.node_folio,
2390 dn.ofs_in_node + i) :
2391 ei.blk + i - 1;
2392
2393 if (!__is_valid_data_blkaddr(blkaddr))
2394 break;
2395
2396 if (!f2fs_is_valid_blkaddr(sbi, blkaddr, DATA_GENERIC)) {
2397 ret = -EFAULT;
2398 goto out_put_dnode;
2399 }
2400 cc->nr_cpages++;
2401
2402 if (!from_dnode && i >= ei.c_len)
2403 break;
2404 }
2405
2406 /* nothing to decompress */
2407 if (cc->nr_cpages == 0) {
2408 ret = 0;
2409 goto out_put_dnode;
2410 }
2411
2412 dic = f2fs_alloc_dic(cc);
2413 if (IS_ERR(dic)) {
2414 ret = PTR_ERR(dic);
2415 goto out_put_dnode;
2416 }
2417
2418 for (i = 0; i < cc->nr_cpages; i++) {
2419 struct folio *folio = page_folio(dic->cpages[i]);
2420 block_t blkaddr;
2421 struct bio_post_read_ctx *ctx;
2422
2423 blkaddr = from_dnode ? data_blkaddr(dn.inode, dn.node_folio,
2424 dn.ofs_in_node + i + 1) :
2425 ei.blk + i;
2426
2427 f2fs_wait_on_block_writeback(inode, blkaddr);
2428
2429 if (f2fs_load_compressed_folio(sbi, folio, blkaddr)) {
2430 if (atomic_dec_and_test(&dic->remaining_pages)) {
2431 f2fs_decompress_cluster(dic, true);
2432 break;
2433 }
2434 continue;
2435 }
2436
2437 if (bio && (!page_is_mergeable(sbi, bio,
2438 *last_block_in_bio, blkaddr) ||
2439 !f2fs_crypt_mergeable_bio(bio, inode, folio->index, NULL))) {
2440 submit_and_realloc:
2441 f2fs_submit_read_bio(sbi, bio, DATA);
2442 bio = NULL;
2443 }
2444
2445 if (!bio)
2446 bio = f2fs_grab_read_bio(inode, cc->vi, blkaddr,
2447 nr_pages - i,
2448 f2fs_ra_op_flags(rac),
2449 folio->index, for_write);
2450
2451 if (!bio_add_folio(bio, folio, blocksize, 0))
2452 goto submit_and_realloc;
2453
2454 ctx = get_post_read_ctx(bio);
2455 ctx->enabled_steps |= STEP_DECOMPRESS;
2456 refcount_inc(&dic->refcnt);
2457
2458 inc_page_count(sbi, F2FS_RD_DATA);
2459 f2fs_update_iostat(sbi, inode, FS_DATA_READ_IO, F2FS_BLKSIZE);
2460 *last_block_in_bio = blkaddr;
2461 }
2462
2463 if (from_dnode)
2464 f2fs_put_dnode(&dn);
2465
2466 *bio_ret = bio;
2467 return 0;
2468
2469 out_put_dnode:
2470 if (from_dnode)
2471 f2fs_put_dnode(&dn);
2472 out:
2473 for (i = 0; i < cc->cluster_size; i++) {
2474 if (cc->rpages[i]) {
2475 ClearPageUptodate(cc->rpages[i]);
2476 unlock_page(cc->rpages[i]);
2477 }
2478 }
2479 *bio_ret = bio;
2480 return ret;
2481 }
2482 #endif
2483
ffs_find_or_alloc(struct folio * folio)2484 static struct f2fs_folio_state *ffs_find_or_alloc(struct folio *folio)
2485 {
2486 struct f2fs_folio_state *ffs = folio->private;
2487
2488 if (ffs)
2489 return ffs;
2490
2491 ffs = f2fs_kmem_cache_alloc(ffs_entry_slab,
2492 GFP_NOIO | __GFP_ZERO, true, NULL);
2493
2494 spin_lock_init(&ffs->state_lock);
2495 folio_attach_private(folio, ffs);
2496 return ffs;
2497 }
2498
ffs_detach_free(struct folio * folio)2499 static void ffs_detach_free(struct folio *folio)
2500 {
2501 struct f2fs_folio_state *ffs;
2502
2503 if (!folio_test_large(folio)) {
2504 folio_detach_private(folio);
2505 return;
2506 }
2507
2508 ffs = folio_detach_private(folio);
2509 if (!ffs)
2510 return;
2511
2512 WARN_ON_ONCE(ffs->read_pages_pending != 0);
2513 kmem_cache_free(ffs_entry_slab, ffs);
2514 }
2515
f2fs_read_data_large_folio(struct inode * inode,struct fsverity_info * vi,struct readahead_control * rac,struct folio * folio)2516 static int f2fs_read_data_large_folio(struct inode *inode,
2517 struct fsverity_info *vi,
2518 struct readahead_control *rac, struct folio *folio)
2519 {
2520 struct bio *bio = NULL;
2521 sector_t last_block_in_bio = 0;
2522 struct f2fs_map_blocks map = {0, };
2523 pgoff_t index, offset, next_pgofs = 0;
2524 unsigned max_nr_pages = rac ? readahead_count(rac) :
2525 folio_nr_pages(folio);
2526 unsigned nrpages;
2527 struct f2fs_folio_state *ffs;
2528 int ret = 0;
2529 bool folio_in_bio = false;
2530
2531 if (!IS_IMMUTABLE(inode) || f2fs_compressed_file(inode)) {
2532 if (folio)
2533 folio_unlock(folio);
2534 return -EOPNOTSUPP;
2535 }
2536
2537 map.m_seg_type = NO_CHECK_TYPE;
2538
2539 if (rac)
2540 folio = readahead_folio(rac);
2541 next_folio:
2542 if (!folio)
2543 goto out;
2544
2545 f2fs_update_read_folio_count(F2FS_I_SB(inode), folio);
2546
2547 folio_in_bio = false;
2548 index = folio->index;
2549 offset = 0;
2550 ffs = NULL;
2551 nrpages = folio_nr_pages(folio);
2552
2553 for (; nrpages; nrpages--, max_nr_pages--, index++, offset++) {
2554 sector_t block_nr;
2555 /*
2556 * Map blocks using the previous result first.
2557 */
2558 if (map.m_flags & F2FS_MAP_MAPPED) {
2559 if (index > map.m_lblk &&
2560 index < (map.m_lblk + map.m_len))
2561 goto got_it;
2562 } else if (index < next_pgofs) {
2563 /* hole case */
2564 goto got_it;
2565 }
2566
2567 /*
2568 * Then do more f2fs_map_blocks() calls until we are
2569 * done with this page.
2570 */
2571 memset(&map, 0, sizeof(map));
2572 map.m_next_pgofs = &next_pgofs;
2573 map.m_seg_type = NO_CHECK_TYPE;
2574 map.m_lblk = index;
2575 map.m_len = max_nr_pages;
2576
2577 ret = f2fs_map_blocks(inode, &map, F2FS_GET_BLOCK_DEFAULT);
2578 if (ret)
2579 goto err_out;
2580 got_it:
2581 if ((map.m_flags & F2FS_MAP_MAPPED)) {
2582 block_nr = map.m_pblk + index - map.m_lblk;
2583 if (!f2fs_is_valid_blkaddr(F2FS_I_SB(inode), block_nr,
2584 DATA_GENERIC_ENHANCE_READ)) {
2585 ret = -EFSCORRUPTED;
2586 goto err_out;
2587 }
2588 } else {
2589 size_t page_offset = offset << PAGE_SHIFT;
2590 folio_zero_range(folio, page_offset, PAGE_SIZE);
2591 if (vi && !fsverity_verify_blocks(vi, folio, PAGE_SIZE, page_offset)) {
2592 ret = -EIO;
2593 goto err_out;
2594 }
2595 continue;
2596 }
2597
2598 /* We must increment read_pages_pending before possible BIOs submitting
2599 * to prevent from premature folio_end_read() call on folio
2600 */
2601 if (folio_test_large(folio)) {
2602 ffs = ffs_find_or_alloc(folio);
2603
2604 /* set the bitmap to wait */
2605 spin_lock_irq(&ffs->state_lock);
2606 ffs->read_pages_pending++;
2607 spin_unlock_irq(&ffs->state_lock);
2608 }
2609
2610 /*
2611 * This page will go to BIO. Do we need to send this
2612 * BIO off first?
2613 */
2614 if (bio && (!page_is_mergeable(F2FS_I_SB(inode), bio,
2615 last_block_in_bio, block_nr) ||
2616 !f2fs_crypt_mergeable_bio(bio, inode, index, NULL))) {
2617 submit_and_realloc:
2618 f2fs_submit_read_bio(F2FS_I_SB(inode), bio, DATA);
2619 bio = NULL;
2620 }
2621 if (bio == NULL)
2622 bio = f2fs_grab_read_bio(inode, vi,
2623 block_nr, max_nr_pages,
2624 f2fs_ra_op_flags(rac),
2625 index, false);
2626
2627 /*
2628 * If the page is under writeback, we need to wait for
2629 * its completion to see the correct decrypted data.
2630 */
2631 f2fs_wait_on_block_writeback(inode, block_nr);
2632
2633 if (!bio_add_folio(bio, folio, F2FS_BLKSIZE,
2634 offset << PAGE_SHIFT))
2635 goto submit_and_realloc;
2636
2637 folio_in_bio = true;
2638 inc_page_count(F2FS_I_SB(inode), F2FS_RD_DATA);
2639 f2fs_update_iostat(F2FS_I_SB(inode), NULL, FS_DATA_READ_IO,
2640 F2FS_BLKSIZE);
2641 last_block_in_bio = block_nr;
2642 }
2643 trace_f2fs_read_folio(folio, DATA);
2644 err_out:
2645 if (!folio_in_bio)
2646 folio_end_read(folio, !ret);
2647 if (ret)
2648 goto out;
2649 if (rac) {
2650 folio = readahead_folio(rac);
2651 goto next_folio;
2652 }
2653 out:
2654 f2fs_submit_read_bio(F2FS_I_SB(inode), bio, DATA);
2655 if (ret && folio_in_bio) {
2656 /* Wait bios and clear uptodate. */
2657 folio_lock(folio);
2658 folio_clear_uptodate(folio);
2659 folio_unlock(folio);
2660 }
2661 return ret;
2662 }
2663
2664 /*
2665 * This function was originally taken from fs/mpage.c, and customized for f2fs.
2666 * Major change was from block_size == page_size in f2fs by default.
2667 */
f2fs_mpage_readpages(struct inode * inode,struct fsverity_info * vi,struct readahead_control * rac,struct folio * folio)2668 static int f2fs_mpage_readpages(struct inode *inode, struct fsverity_info *vi,
2669 struct readahead_control *rac, struct folio *folio)
2670 {
2671 struct bio *bio = NULL;
2672 sector_t last_block_in_bio = 0;
2673 struct f2fs_map_blocks map;
2674 #ifdef CONFIG_F2FS_FS_COMPRESSION
2675 struct compress_ctx cc = {
2676 .inode = inode,
2677 .log_cluster_size = F2FS_I(inode)->i_log_cluster_size,
2678 .cluster_size = F2FS_I(inode)->i_cluster_size,
2679 .cluster_idx = NULL_CLUSTER,
2680 .rpages = NULL,
2681 .cpages = NULL,
2682 .nr_rpages = 0,
2683 .nr_cpages = 0,
2684 };
2685 pgoff_t nc_cluster_idx = NULL_CLUSTER;
2686 pgoff_t index;
2687 #endif
2688 pgoff_t next_pgofs = 0;
2689 unsigned nr_pages = rac ? readahead_count(rac) : 1;
2690 struct address_space *mapping = rac ? rac->mapping : folio->mapping;
2691 unsigned max_nr_pages = nr_pages;
2692 int ret = 0;
2693
2694 if (mapping_large_folio_support(mapping))
2695 return f2fs_read_data_large_folio(inode, vi, rac, folio);
2696
2697 #ifdef CONFIG_F2FS_FS_COMPRESSION
2698 if (f2fs_compressed_file(inode)) {
2699 index = rac ? readahead_index(rac) : folio->index;
2700 max_nr_pages = round_up(index + nr_pages, cc.cluster_size) -
2701 round_down(index, cc.cluster_size);
2702 }
2703 #endif
2704
2705 map.m_pblk = 0;
2706 map.m_lblk = 0;
2707 map.m_len = 0;
2708 map.m_flags = 0;
2709 map.m_next_pgofs = &next_pgofs;
2710 map.m_next_extent = NULL;
2711 map.m_seg_type = NO_CHECK_TYPE;
2712 map.m_may_create = false;
2713
2714 for (; nr_pages; nr_pages--) {
2715 if (rac) {
2716 folio = readahead_folio(rac);
2717 prefetchw(&folio->flags);
2718 }
2719
2720 f2fs_update_read_folio_count(F2FS_I_SB(inode), folio);
2721
2722 #ifdef CONFIG_F2FS_FS_COMPRESSION
2723 index = folio->index;
2724
2725 if (!f2fs_compressed_file(inode))
2726 goto read_single_page;
2727
2728 /* there are remained compressed pages, submit them */
2729 if (!f2fs_cluster_can_merge_page(&cc, index)) {
2730 cc.vi = vi;
2731 ret = f2fs_read_multi_pages(&cc, &bio,
2732 max_nr_pages,
2733 &last_block_in_bio,
2734 rac, false);
2735 f2fs_destroy_compress_ctx(&cc, false);
2736 if (ret)
2737 goto set_error_page;
2738 }
2739 if (cc.cluster_idx == NULL_CLUSTER) {
2740 if (nc_cluster_idx == index >> cc.log_cluster_size)
2741 goto read_single_page;
2742
2743 ret = f2fs_is_compressed_cluster(inode, index);
2744 if (ret < 0)
2745 goto set_error_page;
2746 else if (!ret) {
2747 nc_cluster_idx =
2748 index >> cc.log_cluster_size;
2749 goto read_single_page;
2750 }
2751
2752 nc_cluster_idx = NULL_CLUSTER;
2753 }
2754 ret = f2fs_init_compress_ctx(&cc);
2755 if (ret)
2756 goto set_error_page;
2757
2758 f2fs_compress_ctx_add_page(&cc, folio);
2759
2760 goto next_page;
2761 read_single_page:
2762 #endif
2763
2764 ret = f2fs_read_single_page(inode, vi, folio, max_nr_pages,
2765 &map, &bio, &last_block_in_bio,
2766 rac);
2767 if (ret) {
2768 #ifdef CONFIG_F2FS_FS_COMPRESSION
2769 set_error_page:
2770 #endif
2771 folio_zero_segment(folio, 0, folio_size(folio));
2772 folio_unlock(folio);
2773 }
2774 #ifdef CONFIG_F2FS_FS_COMPRESSION
2775 next_page:
2776 #endif
2777
2778 #ifdef CONFIG_F2FS_FS_COMPRESSION
2779 if (f2fs_compressed_file(inode)) {
2780 /* last page */
2781 if (nr_pages == 1 && !f2fs_cluster_is_empty(&cc)) {
2782 cc.vi = vi;
2783 ret = f2fs_read_multi_pages(&cc, &bio,
2784 max_nr_pages,
2785 &last_block_in_bio,
2786 rac, false);
2787 f2fs_destroy_compress_ctx(&cc, false);
2788 }
2789 }
2790 #endif
2791 }
2792 f2fs_submit_read_bio(F2FS_I_SB(inode), bio, DATA);
2793 return ret;
2794 }
2795
f2fs_read_data_folio(struct file * file,struct folio * folio)2796 static int f2fs_read_data_folio(struct file *file, struct folio *folio)
2797 {
2798 struct inode *inode = folio->mapping->host;
2799 struct fsverity_info *vi = NULL;
2800 int ret;
2801
2802 trace_f2fs_readpage(folio, DATA);
2803
2804 if (!f2fs_is_compress_backend_ready(inode)) {
2805 folio_unlock(folio);
2806 return -EOPNOTSUPP;
2807 }
2808
2809 /* If the file has inline data, try to read it directly */
2810 if (f2fs_has_inline_data(inode)) {
2811 ret = f2fs_read_inline_data(inode, folio);
2812 if (ret != -EAGAIN)
2813 return ret;
2814 }
2815
2816 vi = f2fs_need_verity(inode, folio->index);
2817 if (vi)
2818 fsverity_readahead(vi, folio->index, folio_nr_pages(folio));
2819 return f2fs_mpage_readpages(inode, vi, NULL, folio);
2820 }
2821
f2fs_readahead(struct readahead_control * rac)2822 static void f2fs_readahead(struct readahead_control *rac)
2823 {
2824 struct inode *inode = rac->mapping->host;
2825 struct fsverity_info *vi = NULL;
2826
2827 trace_f2fs_readpages(inode, readahead_index(rac), readahead_count(rac));
2828
2829 if (!f2fs_is_compress_backend_ready(inode))
2830 return;
2831
2832 /* If the file has inline data, skip readahead */
2833 if (f2fs_has_inline_data(inode))
2834 return;
2835
2836 vi = f2fs_need_verity(inode, readahead_index(rac));
2837 if (vi)
2838 fsverity_readahead(vi, readahead_index(rac),
2839 readahead_count(rac));
2840 f2fs_mpage_readpages(inode, vi, rac, NULL);
2841 }
2842
check_inplace_update_policy(struct inode * inode,struct f2fs_io_info * fio)2843 static inline bool check_inplace_update_policy(struct inode *inode,
2844 struct f2fs_io_info *fio)
2845 {
2846 struct f2fs_sb_info *sbi = F2FS_I_SB(inode);
2847
2848 if (IS_F2FS_IPU_HONOR_OPU_WRITE(sbi) &&
2849 is_inode_flag_set(inode, FI_OPU_WRITE))
2850 return false;
2851 if (IS_F2FS_IPU_FORCE(sbi))
2852 return true;
2853 if (IS_F2FS_IPU_SSR(sbi) && f2fs_need_SSR(sbi))
2854 return true;
2855 if (IS_F2FS_IPU_UTIL(sbi) && utilization(sbi) > SM_I(sbi)->min_ipu_util)
2856 return true;
2857 if (IS_F2FS_IPU_SSR_UTIL(sbi) && f2fs_need_SSR(sbi) &&
2858 utilization(sbi) > SM_I(sbi)->min_ipu_util)
2859 return true;
2860
2861 /*
2862 * IPU for rewrite async pages
2863 */
2864 if (IS_F2FS_IPU_ASYNC(sbi) && fio && fio->op == REQ_OP_WRITE &&
2865 !(fio->op_flags & REQ_SYNC) && !IS_ENCRYPTED(inode))
2866 return true;
2867
2868 /* this is only set during fdatasync */
2869 if (IS_F2FS_IPU_FSYNC(sbi) && is_inode_flag_set(inode, FI_NEED_IPU))
2870 return true;
2871
2872 if (unlikely(fio && is_sbi_flag_set(sbi, SBI_CP_DISABLED) &&
2873 !f2fs_is_checkpointed_data(sbi, fio->old_blkaddr)))
2874 return true;
2875
2876 return false;
2877 }
2878
f2fs_should_update_inplace(struct inode * inode,struct f2fs_io_info * fio)2879 bool f2fs_should_update_inplace(struct inode *inode, struct f2fs_io_info *fio)
2880 {
2881 /* swap file is migrating in aligned write mode */
2882 if (is_inode_flag_set(inode, FI_ALIGNED_WRITE))
2883 return false;
2884
2885 if (f2fs_is_pinned_file(inode))
2886 return true;
2887
2888 /* if this is cold file, we should overwrite to avoid fragmentation */
2889 if (file_is_cold(inode) && !is_inode_flag_set(inode, FI_OPU_WRITE))
2890 return true;
2891
2892 return check_inplace_update_policy(inode, fio);
2893 }
2894
f2fs_should_update_outplace(struct inode * inode,struct f2fs_io_info * fio)2895 bool f2fs_should_update_outplace(struct inode *inode, struct f2fs_io_info *fio)
2896 {
2897 struct f2fs_sb_info *sbi = F2FS_I_SB(inode);
2898
2899 /* The below cases were checked when setting it. */
2900 if (f2fs_is_pinned_file(inode))
2901 return false;
2902 if (fio && is_sbi_flag_set(sbi, SBI_NEED_FSCK))
2903 return true;
2904 if (f2fs_lfs_mode(sbi))
2905 return true;
2906 if (S_ISDIR(inode->i_mode))
2907 return true;
2908 if (IS_NOQUOTA(inode))
2909 return true;
2910 if (f2fs_used_in_atomic_write(inode))
2911 return true;
2912 /* rewrite low ratio compress data w/ OPU mode to avoid fragmentation */
2913 if (f2fs_compressed_file(inode) &&
2914 F2FS_OPTION(sbi).compress_mode == COMPR_MODE_USER &&
2915 is_inode_flag_set(inode, FI_ENABLE_COMPRESS))
2916 return true;
2917
2918 /* swap file is migrating in aligned write mode */
2919 if (is_inode_flag_set(inode, FI_ALIGNED_WRITE))
2920 return true;
2921
2922 if (is_inode_flag_set(inode, FI_OPU_WRITE))
2923 return true;
2924
2925 if (fio) {
2926 if (page_private_gcing(fio->page))
2927 return true;
2928 if (unlikely(is_sbi_flag_set(sbi, SBI_CP_DISABLED) &&
2929 f2fs_is_checkpointed_data(sbi, fio->old_blkaddr)))
2930 return true;
2931 }
2932 return false;
2933 }
2934
need_inplace_update(struct f2fs_io_info * fio)2935 static inline bool need_inplace_update(struct f2fs_io_info *fio)
2936 {
2937 struct inode *inode = fio_inode(fio);
2938
2939 if (f2fs_should_update_outplace(inode, fio))
2940 return false;
2941
2942 return f2fs_should_update_inplace(inode, fio);
2943 }
2944
f2fs_do_write_data_page(struct f2fs_io_info * fio)2945 int f2fs_do_write_data_page(struct f2fs_io_info *fio)
2946 {
2947 struct folio *folio = fio->folio;
2948 struct inode *inode = folio->mapping->host;
2949 struct dnode_of_data dn;
2950 struct node_info ni;
2951 struct f2fs_lock_context lc;
2952 bool ipu_force = false;
2953 bool atomic_commit;
2954 int err = 0;
2955
2956 /* Use COW inode to make dnode_of_data for atomic write */
2957 atomic_commit = f2fs_is_atomic_file(inode) &&
2958 folio_test_f2fs_atomic(folio);
2959 if (atomic_commit)
2960 set_new_dnode(&dn, F2FS_I(inode)->cow_inode, NULL, NULL, 0);
2961 else
2962 set_new_dnode(&dn, inode, NULL, NULL, 0);
2963
2964 if (need_inplace_update(fio) &&
2965 f2fs_lookup_read_extent_cache_block(inode, folio->index,
2966 &fio->old_blkaddr)) {
2967 if (!f2fs_is_valid_blkaddr(fio->sbi, fio->old_blkaddr,
2968 DATA_GENERIC_ENHANCE))
2969 return -EFSCORRUPTED;
2970
2971 ipu_force = true;
2972 fio->need_lock = LOCK_DONE;
2973 goto got_it;
2974 }
2975
2976 if (is_sbi_flag_set(fio->sbi, SBI_ENABLE_CHECKPOINT) &&
2977 time_to_inject(fio->sbi, FAULT_SKIP_WRITE))
2978 return -EINVAL;
2979
2980 /* Deadlock due to between page->lock and f2fs_lock_op */
2981 if (fio->need_lock == LOCK_REQ && !f2fs_trylock_op(fio->sbi, &lc))
2982 return -EAGAIN;
2983
2984 err = f2fs_get_dnode_of_data(&dn, folio->index, LOOKUP_NODE);
2985 if (err)
2986 goto out;
2987
2988 fio->old_blkaddr = dn.data_blkaddr;
2989
2990 /* This page is already truncated */
2991 if (fio->old_blkaddr == NULL_ADDR) {
2992 folio_clear_uptodate(folio);
2993 folio_clear_f2fs_gcing(folio);
2994 goto out_writepage;
2995 }
2996 got_it:
2997 if (__is_valid_data_blkaddr(fio->old_blkaddr) &&
2998 !f2fs_is_valid_blkaddr(fio->sbi, fio->old_blkaddr,
2999 DATA_GENERIC_ENHANCE)) {
3000 err = -EFSCORRUPTED;
3001 goto out_writepage;
3002 }
3003
3004 /* wait for GCed page writeback via META_MAPPING */
3005 if (fio->meta_gc)
3006 f2fs_wait_on_block_writeback(inode, fio->old_blkaddr);
3007
3008 /*
3009 * If current allocation needs SSR,
3010 * it had better in-place writes for updated data.
3011 */
3012 if (ipu_force ||
3013 (__is_valid_data_blkaddr(fio->old_blkaddr) &&
3014 need_inplace_update(fio))) {
3015 folio_start_writeback(folio);
3016 f2fs_put_dnode(&dn);
3017 if (fio->need_lock == LOCK_REQ)
3018 f2fs_unlock_op(fio->sbi, &lc);
3019 err = f2fs_inplace_write_data(fio);
3020 if (err)
3021 folio_end_writeback(folio);
3022 else
3023 set_inode_flag(inode, FI_UPDATE_WRITE);
3024 trace_f2fs_do_write_data_page(folio, IPU);
3025 return err;
3026 }
3027
3028 if (fio->need_lock == LOCK_RETRY) {
3029 if (!f2fs_trylock_op(fio->sbi, &lc)) {
3030 err = -EAGAIN;
3031 goto out_writepage;
3032 }
3033 fio->need_lock = LOCK_REQ;
3034 }
3035
3036 err = f2fs_get_node_info(fio->sbi, dn.nid, &ni, false);
3037 if (err)
3038 goto out_writepage;
3039
3040 fio->version = ni.version;
3041
3042 folio_start_writeback(folio);
3043
3044 if (fio->compr_blocks && fio->old_blkaddr == COMPRESS_ADDR)
3045 f2fs_i_compr_blocks_update(inode, fio->compr_blocks - 1, false);
3046
3047 /* LFS mode write path */
3048 f2fs_outplace_write_data(&dn, fio);
3049 trace_f2fs_do_write_data_page(folio, OPU);
3050 set_inode_flag(inode, FI_APPEND_WRITE);
3051 if (atomic_commit)
3052 folio_clear_f2fs_atomic(folio);
3053 out_writepage:
3054 f2fs_put_dnode(&dn);
3055 out:
3056 if (fio->need_lock == LOCK_REQ)
3057 f2fs_unlock_op(fio->sbi, &lc);
3058 return err;
3059 }
3060
f2fs_write_single_data_page(struct folio * folio,int * submitted,struct bio ** bio,sector_t * last_block,struct writeback_control * wbc,enum iostat_type io_type,int compr_blocks,bool allow_balance)3061 int f2fs_write_single_data_page(struct folio *folio, int *submitted,
3062 struct bio **bio,
3063 sector_t *last_block,
3064 struct writeback_control *wbc,
3065 enum iostat_type io_type,
3066 int compr_blocks,
3067 bool allow_balance)
3068 {
3069 struct inode *inode = folio->mapping->host;
3070 struct f2fs_sb_info *sbi = F2FS_I_SB(inode);
3071 loff_t i_size = i_size_read(inode);
3072 const pgoff_t end_index = ((unsigned long long)i_size)
3073 >> PAGE_SHIFT;
3074 loff_t psize = (loff_t)(folio->index + 1) << PAGE_SHIFT;
3075 unsigned offset = 0;
3076 bool need_balance_fs = false;
3077 bool quota_inode = IS_NOQUOTA(inode);
3078 int err = 0;
3079 struct f2fs_io_info fio = {
3080 .sbi = sbi,
3081 .ino = inode->i_ino,
3082 .type = DATA,
3083 .op = REQ_OP_WRITE,
3084 .op_flags = wbc_to_write_flags(wbc),
3085 .old_blkaddr = NULL_ADDR,
3086 .folio = folio,
3087 .encrypted_page = NULL,
3088 .submitted = 0,
3089 .compr_blocks = compr_blocks,
3090 .need_lock = compr_blocks ? LOCK_DONE : LOCK_RETRY,
3091 .meta_gc = f2fs_meta_inode_gc_required(inode) ? 1 : 0,
3092 .io_type = io_type,
3093 .io_wbc = wbc,
3094 .bio = bio,
3095 .last_block = last_block,
3096 };
3097
3098 trace_f2fs_writepage(folio, DATA);
3099
3100 /* we should bypass data pages to proceed the kworker jobs */
3101 if (unlikely(f2fs_cp_error(sbi))) {
3102 mapping_set_error(folio->mapping, -EIO);
3103 /*
3104 * don't drop any dirty dentry pages for keeping lastest
3105 * directory structure.
3106 */
3107 if (S_ISDIR(inode->i_mode) &&
3108 !is_sbi_flag_set(sbi, SBI_IS_CLOSE))
3109 goto redirty_out;
3110
3111 /* keep data pages in remount-ro mode */
3112 if (F2FS_OPTION(sbi).errors == MOUNT_ERRORS_READONLY)
3113 goto redirty_out;
3114 goto out;
3115 }
3116
3117 if (unlikely(is_sbi_flag_set(sbi, SBI_POR_DOING)))
3118 goto redirty_out;
3119
3120 if (folio->index < end_index ||
3121 f2fs_verity_in_progress(inode) ||
3122 compr_blocks)
3123 goto write;
3124
3125 /*
3126 * If the offset is out-of-range of file size,
3127 * this page does not have to be written to disk.
3128 */
3129 offset = i_size & (PAGE_SIZE - 1);
3130 if ((folio->index >= end_index + 1) || !offset)
3131 goto out;
3132
3133 folio_zero_segment(folio, offset, folio_size(folio));
3134 write:
3135 /* Dentry/quota blocks are controlled by checkpoint */
3136 if (S_ISDIR(inode->i_mode) || quota_inode) {
3137 struct f2fs_lock_context lc;
3138
3139 /*
3140 * We need to wait for node_write to avoid block allocation during
3141 * checkpoint. This can only happen to quota writes which can cause
3142 * the below discard race condition.
3143 */
3144 if (quota_inode)
3145 f2fs_down_read_trace(&sbi->node_write, &lc);
3146
3147 fio.need_lock = LOCK_DONE;
3148 err = f2fs_do_write_data_page(&fio);
3149
3150 if (quota_inode)
3151 f2fs_up_read_trace(&sbi->node_write, &lc);
3152
3153 goto done;
3154 }
3155
3156 need_balance_fs = true;
3157 err = -EAGAIN;
3158 if (f2fs_has_inline_data(inode)) {
3159 err = f2fs_write_inline_data(inode, folio);
3160 if (!err)
3161 goto out;
3162 }
3163
3164 if (err == -EAGAIN) {
3165 err = f2fs_do_write_data_page(&fio);
3166 if (err == -EAGAIN) {
3167 f2fs_bug_on(sbi, compr_blocks);
3168 fio.need_lock = LOCK_REQ;
3169 err = f2fs_do_write_data_page(&fio);
3170 }
3171 }
3172
3173 if (err) {
3174 file_set_keep_isize(inode);
3175 } else {
3176 spin_lock(&F2FS_I(inode)->i_size_lock);
3177 if (F2FS_I(inode)->last_disk_size < psize)
3178 F2FS_I(inode)->last_disk_size = psize;
3179 spin_unlock(&F2FS_I(inode)->i_size_lock);
3180 }
3181
3182 done:
3183 if (err && err != -ENOENT)
3184 goto redirty_out;
3185
3186 out:
3187 inode_dec_dirty_pages(inode);
3188 if (err) {
3189 folio_clear_uptodate(folio);
3190 folio_clear_f2fs_gcing(folio);
3191 }
3192 folio_unlock(folio);
3193 if (!S_ISDIR(inode->i_mode) && !IS_NOQUOTA(inode) &&
3194 !F2FS_I(inode)->wb_task && allow_balance)
3195 f2fs_balance_fs(sbi, need_balance_fs);
3196
3197 if (unlikely(f2fs_cp_error(sbi))) {
3198 f2fs_submit_merged_write(sbi, DATA);
3199 if (bio && *bio)
3200 f2fs_submit_merged_ipu_write(sbi, bio, NULL);
3201 submitted = NULL;
3202 }
3203
3204 if (submitted)
3205 *submitted = fio.submitted;
3206
3207 return 0;
3208
3209 redirty_out:
3210 folio_redirty_for_writepage(wbc, folio);
3211 /*
3212 * pageout() in MM translates EAGAIN, so calls handle_write_error()
3213 * -> mapping_set_error() -> set_bit(AS_EIO, ...).
3214 * file_write_and_wait_range() will see EIO error, which is critical
3215 * to return value of fsync() followed by atomic_write failure to user.
3216 */
3217 folio_unlock(folio);
3218 if (!err)
3219 return 1;
3220 return err;
3221 }
3222
3223 /*
3224 * This function was copied from write_cache_pages from mm/page-writeback.c.
3225 * The major change is making write step of cold data page separately from
3226 * warm/hot data page.
3227 */
f2fs_write_cache_pages(struct address_space * mapping,struct writeback_control * wbc,enum iostat_type io_type)3228 static int f2fs_write_cache_pages(struct address_space *mapping,
3229 struct writeback_control *wbc,
3230 enum iostat_type io_type)
3231 {
3232 int ret = 0;
3233 int done = 0, retry = 0;
3234 struct page *pages_local[F2FS_ONSTACK_PAGES];
3235 struct page **pages = pages_local;
3236 struct folio_batch fbatch;
3237 struct f2fs_sb_info *sbi = F2FS_M_SB(mapping);
3238 struct bio *bio = NULL;
3239 sector_t last_block;
3240 #ifdef CONFIG_F2FS_FS_COMPRESSION
3241 struct inode *inode = mapping->host;
3242 struct compress_ctx cc = {
3243 .inode = inode,
3244 .log_cluster_size = F2FS_I(inode)->i_log_cluster_size,
3245 .cluster_size = F2FS_I(inode)->i_cluster_size,
3246 .cluster_idx = NULL_CLUSTER,
3247 .rpages = NULL,
3248 .nr_rpages = 0,
3249 .cpages = NULL,
3250 .valid_nr_cpages = 0,
3251 .rbuf = NULL,
3252 .cbuf = NULL,
3253 .rlen = PAGE_SIZE * F2FS_I(inode)->i_cluster_size,
3254 .private = NULL,
3255 };
3256 #endif
3257 int nr_folios, p, idx;
3258 int nr_pages;
3259 unsigned int max_pages = F2FS_ONSTACK_PAGES;
3260 pgoff_t index;
3261 pgoff_t end; /* Inclusive */
3262 pgoff_t done_index;
3263 int range_whole = 0;
3264 xa_mark_t tag;
3265 int nwritten = 0;
3266 int submitted = 0;
3267 int i;
3268
3269 #ifdef CONFIG_F2FS_FS_COMPRESSION
3270 if (f2fs_compressed_file(inode) &&
3271 1 << cc.log_cluster_size > F2FS_ONSTACK_PAGES) {
3272 pages = f2fs_kzalloc(sbi, sizeof(struct page *) <<
3273 cc.log_cluster_size, GFP_NOFS | __GFP_NOFAIL);
3274 max_pages = 1 << cc.log_cluster_size;
3275 }
3276 #endif
3277
3278 folio_batch_init(&fbatch);
3279
3280 if (get_dirty_pages(mapping->host) <=
3281 SM_I(F2FS_M_SB(mapping))->min_hot_blocks)
3282 set_inode_flag(mapping->host, FI_HOT_DATA);
3283 else
3284 clear_inode_flag(mapping->host, FI_HOT_DATA);
3285
3286 if (wbc->range_cyclic) {
3287 index = mapping->writeback_index; /* prev offset */
3288 end = -1;
3289 } else {
3290 index = wbc->range_start >> PAGE_SHIFT;
3291 end = wbc->range_end >> PAGE_SHIFT;
3292 if (wbc->range_start == 0 && wbc->range_end == LLONG_MAX)
3293 range_whole = 1;
3294 }
3295 tag = wbc_to_tag(wbc);
3296 retry:
3297 retry = 0;
3298 if (wbc->sync_mode == WB_SYNC_ALL || wbc->tagged_writepages)
3299 tag_pages_for_writeback(mapping, index, end);
3300 done_index = index;
3301 while (!done && !retry && (index <= end)) {
3302 nr_pages = 0;
3303 again:
3304 nr_folios = filemap_get_folios_tag(mapping, &index, end,
3305 tag, &fbatch);
3306 if (nr_folios == 0) {
3307 if (nr_pages)
3308 goto write;
3309 break;
3310 }
3311
3312 for (i = 0; i < nr_folios; i++) {
3313 struct folio *folio = fbatch.folios[i];
3314
3315 idx = 0;
3316 p = folio_nr_pages(folio);
3317 add_more:
3318 pages[nr_pages] = folio_page(folio, idx);
3319 folio_get(folio);
3320 if (++nr_pages == max_pages) {
3321 index = folio->index + idx + 1;
3322 folio_batch_release(&fbatch);
3323 goto write;
3324 }
3325 if (++idx < p)
3326 goto add_more;
3327 }
3328 folio_batch_release(&fbatch);
3329 goto again;
3330 write:
3331 for (i = 0; i < nr_pages; i++) {
3332 struct page *page = pages[i];
3333 struct folio *folio = page_folio(page);
3334 bool need_readd;
3335 readd:
3336 need_readd = false;
3337 #ifdef CONFIG_F2FS_FS_COMPRESSION
3338 if (f2fs_compressed_file(inode)) {
3339 void *fsdata = NULL;
3340 struct page *pagep;
3341 int ret2;
3342
3343 ret = f2fs_init_compress_ctx(&cc);
3344 if (ret) {
3345 done = 1;
3346 break;
3347 }
3348
3349 if (!f2fs_cluster_can_merge_page(&cc,
3350 folio->index)) {
3351 ret = f2fs_write_multi_pages(&cc,
3352 &submitted, wbc, io_type);
3353 if (!ret)
3354 need_readd = true;
3355 goto result;
3356 }
3357
3358 if (unlikely(f2fs_cp_error(sbi)))
3359 goto lock_folio;
3360
3361 if (!f2fs_cluster_is_empty(&cc))
3362 goto lock_folio;
3363
3364 if (f2fs_all_cluster_page_ready(&cc,
3365 pages, i, nr_pages, true))
3366 goto lock_folio;
3367
3368 ret2 = f2fs_prepare_compress_overwrite(
3369 inode, &pagep,
3370 folio->index, &fsdata);
3371 if (ret2 < 0) {
3372 ret = ret2;
3373 done = 1;
3374 break;
3375 } else if (ret2 &&
3376 (!f2fs_compress_write_end(inode,
3377 fsdata, folio->index, 1) ||
3378 !f2fs_all_cluster_page_ready(&cc,
3379 pages, i, nr_pages,
3380 false))) {
3381 retry = 1;
3382 break;
3383 }
3384 }
3385 #endif
3386 /* give a priority to WB_SYNC threads */
3387 if (atomic_read(&sbi->wb_sync_req[DATA]) &&
3388 wbc->sync_mode == WB_SYNC_NONE) {
3389 done = 1;
3390 break;
3391 }
3392 #ifdef CONFIG_F2FS_FS_COMPRESSION
3393 lock_folio:
3394 #endif
3395 done_index = folio->index;
3396 retry_write:
3397 folio_lock(folio);
3398
3399 if (unlikely(folio->mapping != mapping)) {
3400 continue_unlock:
3401 folio_unlock(folio);
3402 continue;
3403 }
3404
3405 if (!folio_test_dirty(folio)) {
3406 /* someone wrote it for us */
3407 goto continue_unlock;
3408 }
3409
3410 if (folio_test_writeback(folio)) {
3411 if (wbc->sync_mode == WB_SYNC_NONE)
3412 goto continue_unlock;
3413 f2fs_folio_wait_writeback(folio, DATA, true, true);
3414 }
3415
3416 if (!folio_clear_dirty_for_io(folio))
3417 goto continue_unlock;
3418
3419 #ifdef CONFIG_F2FS_FS_COMPRESSION
3420 if (f2fs_compressed_file(inode)) {
3421 folio_get(folio);
3422 f2fs_compress_ctx_add_page(&cc, folio);
3423 continue;
3424 }
3425 #endif
3426 submitted = 0;
3427 ret = f2fs_write_single_data_page(folio,
3428 &submitted, &bio, &last_block,
3429 wbc, io_type, 0, true);
3430 #ifdef CONFIG_F2FS_FS_COMPRESSION
3431 result:
3432 #endif
3433 nwritten += submitted;
3434 wbc->nr_to_write -= submitted;
3435
3436 if (unlikely(ret)) {
3437 /*
3438 * keep nr_to_write, since vfs uses this to
3439 * get # of written pages.
3440 */
3441 if (ret == 1) {
3442 ret = 0;
3443 goto next;
3444 } else if (ret == -EAGAIN) {
3445 ret = 0;
3446 if (wbc->sync_mode == WB_SYNC_ALL) {
3447 f2fs_schedule_timeout(
3448 DEFAULT_SCHEDULE_TIMEOUT);
3449 goto retry_write;
3450 }
3451 goto next;
3452 }
3453 done_index = folio_next_index(folio);
3454 done = 1;
3455 break;
3456 }
3457
3458 if (wbc->nr_to_write <= 0 &&
3459 wbc->sync_mode == WB_SYNC_NONE) {
3460 done = 1;
3461 break;
3462 }
3463 next:
3464 if (need_readd)
3465 goto readd;
3466 }
3467 release_pages(pages, nr_pages);
3468 cond_resched();
3469 }
3470 #ifdef CONFIG_F2FS_FS_COMPRESSION
3471 /* flush remained pages in compress cluster */
3472 if (f2fs_compressed_file(inode) && !f2fs_cluster_is_empty(&cc)) {
3473 ret = f2fs_write_multi_pages(&cc, &submitted, wbc, io_type);
3474 nwritten += submitted;
3475 wbc->nr_to_write -= submitted;
3476 if (ret) {
3477 done = 1;
3478 retry = 0;
3479 }
3480 }
3481 if (f2fs_compressed_file(inode))
3482 f2fs_destroy_compress_ctx(&cc, false);
3483 #endif
3484 if (retry) {
3485 index = 0;
3486 end = -1;
3487 goto retry;
3488 }
3489 if (wbc->range_cyclic && !done)
3490 done_index = 0;
3491 if (wbc->range_cyclic || (range_whole && wbc->nr_to_write > 0))
3492 mapping->writeback_index = done_index;
3493
3494 if (nwritten)
3495 f2fs_submit_merged_write_cond(F2FS_M_SB(mapping), mapping->host,
3496 NULL, 0, DATA);
3497 /* submit cached bio of IPU write */
3498 if (bio)
3499 f2fs_submit_merged_ipu_write(sbi, &bio, NULL);
3500
3501 #ifdef CONFIG_F2FS_FS_COMPRESSION
3502 if (pages != pages_local)
3503 kfree(pages);
3504 #endif
3505
3506 return ret;
3507 }
3508
__should_serialize_io(struct inode * inode,struct writeback_control * wbc)3509 static inline bool __should_serialize_io(struct inode *inode,
3510 struct writeback_control *wbc)
3511 {
3512 /* to avoid deadlock in path of data flush */
3513 if (F2FS_I(inode)->wb_task)
3514 return false;
3515
3516 if (!S_ISREG(inode->i_mode))
3517 return false;
3518 if (IS_NOQUOTA(inode))
3519 return false;
3520
3521 if (f2fs_is_pinned_file(inode))
3522 return false;
3523 if (f2fs_need_compress_data(inode))
3524 return true;
3525 if (wbc->sync_mode != WB_SYNC_ALL)
3526 return true;
3527 if (get_dirty_pages(inode) >= SM_I(F2FS_I_SB(inode))->min_seq_blocks)
3528 return true;
3529 return false;
3530 }
3531
account_writeback(struct inode * inode,bool inc)3532 static inline void account_writeback(struct inode *inode, bool inc)
3533 {
3534 if (!f2fs_sb_has_compression(F2FS_I_SB(inode)))
3535 return;
3536
3537 f2fs_down_read(&F2FS_I(inode)->i_sem);
3538 if (inc)
3539 atomic_inc(&F2FS_I(inode)->writeback);
3540 else
3541 atomic_dec(&F2FS_I(inode)->writeback);
3542 f2fs_up_read(&F2FS_I(inode)->i_sem);
3543 }
3544
update_skipped_write(struct f2fs_sb_info * sbi,struct writeback_control * wbc)3545 static inline void update_skipped_write(struct f2fs_sb_info *sbi,
3546 struct writeback_control *wbc)
3547 {
3548 long skipped = wbc->pages_skipped;
3549
3550 if (is_sbi_flag_set(sbi, SBI_ENABLE_CHECKPOINT) && skipped &&
3551 wbc->sync_mode == WB_SYNC_ALL)
3552 atomic_add(skipped, &sbi->nr_pages[F2FS_SKIPPED_WRITE]);
3553 }
3554
__f2fs_write_data_pages(struct address_space * mapping,struct writeback_control * wbc,enum iostat_type io_type)3555 static int __f2fs_write_data_pages(struct address_space *mapping,
3556 struct writeback_control *wbc,
3557 enum iostat_type io_type)
3558 {
3559 struct inode *inode = mapping->host;
3560 struct f2fs_sb_info *sbi = F2FS_I_SB(inode);
3561 struct blk_plug plug;
3562 int ret;
3563 bool locked = false;
3564
3565 /* skip writing if there is no dirty page in this inode */
3566 if (!get_dirty_pages(inode) && wbc->sync_mode == WB_SYNC_NONE)
3567 return 0;
3568
3569 /* during POR, we don't need to trigger writepage at all. */
3570 if (unlikely(is_sbi_flag_set(sbi, SBI_POR_DOING)))
3571 goto skip_write;
3572
3573 if ((S_ISDIR(inode->i_mode) || IS_NOQUOTA(inode)) &&
3574 wbc->sync_mode == WB_SYNC_NONE &&
3575 get_dirty_pages(inode) < nr_pages_to_skip(sbi, DATA) &&
3576 f2fs_available_free_memory(sbi, DIRTY_DENTS))
3577 goto skip_write;
3578
3579 /* skip writing in file defragment preparing stage */
3580 if (is_inode_flag_set(inode, FI_SKIP_WRITES))
3581 goto skip_write;
3582
3583 trace_f2fs_writepages(mapping->host, wbc, DATA);
3584
3585 /* to avoid spliting IOs due to mixed WB_SYNC_ALL and WB_SYNC_NONE */
3586 if (wbc->sync_mode == WB_SYNC_ALL)
3587 atomic_inc(&sbi->wb_sync_req[DATA]);
3588 else if (atomic_read(&sbi->wb_sync_req[DATA])) {
3589 /* to avoid potential deadlock */
3590 if (current->plug)
3591 blk_finish_plug(current->plug);
3592 goto skip_write;
3593 }
3594
3595 if (__should_serialize_io(inode, wbc)) {
3596 mutex_lock(&sbi->writepages);
3597 locked = true;
3598 }
3599
3600 account_writeback(inode, true);
3601
3602 blk_start_plug(&plug);
3603 ret = f2fs_write_cache_pages(mapping, wbc, io_type);
3604 blk_finish_plug(&plug);
3605
3606 account_writeback(inode, false);
3607
3608 if (locked)
3609 mutex_unlock(&sbi->writepages);
3610
3611 if (wbc->sync_mode == WB_SYNC_ALL)
3612 atomic_dec(&sbi->wb_sync_req[DATA]);
3613 /*
3614 * if some pages were truncated, we cannot guarantee its mapping->host
3615 * to detect pending bios.
3616 */
3617
3618 f2fs_remove_dirty_inode(inode);
3619
3620 /*
3621 * f2fs_write_cache_pages() has retry logic for EAGAIN case which is
3622 * common when racing w/ checkpoint, so only update skipped write
3623 * when ret is non-zero.
3624 */
3625 if (ret)
3626 update_skipped_write(sbi, wbc);
3627 return ret;
3628
3629 skip_write:
3630 wbc->pages_skipped += get_dirty_pages(inode);
3631 update_skipped_write(sbi, wbc);
3632 trace_f2fs_writepages(mapping->host, wbc, DATA);
3633 return 0;
3634 }
3635
f2fs_write_data_pages(struct address_space * mapping,struct writeback_control * wbc)3636 static int f2fs_write_data_pages(struct address_space *mapping,
3637 struct writeback_control *wbc)
3638 {
3639 struct inode *inode = mapping->host;
3640
3641 return __f2fs_write_data_pages(mapping, wbc,
3642 F2FS_I(inode)->cp_task == current ?
3643 FS_CP_DATA_IO : FS_DATA_IO);
3644 }
3645
f2fs_write_failed(struct inode * inode,loff_t to)3646 void f2fs_write_failed(struct inode *inode, loff_t to)
3647 {
3648 loff_t i_size = i_size_read(inode);
3649
3650 if (IS_NOQUOTA(inode))
3651 return;
3652
3653 /* In the fs-verity case, f2fs_end_enable_verity() does the truncate */
3654 if (to > i_size && !f2fs_verity_in_progress(inode)) {
3655 f2fs_down_write(&F2FS_I(inode)->i_gc_rwsem[WRITE]);
3656 filemap_invalidate_lock(inode->i_mapping);
3657
3658 truncate_pagecache(inode, i_size);
3659 f2fs_truncate_blocks(inode, i_size, true);
3660
3661 filemap_invalidate_unlock(inode->i_mapping);
3662 f2fs_up_write(&F2FS_I(inode)->i_gc_rwsem[WRITE]);
3663 }
3664 }
3665
prepare_write_begin(struct f2fs_sb_info * sbi,struct folio * folio,loff_t pos,unsigned int len,block_t * blk_addr,bool * node_changed)3666 static int prepare_write_begin(struct f2fs_sb_info *sbi,
3667 struct folio *folio, loff_t pos, unsigned int len,
3668 block_t *blk_addr, bool *node_changed)
3669 {
3670 struct inode *inode = folio->mapping->host;
3671 pgoff_t index = folio->index;
3672 struct dnode_of_data dn;
3673 struct f2fs_lock_context lc;
3674 struct folio *ifolio;
3675 bool locked = false;
3676 int flag = F2FS_GET_BLOCK_PRE_AIO;
3677 int err = 0;
3678
3679 if (!f2fs_has_inline_data(inode) && !f2fs_compressed_file(inode) &&
3680 (pos & PAGE_MASK) < i_size_read(inode) &&
3681 f2fs_lookup_read_extent_cache_block(inode, index, blk_addr))
3682 return 0;
3683
3684 /*
3685 * If a whole page is being written and we already preallocated all the
3686 * blocks, then there is no need to get a block address now.
3687 */
3688 if (len == PAGE_SIZE && is_inode_flag_set(inode, FI_PREALLOCATED_ALL))
3689 return 0;
3690
3691 /* f2fs_lock_op avoids race between write CP and convert_inline_page */
3692 if (f2fs_has_inline_data(inode)) {
3693 if (pos + len > MAX_INLINE_DATA(inode))
3694 flag = F2FS_GET_BLOCK_DEFAULT;
3695 f2fs_map_lock(sbi, &lc, flag);
3696 locked = true;
3697 } else if ((pos & PAGE_MASK) >= i_size_read(inode)) {
3698 f2fs_map_lock(sbi, &lc, flag);
3699 locked = true;
3700 }
3701
3702 restart:
3703 /* check inline_data */
3704 ifolio = f2fs_get_inode_folio(sbi, inode->i_ino);
3705 if (IS_ERR(ifolio)) {
3706 err = PTR_ERR(ifolio);
3707 goto unlock_out;
3708 }
3709
3710 set_new_dnode(&dn, inode, ifolio, ifolio, 0);
3711
3712 if (f2fs_has_inline_data(inode)) {
3713 if (pos + len <= MAX_INLINE_DATA(inode)) {
3714 f2fs_do_read_inline_data(folio, ifolio);
3715 set_inode_flag(inode, FI_DATA_EXIST);
3716 if (inode->i_nlink)
3717 folio_set_f2fs_inline(ifolio);
3718 goto out;
3719 }
3720 err = f2fs_convert_inline_folio(&dn, folio);
3721 if (err || dn.data_blkaddr != NULL_ADDR)
3722 goto out;
3723 }
3724
3725 if (!f2fs_lookup_read_extent_cache_block(inode, index,
3726 &dn.data_blkaddr)) {
3727 if (IS_DEVICE_ALIASING(inode)) {
3728 err = -ENODATA;
3729 goto out;
3730 }
3731
3732 if (locked) {
3733 err = f2fs_reserve_block(&dn, index);
3734 goto out;
3735 }
3736
3737 /* hole case */
3738 err = f2fs_get_dnode_of_data(&dn, index, LOOKUP_NODE);
3739 if (!err && dn.data_blkaddr != NULL_ADDR)
3740 goto out;
3741 f2fs_put_dnode(&dn);
3742 f2fs_map_lock(sbi, &lc, F2FS_GET_BLOCK_PRE_AIO);
3743 WARN_ON(flag != F2FS_GET_BLOCK_PRE_AIO);
3744 locked = true;
3745 goto restart;
3746 }
3747 out:
3748 if (!err) {
3749 /* convert_inline_page can make node_changed */
3750 *blk_addr = dn.data_blkaddr;
3751 *node_changed = dn.node_changed;
3752 }
3753 f2fs_put_dnode(&dn);
3754 unlock_out:
3755 if (locked)
3756 f2fs_map_unlock(sbi, &lc, flag);
3757 return err;
3758 }
3759
__find_data_block(struct inode * inode,pgoff_t index,block_t * blk_addr)3760 static int __find_data_block(struct inode *inode, pgoff_t index,
3761 block_t *blk_addr)
3762 {
3763 struct dnode_of_data dn;
3764 struct folio *ifolio;
3765 int err = 0;
3766
3767 ifolio = f2fs_get_inode_folio(F2FS_I_SB(inode), inode->i_ino);
3768 if (IS_ERR(ifolio))
3769 return PTR_ERR(ifolio);
3770
3771 set_new_dnode(&dn, inode, ifolio, ifolio, 0);
3772
3773 if (!f2fs_lookup_read_extent_cache_block(inode, index,
3774 &dn.data_blkaddr)) {
3775 /* hole case */
3776 err = f2fs_get_dnode_of_data(&dn, index, LOOKUP_NODE);
3777 if (err) {
3778 dn.data_blkaddr = NULL_ADDR;
3779 err = 0;
3780 }
3781 }
3782 *blk_addr = dn.data_blkaddr;
3783 f2fs_put_dnode(&dn);
3784 return err;
3785 }
3786
__reserve_data_block(struct inode * inode,pgoff_t index,block_t * blk_addr,bool * node_changed)3787 static int __reserve_data_block(struct inode *inode, pgoff_t index,
3788 block_t *blk_addr, bool *node_changed)
3789 {
3790 struct f2fs_sb_info *sbi = F2FS_I_SB(inode);
3791 struct dnode_of_data dn;
3792 struct f2fs_lock_context lc;
3793 struct folio *ifolio;
3794 int err = 0;
3795
3796 f2fs_map_lock(sbi, &lc, F2FS_GET_BLOCK_PRE_AIO);
3797
3798 ifolio = f2fs_get_inode_folio(sbi, inode->i_ino);
3799 if (IS_ERR(ifolio)) {
3800 err = PTR_ERR(ifolio);
3801 goto unlock_out;
3802 }
3803 set_new_dnode(&dn, inode, ifolio, ifolio, 0);
3804
3805 if (!f2fs_lookup_read_extent_cache_block(dn.inode, index,
3806 &dn.data_blkaddr))
3807 err = f2fs_reserve_block(&dn, index);
3808
3809 *blk_addr = dn.data_blkaddr;
3810 *node_changed = dn.node_changed;
3811 f2fs_put_dnode(&dn);
3812
3813 unlock_out:
3814 f2fs_map_unlock(sbi, &lc, F2FS_GET_BLOCK_PRE_AIO);
3815 return err;
3816 }
3817
prepare_atomic_write_begin(struct f2fs_sb_info * sbi,struct folio * folio,loff_t pos,unsigned int len,block_t * blk_addr,bool * node_changed)3818 static int prepare_atomic_write_begin(struct f2fs_sb_info *sbi,
3819 struct folio *folio, loff_t pos, unsigned int len,
3820 block_t *blk_addr, bool *node_changed)
3821 {
3822 struct inode *inode = folio->mapping->host;
3823 struct inode *cow_inode = F2FS_I(inode)->cow_inode;
3824 pgoff_t index = folio->index;
3825 int err = 0;
3826 block_t ori_blk_addr = NULL_ADDR;
3827 bool cow_has_reserved_block = false;
3828
3829 /* If pos is beyond the end of file, reserve a new block in COW inode */
3830 if ((pos & PAGE_MASK) >= i_size_read(inode))
3831 goto reserve_block;
3832
3833 /* Look for the block in COW inode first */
3834 err = __find_data_block(cow_inode, index, blk_addr);
3835 if (err)
3836 return err;
3837
3838 if (__is_valid_data_blkaddr(*blk_addr))
3839 return 0;
3840
3841 if (*blk_addr == NEW_ADDR)
3842 cow_has_reserved_block = true;
3843
3844 if (is_inode_flag_set(inode, FI_ATOMIC_REPLACE))
3845 goto reserve_block;
3846
3847 /* Look for the block in the original inode */
3848 err = __find_data_block(inode, index, &ori_blk_addr);
3849 if (err)
3850 return err;
3851
3852 reserve_block:
3853 /* Finally, we should reserve a new block in COW inode for the update */
3854 if (!cow_has_reserved_block) {
3855 err = __reserve_data_block(cow_inode, index, blk_addr,
3856 node_changed);
3857 if (err)
3858 return err;
3859 inc_atomic_write_cnt(inode);
3860 }
3861
3862 if (ori_blk_addr != NULL_ADDR)
3863 *blk_addr = ori_blk_addr;
3864 return 0;
3865 }
3866
f2fs_write_begin(const struct kiocb * iocb,struct address_space * mapping,loff_t pos,unsigned len,struct folio ** foliop,void ** fsdata)3867 static int f2fs_write_begin(const struct kiocb *iocb,
3868 struct address_space *mapping,
3869 loff_t pos, unsigned len, struct folio **foliop,
3870 void **fsdata)
3871 {
3872 struct inode *inode = mapping->host;
3873 struct f2fs_sb_info *sbi = F2FS_I_SB(inode);
3874 struct folio *folio;
3875 pgoff_t index = pos >> PAGE_SHIFT;
3876 bool need_balance = false;
3877 block_t blkaddr = NULL_ADDR;
3878 int err = 0;
3879
3880 trace_f2fs_write_begin(inode, pos, len);
3881
3882 if (!f2fs_is_checkpoint_ready(sbi)) {
3883 err = -ENOSPC;
3884 goto fail;
3885 }
3886
3887 /*
3888 * We should check this at this moment to avoid deadlock on inode page
3889 * and #0 page. The locking rule for inline_data conversion should be:
3890 * folio_lock(folio #0) -> folio_lock(inode_page)
3891 */
3892 if (index != 0) {
3893 err = f2fs_convert_inline_inode(inode);
3894 if (err)
3895 goto fail;
3896 }
3897
3898 #ifdef CONFIG_F2FS_FS_COMPRESSION
3899 if (f2fs_compressed_file(inode)) {
3900 int ret;
3901 struct page *page;
3902
3903 *fsdata = NULL;
3904
3905 if (len == PAGE_SIZE && !(f2fs_is_atomic_file(inode)))
3906 goto repeat;
3907
3908 ret = f2fs_prepare_compress_overwrite(inode, &page,
3909 index, fsdata);
3910 if (ret < 0) {
3911 err = ret;
3912 goto fail;
3913 } else if (ret) {
3914 *foliop = page_folio(page);
3915 return 0;
3916 }
3917 }
3918 #endif
3919
3920 repeat:
3921 /*
3922 * Do not use FGP_STABLE to avoid deadlock.
3923 * Will wait that below with our IO control.
3924 */
3925 folio = f2fs_filemap_get_folio(mapping, index,
3926 FGP_LOCK | FGP_WRITE | FGP_CREAT,
3927 mapping_gfp_mask(mapping));
3928 if (IS_ERR(folio)) {
3929 err = PTR_ERR(folio);
3930 goto fail;
3931 }
3932
3933 /* TODO: cluster can be compressed due to race with .writepage */
3934
3935 *foliop = folio;
3936
3937 if (f2fs_is_atomic_file(inode))
3938 err = prepare_atomic_write_begin(sbi, folio, pos, len,
3939 &blkaddr, &need_balance);
3940 else
3941 err = prepare_write_begin(sbi, folio, pos, len,
3942 &blkaddr, &need_balance);
3943 if (err)
3944 goto put_folio;
3945
3946 if (need_balance && !IS_NOQUOTA(inode) &&
3947 has_not_enough_free_secs(sbi, 0, 0)) {
3948 folio_unlock(folio);
3949 f2fs_balance_fs(sbi, true);
3950 folio_lock(folio);
3951 if (folio->mapping != mapping) {
3952 /* The folio got truncated from under us */
3953 folio_unlock(folio);
3954 folio_put(folio);
3955 goto repeat;
3956 }
3957 }
3958
3959 f2fs_folio_wait_writeback(folio, DATA, false, true);
3960
3961 if (len == folio_size(folio) || folio_test_uptodate(folio))
3962 return 0;
3963
3964 if (!(pos & (PAGE_SIZE - 1)) && (pos + len) >= i_size_read(inode) &&
3965 !f2fs_verity_in_progress(inode)) {
3966 folio_zero_segment(folio, len, folio_size(folio));
3967 return 0;
3968 }
3969
3970 if (blkaddr == NEW_ADDR) {
3971 folio_zero_segment(folio, 0, folio_size(folio));
3972 folio_mark_uptodate(folio);
3973 } else {
3974 if (!f2fs_is_valid_blkaddr(sbi, blkaddr,
3975 DATA_GENERIC_ENHANCE_READ)) {
3976 err = -EFSCORRUPTED;
3977 goto put_folio;
3978 }
3979 /*
3980 * Although the block may be stored in the COW inode, the folio
3981 * belongs to @inode and its data was encrypted (or not) using
3982 * @inode's context (see f2fs_set_bio_crypt_ctx()). Read with
3983 * @inode so the post-read decryption decision matches the
3984 * folio's owner; otherwise an unencrypted @inode whose COW inode
3985 * is encrypted hits a NULL ->i_crypt_info on decryption.
3986 */
3987 f2fs_submit_page_read(inode,
3988 NULL, /* can't write to fsverity files */
3989 folio, blkaddr, 0, true);
3990
3991 folio_lock(folio);
3992 if (unlikely(folio->mapping != mapping)) {
3993 folio_unlock(folio);
3994 folio_put(folio);
3995 goto repeat;
3996 }
3997 if (unlikely(!folio_test_uptodate(folio))) {
3998 err = -EIO;
3999 goto put_folio;
4000 }
4001 }
4002 return 0;
4003
4004 put_folio:
4005 f2fs_folio_put(folio, true);
4006 fail:
4007 f2fs_write_failed(inode, pos + len);
4008 return err;
4009 }
4010
f2fs_write_end(const struct kiocb * iocb,struct address_space * mapping,loff_t pos,unsigned len,unsigned copied,struct folio * folio,void * fsdata)4011 static int f2fs_write_end(const struct kiocb *iocb,
4012 struct address_space *mapping,
4013 loff_t pos, unsigned len, unsigned copied,
4014 struct folio *folio, void *fsdata)
4015 {
4016 struct inode *inode = folio->mapping->host;
4017
4018 trace_f2fs_write_end(inode, pos, len, copied);
4019
4020 /*
4021 * This should be come from len == PAGE_SIZE, and we expect copied
4022 * should be PAGE_SIZE. Otherwise, we treat it with zero copied and
4023 * let generic_perform_write() try to copy data again through copied=0.
4024 */
4025 if (!folio_test_uptodate(folio)) {
4026 if (unlikely(copied != len))
4027 copied = 0;
4028 else
4029 folio_mark_uptodate(folio);
4030 }
4031
4032 #ifdef CONFIG_F2FS_FS_COMPRESSION
4033 /* overwrite compressed file */
4034 if (f2fs_compressed_file(inode) && fsdata) {
4035 f2fs_compress_write_end(inode, fsdata, folio->index, copied);
4036 f2fs_update_time(F2FS_I_SB(inode), REQ_TIME);
4037
4038 if (pos + copied > i_size_read(inode) &&
4039 !f2fs_verity_in_progress(inode))
4040 f2fs_i_size_write(inode, pos + copied);
4041 return copied;
4042 }
4043 #endif
4044
4045 if (!copied)
4046 goto unlock_out;
4047
4048 folio_mark_dirty(folio);
4049
4050 if (f2fs_is_atomic_file(inode))
4051 folio_set_f2fs_atomic(folio);
4052
4053 if (pos + copied > i_size_read(inode) &&
4054 !f2fs_verity_in_progress(inode)) {
4055 f2fs_i_size_write(inode, pos + copied);
4056 if (f2fs_is_atomic_file(inode))
4057 f2fs_i_size_write(F2FS_I(inode)->cow_inode,
4058 pos + copied);
4059 }
4060 unlock_out:
4061 f2fs_folio_put(folio, true);
4062 f2fs_update_time(F2FS_I_SB(inode), REQ_TIME);
4063 return copied;
4064 }
4065
f2fs_invalidate_folio(struct folio * folio,size_t offset,size_t length)4066 void f2fs_invalidate_folio(struct folio *folio, size_t offset, size_t length)
4067 {
4068 struct inode *inode = folio->mapping->host;
4069 struct f2fs_sb_info *sbi = F2FS_I_SB(inode);
4070
4071 if (inode->i_ino >= F2FS_ROOT_INO(sbi) &&
4072 (offset || length != folio_size(folio)))
4073 return;
4074
4075 if (folio_test_dirty(folio)) {
4076 if (inode->i_ino == F2FS_META_INO(sbi)) {
4077 dec_page_count(sbi, F2FS_DIRTY_META);
4078 } else if (inode->i_ino == F2FS_NODE_INO(sbi)) {
4079 dec_page_count(sbi, F2FS_DIRTY_NODES);
4080 } else {
4081 inode_dec_dirty_pages(inode);
4082 f2fs_remove_dirty_inode(inode);
4083 }
4084 }
4085
4086 if (offset || length != folio_size(folio))
4087 return;
4088
4089 folio_cancel_dirty(folio);
4090 ffs_detach_free(folio);
4091 }
4092
f2fs_release_folio(struct folio * folio,gfp_t wait)4093 bool f2fs_release_folio(struct folio *folio, gfp_t wait)
4094 {
4095 /* If this is dirty folio, keep private data */
4096 if (folio_test_dirty(folio))
4097 return false;
4098
4099 ffs_detach_free(folio);
4100 return true;
4101 }
4102
f2fs_dirty_data_folio(struct address_space * mapping,struct folio * folio)4103 static bool f2fs_dirty_data_folio(struct address_space *mapping,
4104 struct folio *folio)
4105 {
4106 struct inode *inode = mapping->host;
4107
4108 trace_f2fs_set_page_dirty(folio, DATA);
4109
4110 if (!folio_test_uptodate(folio))
4111 folio_mark_uptodate(folio);
4112 BUG_ON(folio_test_swapcache(folio));
4113
4114 if (filemap_dirty_folio(mapping, folio)) {
4115 f2fs_update_dirty_folio(inode, folio);
4116 return true;
4117 }
4118 return false;
4119 }
4120
4121
f2fs_bmap_compress(struct inode * inode,sector_t block)4122 static sector_t f2fs_bmap_compress(struct inode *inode, sector_t block)
4123 {
4124 #ifdef CONFIG_F2FS_FS_COMPRESSION
4125 struct dnode_of_data dn;
4126 sector_t start_idx, blknr = 0;
4127 int ret;
4128
4129 start_idx = round_down(block, F2FS_I(inode)->i_cluster_size);
4130
4131 set_new_dnode(&dn, inode, NULL, NULL, 0);
4132 ret = f2fs_get_dnode_of_data(&dn, start_idx, LOOKUP_NODE);
4133 if (ret)
4134 return 0;
4135
4136 if (dn.data_blkaddr != COMPRESS_ADDR) {
4137 dn.ofs_in_node += block - start_idx;
4138 blknr = f2fs_data_blkaddr(&dn);
4139 if (!__is_valid_data_blkaddr(blknr))
4140 blknr = 0;
4141 }
4142
4143 f2fs_put_dnode(&dn);
4144 return blknr;
4145 #else
4146 return 0;
4147 #endif
4148 }
4149
4150
f2fs_bmap(struct address_space * mapping,sector_t block)4151 static sector_t f2fs_bmap(struct address_space *mapping, sector_t block)
4152 {
4153 struct inode *inode = mapping->host;
4154 sector_t blknr = 0;
4155
4156 if (f2fs_has_inline_data(inode))
4157 goto out;
4158
4159 /* make sure allocating whole blocks */
4160 if (mapping_tagged(mapping, PAGECACHE_TAG_DIRTY))
4161 filemap_write_and_wait(mapping);
4162
4163 /* Block number less than F2FS MAX BLOCKS */
4164 if (unlikely(block >= max_file_blocks(inode)))
4165 goto out;
4166
4167 if (f2fs_compressed_file(inode)) {
4168 blknr = f2fs_bmap_compress(inode, block);
4169 } else {
4170 struct f2fs_map_blocks map;
4171
4172 memset(&map, 0, sizeof(map));
4173 map.m_lblk = block;
4174 map.m_len = 1;
4175 map.m_next_pgofs = NULL;
4176 map.m_seg_type = NO_CHECK_TYPE;
4177
4178 if (!f2fs_map_blocks(inode, &map, F2FS_GET_BLOCK_BMAP))
4179 blknr = map.m_pblk;
4180 }
4181 out:
4182 trace_f2fs_bmap(inode, block, blknr);
4183 return blknr;
4184 }
4185
4186 #ifdef CONFIG_SWAP
f2fs_migrate_blocks(struct inode * inode,block_t start_blk,unsigned int blkcnt)4187 static int f2fs_migrate_blocks(struct inode *inode, block_t start_blk,
4188 unsigned int blkcnt)
4189 {
4190 struct f2fs_sb_info *sbi = F2FS_I_SB(inode);
4191 unsigned int blkofs;
4192 unsigned int blk_per_sec = BLKS_PER_SEC(sbi);
4193 unsigned int end_blk = start_blk + blkcnt - 1;
4194 unsigned int secidx = start_blk / blk_per_sec;
4195 unsigned int end_sec;
4196 int ret = 0;
4197
4198 if (!blkcnt)
4199 return 0;
4200 end_sec = end_blk / blk_per_sec;
4201
4202 f2fs_down_write(&F2FS_I(inode)->i_gc_rwsem[WRITE]);
4203 filemap_invalidate_lock(inode->i_mapping);
4204
4205 set_inode_flag(inode, FI_ALIGNED_WRITE);
4206 set_inode_flag(inode, FI_OPU_WRITE);
4207
4208 for (; secidx <= end_sec; secidx++) {
4209 unsigned int blkofs_end = secidx == end_sec ?
4210 end_blk % blk_per_sec : blk_per_sec - 1;
4211
4212 f2fs_down_write(&sbi->pin_sem);
4213
4214 ret = f2fs_allocate_pinning_section(sbi);
4215 if (ret) {
4216 f2fs_up_write(&sbi->pin_sem);
4217 break;
4218 }
4219
4220 set_inode_flag(inode, FI_SKIP_WRITES);
4221
4222 for (blkofs = 0; blkofs <= blkofs_end; blkofs++) {
4223 struct folio *folio;
4224 unsigned int blkidx = secidx * blk_per_sec + blkofs;
4225
4226 folio = f2fs_get_lock_data_folio(inode, blkidx, true);
4227 if (IS_ERR(folio)) {
4228 f2fs_up_write(&sbi->pin_sem);
4229 ret = PTR_ERR(folio);
4230 goto done;
4231 }
4232
4233 folio_mark_dirty(folio);
4234 f2fs_folio_put(folio, true);
4235 }
4236
4237 clear_inode_flag(inode, FI_SKIP_WRITES);
4238
4239 ret = filemap_fdatawrite(inode->i_mapping);
4240
4241 f2fs_up_write(&sbi->pin_sem);
4242
4243 if (ret)
4244 break;
4245 }
4246
4247 done:
4248 clear_inode_flag(inode, FI_SKIP_WRITES);
4249 clear_inode_flag(inode, FI_OPU_WRITE);
4250 clear_inode_flag(inode, FI_ALIGNED_WRITE);
4251
4252 filemap_invalidate_unlock(inode->i_mapping);
4253 f2fs_up_write(&F2FS_I(inode)->i_gc_rwsem[WRITE]);
4254
4255 return ret;
4256 }
4257
check_swap_activate(struct swap_info_struct * sis,struct file * swap_file,sector_t * span)4258 static int check_swap_activate(struct swap_info_struct *sis,
4259 struct file *swap_file, sector_t *span)
4260 {
4261 struct address_space *mapping = swap_file->f_mapping;
4262 struct inode *inode = mapping->host;
4263 struct f2fs_sb_info *sbi = F2FS_I_SB(inode);
4264 block_t cur_lblock;
4265 block_t last_lblock;
4266 block_t pblock;
4267 block_t lowest_pblock = -1;
4268 block_t highest_pblock = 0;
4269 int nr_extents = 0;
4270 unsigned int nr_pblocks;
4271 unsigned int blks_per_sec = BLKS_PER_SEC(sbi);
4272 unsigned int not_aligned = 0;
4273 int ret = 0;
4274
4275 /*
4276 * Map all the blocks into the extent list. This code doesn't try
4277 * to be very smart.
4278 */
4279 cur_lblock = 0;
4280 last_lblock = F2FS_BYTES_TO_BLK(i_size_read(inode));
4281
4282 while (cur_lblock < last_lblock && cur_lblock < sis->max) {
4283 struct f2fs_map_blocks map;
4284 bool last_extent = false;
4285 retry:
4286 cond_resched();
4287
4288 memset(&map, 0, sizeof(map));
4289 map.m_lblk = cur_lblock;
4290 map.m_len = last_lblock - cur_lblock;
4291 map.m_next_pgofs = NULL;
4292 map.m_next_extent = NULL;
4293 map.m_seg_type = NO_CHECK_TYPE;
4294 map.m_may_create = false;
4295
4296 ret = f2fs_map_blocks(inode, &map, F2FS_GET_BLOCK_FIEMAP);
4297 if (ret)
4298 goto out;
4299
4300 /* hole */
4301 if (!(map.m_flags & F2FS_MAP_FLAGS)) {
4302 f2fs_err(sbi, "Swapfile has holes");
4303 ret = -EINVAL;
4304 goto out;
4305 }
4306
4307 pblock = map.m_pblk;
4308 nr_pblocks = map.m_len;
4309
4310 if (!last_extent &&
4311 ((pblock - SM_I(sbi)->main_blkaddr) % blks_per_sec ||
4312 nr_pblocks % blks_per_sec ||
4313 f2fs_is_sequential_zone_area(sbi, pblock))) {
4314 not_aligned++;
4315
4316 nr_pblocks = roundup(nr_pblocks, blks_per_sec);
4317 if (cur_lblock + nr_pblocks > sis->max)
4318 nr_pblocks -= blks_per_sec;
4319
4320 /* this extent is last one */
4321 if (!nr_pblocks) {
4322 nr_pblocks = last_lblock - cur_lblock;
4323 last_extent = true;
4324 }
4325
4326 ret = f2fs_migrate_blocks(inode, cur_lblock,
4327 nr_pblocks);
4328 if (ret) {
4329 if (ret == -ENOENT)
4330 ret = -EINVAL;
4331 goto out;
4332 }
4333
4334 /* lookup block mapping info after block migration */
4335 goto retry;
4336 }
4337
4338 if (cur_lblock + nr_pblocks >= sis->max)
4339 nr_pblocks = sis->max - cur_lblock;
4340
4341 if (cur_lblock) { /* exclude the header page */
4342 if (pblock < lowest_pblock)
4343 lowest_pblock = pblock;
4344 if (pblock + nr_pblocks - 1 > highest_pblock)
4345 highest_pblock = pblock + nr_pblocks - 1;
4346 }
4347
4348 /*
4349 * We found a PAGE_SIZE-length, PAGE_SIZE-aligned run of blocks
4350 */
4351 ret = add_swap_extent(sis, cur_lblock, nr_pblocks, pblock);
4352 if (ret < 0)
4353 goto out;
4354 nr_extents += ret;
4355 cur_lblock += nr_pblocks;
4356 }
4357 ret = nr_extents;
4358 *span = 1 + highest_pblock - lowest_pblock;
4359 if (cur_lblock == 0)
4360 cur_lblock = 1; /* force Empty message */
4361 sis->max = cur_lblock;
4362 sis->pages = cur_lblock - 1;
4363 out:
4364 if (not_aligned)
4365 f2fs_warn(sbi, "Swapfile (%u) is not align to section: 1) creat(), 2) ioctl(F2FS_IOC_SET_PIN_FILE), 3) fallocate(%lu * N)",
4366 not_aligned, blks_per_sec * F2FS_BLKSIZE);
4367 return ret;
4368 }
4369
f2fs_swap_activate(struct swap_info_struct * sis,struct file * file,sector_t * span)4370 static int f2fs_swap_activate(struct swap_info_struct *sis, struct file *file,
4371 sector_t *span)
4372 {
4373 struct inode *inode = file_inode(file);
4374 struct f2fs_sb_info *sbi = F2FS_I_SB(inode);
4375 int ret;
4376
4377 if (!S_ISREG(inode->i_mode))
4378 return -EINVAL;
4379
4380 if (f2fs_readonly(sbi->sb))
4381 return -EROFS;
4382
4383 if (f2fs_lfs_mode(sbi) && !f2fs_sb_has_blkzoned(sbi)) {
4384 f2fs_err(sbi, "Swapfile not supported in LFS mode");
4385 return -EINVAL;
4386 }
4387
4388 ret = f2fs_convert_inline_inode(inode);
4389 if (ret)
4390 return ret;
4391
4392 if (!f2fs_disable_compressed_file(inode))
4393 return -EINVAL;
4394
4395 ret = filemap_fdatawrite(inode->i_mapping);
4396 if (ret < 0)
4397 return ret;
4398
4399 f2fs_precache_extents(inode);
4400
4401 ret = check_swap_activate(sis, file, span);
4402 if (ret < 0)
4403 return ret;
4404
4405 stat_inc_swapfile_inode(inode);
4406 set_inode_flag(inode, FI_PIN_FILE);
4407 f2fs_update_time(sbi, REQ_TIME);
4408 return ret;
4409 }
4410
f2fs_swap_deactivate(struct file * file)4411 static void f2fs_swap_deactivate(struct file *file)
4412 {
4413 struct inode *inode = file_inode(file);
4414
4415 stat_dec_swapfile_inode(inode);
4416 clear_inode_flag(inode, FI_PIN_FILE);
4417 }
4418 #else
f2fs_swap_activate(struct swap_info_struct * sis,struct file * file,sector_t * span)4419 static int f2fs_swap_activate(struct swap_info_struct *sis, struct file *file,
4420 sector_t *span)
4421 {
4422 return -EOPNOTSUPP;
4423 }
4424
f2fs_swap_deactivate(struct file * file)4425 static void f2fs_swap_deactivate(struct file *file)
4426 {
4427 }
4428 #endif
4429
4430 const struct address_space_operations f2fs_dblock_aops = {
4431 .read_folio = f2fs_read_data_folio,
4432 .readahead = f2fs_readahead,
4433 .writepages = f2fs_write_data_pages,
4434 .write_begin = f2fs_write_begin,
4435 .write_end = f2fs_write_end,
4436 .dirty_folio = f2fs_dirty_data_folio,
4437 .migrate_folio = filemap_migrate_folio,
4438 .invalidate_folio = f2fs_invalidate_folio,
4439 .release_folio = f2fs_release_folio,
4440 .bmap = f2fs_bmap,
4441 .swap_activate = f2fs_swap_activate,
4442 .swap_deactivate = f2fs_swap_deactivate,
4443 };
4444
f2fs_clear_page_cache_dirty_tag(struct folio * folio)4445 void f2fs_clear_page_cache_dirty_tag(struct folio *folio)
4446 {
4447 struct address_space *mapping = folio->mapping;
4448 unsigned long flags;
4449
4450 xa_lock_irqsave(&mapping->i_pages, flags);
4451 __xa_clear_mark(&mapping->i_pages, folio->index,
4452 PAGECACHE_TAG_DIRTY);
4453 xa_unlock_irqrestore(&mapping->i_pages, flags);
4454 }
4455
f2fs_init_post_read_processing(void)4456 int __init f2fs_init_post_read_processing(void)
4457 {
4458 bio_post_read_ctx_cache =
4459 kmem_cache_create("f2fs_bio_post_read_ctx",
4460 sizeof(struct bio_post_read_ctx), 0, 0, NULL);
4461 if (!bio_post_read_ctx_cache)
4462 goto fail;
4463 bio_post_read_ctx_pool =
4464 mempool_create_slab_pool(NUM_PREALLOC_POST_READ_CTXS,
4465 bio_post_read_ctx_cache);
4466 if (!bio_post_read_ctx_pool)
4467 goto fail_free_cache;
4468 return 0;
4469
4470 fail_free_cache:
4471 kmem_cache_destroy(bio_post_read_ctx_cache);
4472 fail:
4473 return -ENOMEM;
4474 }
4475
f2fs_destroy_post_read_processing(void)4476 void f2fs_destroy_post_read_processing(void)
4477 {
4478 mempool_destroy(bio_post_read_ctx_pool);
4479 kmem_cache_destroy(bio_post_read_ctx_cache);
4480 }
4481
f2fs_init_wq(struct f2fs_sb_info * sbi)4482 int f2fs_init_wq(struct f2fs_sb_info *sbi)
4483 {
4484 sbi->wq = alloc_workqueue("f2fs_wq", WQ_UNBOUND | WQ_HIGHPRI,
4485 num_online_cpus());
4486 if (!sbi->wq)
4487 return -ENOMEM;
4488
4489 sbi->evict_wq = alloc_workqueue("f2fs_evict_wq",
4490 WQ_UNBOUND | WQ_HIGHPRI, num_online_cpus());
4491 if (!sbi->evict_wq) {
4492 destroy_workqueue(sbi->wq);
4493 return -ENOMEM;
4494 }
4495 return 0;
4496 }
4497
f2fs_destroy_wq(struct f2fs_sb_info * sbi)4498 void f2fs_destroy_wq(struct f2fs_sb_info *sbi)
4499 {
4500 if (sbi->wq)
4501 destroy_workqueue(sbi->wq);
4502 if (sbi->evict_wq)
4503 destroy_workqueue(sbi->evict_wq);
4504 }
4505
f2fs_init_bio_entry_cache(void)4506 int __init f2fs_init_bio_entry_cache(void)
4507 {
4508 bio_entry_slab = f2fs_kmem_cache_create("f2fs_bio_entry_slab",
4509 sizeof(struct bio_entry));
4510
4511 if (!bio_entry_slab)
4512 return -ENOMEM;
4513
4514 ffs_entry_slab = f2fs_kmem_cache_create("f2fs_ffs_slab",
4515 sizeof(struct f2fs_folio_state));
4516
4517 if (!ffs_entry_slab) {
4518 kmem_cache_destroy(bio_entry_slab);
4519 return -ENOMEM;
4520 }
4521
4522 return 0;
4523 }
4524
f2fs_destroy_bio_entry_cache(void)4525 void f2fs_destroy_bio_entry_cache(void)
4526 {
4527 kmem_cache_destroy(bio_entry_slab);
4528 kmem_cache_destroy(ffs_entry_slab);
4529 }
4530
f2fs_iomap_begin(struct inode * inode,loff_t offset,loff_t length,unsigned int flags,struct iomap * iomap,struct iomap * srcmap)4531 static int f2fs_iomap_begin(struct inode *inode, loff_t offset, loff_t length,
4532 unsigned int flags, struct iomap *iomap,
4533 struct iomap *srcmap)
4534 {
4535 struct f2fs_map_blocks map = { NULL, };
4536 pgoff_t next_pgofs = 0;
4537 int err;
4538
4539 map.m_lblk = F2FS_BYTES_TO_BLK(offset);
4540 map.m_len = F2FS_BYTES_TO_BLK(offset + length - 1) - map.m_lblk + 1;
4541 map.m_next_pgofs = &next_pgofs;
4542 map.m_seg_type = f2fs_rw_hint_to_seg_type(F2FS_I_SB(inode),
4543 inode->i_write_hint);
4544 if (flags & IOMAP_WRITE && iomap->private) {
4545 map.m_last_pblk = (unsigned long)iomap->private;
4546 iomap->private = NULL;
4547 }
4548
4549 /*
4550 * If the blocks being overwritten are already allocated,
4551 * f2fs_map_lock and f2fs_balance_fs are not necessary.
4552 */
4553 if ((flags & IOMAP_WRITE) &&
4554 !__f2fs_overwrite_io(inode, offset, length, true))
4555 map.m_may_create = true;
4556
4557 err = f2fs_map_blocks(inode, &map, F2FS_GET_BLOCK_DIO);
4558 if (err)
4559 return err;
4560
4561 iomap->offset = F2FS_BLK_TO_BYTES(map.m_lblk);
4562
4563 /*
4564 * Sometimes I/O to an encrypted file has to be broken up to guarantee
4565 * DUN contiguity. Handle this by limiting the length of the mapping
4566 * returned.
4567 */
4568 map.m_len = fscrypt_limit_io_blocks(inode, map.m_lblk, map.m_len);
4569
4570 /*
4571 * We should never see delalloc or compressed extents here based on
4572 * prior flushing and checks.
4573 */
4574 if (WARN_ON_ONCE(map.m_pblk == COMPRESS_ADDR))
4575 return -EINVAL;
4576
4577 if (map.m_flags & F2FS_MAP_MAPPED) {
4578 if (WARN_ON_ONCE(map.m_pblk == NEW_ADDR))
4579 return -EINVAL;
4580
4581 iomap->length = F2FS_BLK_TO_BYTES(map.m_len);
4582 iomap->type = IOMAP_MAPPED;
4583 iomap->flags |= IOMAP_F_MERGED;
4584 iomap->bdev = map.m_bdev;
4585 iomap->addr = F2FS_BLK_TO_BYTES(map.m_pblk);
4586
4587 if (flags & IOMAP_WRITE && map.m_last_pblk)
4588 iomap->private = (void *)map.m_last_pblk;
4589 } else {
4590 if (flags & IOMAP_WRITE)
4591 return -ENOTBLK;
4592
4593 if (map.m_pblk == NULL_ADDR) {
4594 iomap->length = F2FS_BLK_TO_BYTES(next_pgofs) -
4595 iomap->offset;
4596 iomap->type = IOMAP_HOLE;
4597 } else if (map.m_pblk == NEW_ADDR) {
4598 iomap->length = F2FS_BLK_TO_BYTES(map.m_len);
4599 iomap->type = IOMAP_UNWRITTEN;
4600 } else {
4601 f2fs_bug_on(F2FS_I_SB(inode), 1);
4602 }
4603 iomap->addr = IOMAP_NULL_ADDR;
4604 }
4605
4606 if (map.m_flags & F2FS_MAP_NEW)
4607 iomap->flags |= IOMAP_F_NEW;
4608 if ((inode_state_read_once(inode) & I_DIRTY_DATASYNC) ||
4609 offset + length > i_size_read(inode))
4610 iomap->flags |= IOMAP_F_DIRTY;
4611
4612 return 0;
4613 }
4614
4615 static DEFINE_IOMAP_ITER_NEXT(f2fs_iomap_next, f2fs_iomap_begin);
4616
4617 const struct iomap_ops f2fs_iomap_ops = {
4618 .iomap_next = f2fs_iomap_next,
4619 };
4620