1 // SPDX-License-Identifier: GPL-2.0-only
2 /*
3 * Copyright (c) 2025, Christoph Hellwig.
4 * Copyright (c) 2025, Western Digital Corporation or its affiliates.
5 *
6 * Zoned Loop Device driver - exports a zoned block device using one file per
7 * zone as backing storage.
8 */
9 #define pr_fmt(fmt) KBUILD_MODNAME ": " fmt
10
11 #include <linux/module.h>
12 #include <linux/blk-mq.h>
13 #include <linux/blkzoned.h>
14 #include <linux/pagemap.h>
15 #include <linux/miscdevice.h>
16 #include <linux/falloc.h>
17 #include <linux/mutex.h>
18 #include <linux/parser.h>
19 #include <linux/seq_file.h>
20 #include <linux/xattr.h>
21
22 /*
23 * Options for adding (and removing) a device.
24 */
25 enum {
26 ZLOOP_OPT_ERR = 0,
27 ZLOOP_OPT_ID = (1 << 0),
28 ZLOOP_OPT_CAPACITY = (1 << 1),
29 ZLOOP_OPT_ZONE_SIZE = (1 << 2),
30 ZLOOP_OPT_ZONE_CAPACITY = (1 << 3),
31 ZLOOP_OPT_NR_CONV_ZONES = (1 << 4),
32 ZLOOP_OPT_BASE_DIR = (1 << 5),
33 ZLOOP_OPT_NR_QUEUES = (1 << 6),
34 ZLOOP_OPT_QUEUE_DEPTH = (1 << 7),
35 ZLOOP_OPT_BUFFERED_IO = (1 << 8),
36 ZLOOP_OPT_ZONE_APPEND = (1 << 9),
37 ZLOOP_OPT_ORDERED_ZONE_APPEND = (1 << 10),
38 ZLOOP_OPT_DISCARD_WRITE_CACHE = (1 << 11),
39 ZLOOP_OPT_MAX_OPEN_ZONES = (1 << 12),
40 };
41
42 static const match_table_t zloop_opt_tokens = {
43 { ZLOOP_OPT_ID, "id=%d" },
44 { ZLOOP_OPT_CAPACITY, "capacity_mb=%u" },
45 { ZLOOP_OPT_ZONE_SIZE, "zone_size_mb=%u" },
46 { ZLOOP_OPT_ZONE_CAPACITY, "zone_capacity_mb=%u" },
47 { ZLOOP_OPT_NR_CONV_ZONES, "conv_zones=%u" },
48 { ZLOOP_OPT_BASE_DIR, "base_dir=%s" },
49 { ZLOOP_OPT_NR_QUEUES, "nr_queues=%u" },
50 { ZLOOP_OPT_QUEUE_DEPTH, "queue_depth=%u" },
51 { ZLOOP_OPT_BUFFERED_IO, "buffered_io" },
52 { ZLOOP_OPT_ZONE_APPEND, "zone_append=%u" },
53 { ZLOOP_OPT_ORDERED_ZONE_APPEND, "ordered_zone_append" },
54 { ZLOOP_OPT_DISCARD_WRITE_CACHE, "discard_write_cache" },
55 { ZLOOP_OPT_MAX_OPEN_ZONES, "max_open_zones=%u" },
56 { ZLOOP_OPT_ERR, NULL }
57 };
58
59 /* Default values for the "add" operation. */
60 #define ZLOOP_DEF_ID -1
61 #define ZLOOP_DEF_ZONE_SIZE ((256ULL * SZ_1M) >> SECTOR_SHIFT)
62 #define ZLOOP_DEF_NR_ZONES 64
63 #define ZLOOP_DEF_NR_CONV_ZONES 8
64 #define ZLOOP_DEF_MAX_OPEN_ZONES 0
65 #define ZLOOP_DEF_BASE_DIR "/var/local/zloop"
66 #define ZLOOP_DEF_NR_QUEUES 1
67 #define ZLOOP_DEF_QUEUE_DEPTH 128
68 #define ZLOOP_DEF_BUFFERED_IO false
69 #define ZLOOP_DEF_ZONE_APPEND true
70 #define ZLOOP_DEF_ORDERED_ZONE_APPEND false
71
72 /* Arbitrary limit on the zone size (16GB). */
73 #define ZLOOP_MAX_ZONE_SIZE_MB 16384
74
75 struct zloop_options {
76 unsigned int mask;
77 int id;
78 sector_t capacity;
79 sector_t zone_size;
80 sector_t zone_capacity;
81 unsigned int nr_conv_zones;
82 unsigned int max_open_zones;
83 char *base_dir;
84 unsigned int nr_queues;
85 unsigned int queue_depth;
86 bool buffered_io;
87 bool zone_append;
88 bool ordered_zone_append;
89 bool discard_write_cache;
90 };
91
92 /*
93 * Device states.
94 */
95 enum {
96 Zlo_creating = 0,
97 Zlo_live,
98 Zlo_deleting,
99 };
100
101 enum zloop_zone_flags {
102 ZLOOP_ZONE_CONV = 0,
103 ZLOOP_ZONE_SEQ_ERROR,
104 };
105
106 /*
107 * Zone descriptor.
108 * Locking order: z.lock -> z.wp_lock -> zlo.open_zones_lock
109 */
110 struct zloop_zone {
111 struct list_head open_zone_entry;
112 struct file *file;
113
114 unsigned long flags;
115 struct mutex lock;
116 spinlock_t wp_lock;
117 enum blk_zone_cond cond;
118 sector_t start;
119 sector_t wp;
120
121 gfp_t old_gfp_mask;
122 };
123
124 struct zloop_device {
125 unsigned int id;
126 unsigned int state;
127
128 struct blk_mq_tag_set tag_set;
129 struct gendisk *disk;
130
131 struct workqueue_struct *workqueue;
132 bool buffered_io;
133 bool zone_append;
134 bool ordered_zone_append;
135 bool discard_write_cache;
136
137 const char *base_dir;
138 struct file *data_dir;
139
140 unsigned int zone_shift;
141 sector_t zone_size;
142 sector_t zone_capacity;
143 unsigned int nr_zones;
144 unsigned int nr_conv_zones;
145 unsigned int max_open_zones;
146 unsigned int block_size;
147 unsigned int dio_mem_align;
148
149 spinlock_t open_zones_lock;
150 struct list_head open_zones_lru_list;
151 unsigned int nr_open_zones;
152
153 struct zloop_zone zones[] __counted_by(nr_zones);
154 };
155
156 struct zloop_cmd {
157 struct work_struct work;
158 atomic_t ref;
159 sector_t sector;
160 sector_t nr_sectors;
161 long ret;
162 struct kiocb iocb;
163 struct bio_vec *bvec;
164 };
165
166 static DEFINE_IDR(zloop_index_idr);
167 static DEFINE_MUTEX(zloop_ctl_mutex);
168
rq_zone_no(struct request * rq)169 static unsigned int rq_zone_no(struct request *rq)
170 {
171 struct zloop_device *zlo = rq->q->queuedata;
172
173 return blk_rq_pos(rq) >> zlo->zone_shift;
174 }
175
176 /*
177 * Open an already open zone. This is mostly a no-op, except for the imp open ->
178 * exp open condition change that may happen. We also move a zone at the tail of
179 * the list of open zones so that if we need to
180 * implicitly close one open zone, we can do so in LRU order.
181 */
zloop_lru_rotate_open_zone(struct zloop_device * zlo,struct zloop_zone * zone)182 static inline void zloop_lru_rotate_open_zone(struct zloop_device *zlo,
183 struct zloop_zone *zone)
184 {
185 if (zlo->max_open_zones) {
186 spin_lock(&zlo->open_zones_lock);
187 list_move_tail(&zone->open_zone_entry,
188 &zlo->open_zones_lru_list);
189 spin_unlock(&zlo->open_zones_lock);
190 }
191 }
192
zloop_lru_remove_open_zone(struct zloop_device * zlo,struct zloop_zone * zone)193 static inline void zloop_lru_remove_open_zone(struct zloop_device *zlo,
194 struct zloop_zone *zone)
195 {
196 if (zone->cond == BLK_ZONE_COND_IMP_OPEN ||
197 zone->cond == BLK_ZONE_COND_EXP_OPEN) {
198 spin_lock(&zlo->open_zones_lock);
199 list_del_init(&zone->open_zone_entry);
200 zlo->nr_open_zones--;
201 spin_unlock(&zlo->open_zones_lock);
202 }
203 }
204
zloop_can_open_zone(struct zloop_device * zlo)205 static inline bool zloop_can_open_zone(struct zloop_device *zlo)
206 {
207 return !zlo->max_open_zones || zlo->nr_open_zones < zlo->max_open_zones;
208 }
209
210 /*
211 * If we have reached the maximum open zones limit, attempt to close an
212 * implicitly open zone (if we have any) so that we can implicitly open another
213 * zone without exceeding the maximum number of open zones.
214 */
zloop_close_imp_open_zone(struct zloop_device * zlo)215 static bool zloop_close_imp_open_zone(struct zloop_device *zlo)
216 {
217 struct zloop_zone *zone;
218
219 lockdep_assert_held(&zlo->open_zones_lock);
220
221 if (zloop_can_open_zone(zlo))
222 return true;
223
224 list_for_each_entry(zone, &zlo->open_zones_lru_list, open_zone_entry) {
225 if (zone->cond == BLK_ZONE_COND_IMP_OPEN) {
226 zone->cond = BLK_ZONE_COND_CLOSED;
227 list_del_init(&zone->open_zone_entry);
228 zlo->nr_open_zones--;
229 return true;
230 }
231 }
232
233 return false;
234 }
235
zloop_open_closed_or_empty_zone(struct zloop_device * zlo,struct zloop_zone * zone,bool explicit)236 static bool zloop_open_closed_or_empty_zone(struct zloop_device *zlo,
237 struct zloop_zone *zone,
238 bool explicit)
239 {
240 spin_lock(&zlo->open_zones_lock);
241
242 if (explicit) {
243 /*
244 * Explicit open: we cannot allow this if we have reached the
245 * maximum open zones limit.
246 */
247 if (!zloop_can_open_zone(zlo))
248 goto fail;
249 zone->cond = BLK_ZONE_COND_EXP_OPEN;
250 } else {
251 /*
252 * Implicit open case: if we have reached the maximum open zones
253 * limit, try to close an implicitly open zone first.
254 */
255 if (!zloop_close_imp_open_zone(zlo))
256 goto fail;
257 zone->cond = BLK_ZONE_COND_IMP_OPEN;
258 }
259
260 zlo->nr_open_zones++;
261 list_add_tail(&zone->open_zone_entry,
262 &zlo->open_zones_lru_list);
263
264 spin_unlock(&zlo->open_zones_lock);
265
266 return true;
267
268 fail:
269 spin_unlock(&zlo->open_zones_lock);
270
271 return false;
272 }
273
zloop_do_open_zone(struct zloop_device * zlo,struct zloop_zone * zone,bool explicit)274 static bool zloop_do_open_zone(struct zloop_device *zlo,
275 struct zloop_zone *zone, bool explicit)
276 {
277 switch (zone->cond) {
278 case BLK_ZONE_COND_IMP_OPEN:
279 case BLK_ZONE_COND_EXP_OPEN:
280 if (explicit)
281 zone->cond = BLK_ZONE_COND_EXP_OPEN;
282 zloop_lru_rotate_open_zone(zlo, zone);
283 return true;
284 case BLK_ZONE_COND_EMPTY:
285 case BLK_ZONE_COND_CLOSED:
286 return zloop_open_closed_or_empty_zone(zlo, zone, explicit);
287 default:
288 return false;
289 }
290 }
291
zloop_mark_full(struct zloop_device * zlo,struct zloop_zone * zone)292 static void zloop_mark_full(struct zloop_device *zlo, struct zloop_zone *zone)
293 {
294 lockdep_assert_held(&zone->wp_lock);
295
296 zloop_lru_remove_open_zone(zlo, zone);
297 zone->cond = BLK_ZONE_COND_FULL;
298 zone->wp = ULLONG_MAX;
299 }
300
zloop_mark_empty(struct zloop_device * zlo,struct zloop_zone * zone)301 static void zloop_mark_empty(struct zloop_device *zlo, struct zloop_zone *zone)
302 {
303 lockdep_assert_held(&zone->wp_lock);
304
305 zloop_lru_remove_open_zone(zlo, zone);
306 zone->cond = BLK_ZONE_COND_EMPTY;
307 zone->wp = zone->start;
308 }
309
zloop_update_seq_zone(struct zloop_device * zlo,unsigned int zone_no)310 static int zloop_update_seq_zone(struct zloop_device *zlo, unsigned int zone_no)
311 {
312 struct zloop_zone *zone = &zlo->zones[zone_no];
313 struct kstat stat;
314 sector_t file_sectors;
315 int ret;
316
317 lockdep_assert_held(&zone->lock);
318
319 ret = vfs_getattr(&zone->file->f_path, &stat, STATX_SIZE, 0);
320 if (ret < 0) {
321 pr_err("Failed to get zone %u file stat (err=%d)\n",
322 zone_no, ret);
323 set_bit(ZLOOP_ZONE_SEQ_ERROR, &zone->flags);
324 return ret;
325 }
326
327 file_sectors = stat.size >> SECTOR_SHIFT;
328 if (file_sectors > zlo->zone_capacity) {
329 pr_err("Zone %u file too large (%llu sectors > %llu)\n",
330 zone_no, file_sectors, zlo->zone_capacity);
331 return -EINVAL;
332 }
333
334 if (!IS_ALIGNED(stat.size, zlo->block_size)) {
335 pr_err("Zone %u file size (%llu) not aligned to block size %u\n",
336 zone_no, stat.size, zlo->block_size);
337 return -EINVAL;
338 }
339
340 spin_lock(&zone->wp_lock);
341 if (!file_sectors) {
342 zloop_mark_empty(zlo, zone);
343 } else if (file_sectors == zlo->zone_capacity) {
344 zloop_mark_full(zlo, zone);
345 } else {
346 if (zone->cond != BLK_ZONE_COND_IMP_OPEN &&
347 zone->cond != BLK_ZONE_COND_EXP_OPEN)
348 zone->cond = BLK_ZONE_COND_CLOSED;
349 zone->wp = zone->start + file_sectors;
350 }
351 spin_unlock(&zone->wp_lock);
352
353 return 0;
354 }
355
zloop_open_zone(struct zloop_device * zlo,unsigned int zone_no)356 static int zloop_open_zone(struct zloop_device *zlo, unsigned int zone_no)
357 {
358 struct zloop_zone *zone = &zlo->zones[zone_no];
359 int ret = 0;
360
361 if (test_bit(ZLOOP_ZONE_CONV, &zone->flags))
362 return -EIO;
363
364 mutex_lock(&zone->lock);
365
366 if (test_and_clear_bit(ZLOOP_ZONE_SEQ_ERROR, &zone->flags)) {
367 ret = zloop_update_seq_zone(zlo, zone_no);
368 if (ret)
369 goto unlock;
370 }
371
372 if (!zloop_do_open_zone(zlo, zone, true))
373 ret = -EIO;
374
375 unlock:
376 mutex_unlock(&zone->lock);
377
378 return ret;
379 }
380
zloop_close_zone(struct zloop_device * zlo,unsigned int zone_no)381 static int zloop_close_zone(struct zloop_device *zlo, unsigned int zone_no)
382 {
383 struct zloop_zone *zone = &zlo->zones[zone_no];
384 int ret = 0;
385
386 if (test_bit(ZLOOP_ZONE_CONV, &zone->flags))
387 return -EIO;
388
389 mutex_lock(&zone->lock);
390
391 if (test_and_clear_bit(ZLOOP_ZONE_SEQ_ERROR, &zone->flags)) {
392 ret = zloop_update_seq_zone(zlo, zone_no);
393 if (ret)
394 goto unlock;
395 }
396
397 switch (zone->cond) {
398 case BLK_ZONE_COND_CLOSED:
399 break;
400 case BLK_ZONE_COND_IMP_OPEN:
401 case BLK_ZONE_COND_EXP_OPEN:
402 spin_lock(&zone->wp_lock);
403 zloop_lru_remove_open_zone(zlo, zone);
404 if (zone->wp == zone->start)
405 zone->cond = BLK_ZONE_COND_EMPTY;
406 else
407 zone->cond = BLK_ZONE_COND_CLOSED;
408 spin_unlock(&zone->wp_lock);
409 break;
410 case BLK_ZONE_COND_EMPTY:
411 case BLK_ZONE_COND_FULL:
412 default:
413 ret = -EIO;
414 break;
415 }
416
417 unlock:
418 mutex_unlock(&zone->lock);
419
420 return ret;
421 }
422
zloop_reset_zone(struct zloop_device * zlo,unsigned int zone_no)423 static int zloop_reset_zone(struct zloop_device *zlo, unsigned int zone_no)
424 {
425 struct zloop_zone *zone = &zlo->zones[zone_no];
426 int ret = 0;
427
428 if (test_bit(ZLOOP_ZONE_CONV, &zone->flags))
429 return -EIO;
430
431 mutex_lock(&zone->lock);
432
433 if (!test_bit(ZLOOP_ZONE_SEQ_ERROR, &zone->flags) &&
434 zone->cond == BLK_ZONE_COND_EMPTY)
435 goto unlock;
436
437 if (vfs_truncate(&zone->file->f_path, 0)) {
438 set_bit(ZLOOP_ZONE_SEQ_ERROR, &zone->flags);
439 ret = -EIO;
440 goto unlock;
441 }
442
443 spin_lock(&zone->wp_lock);
444 zloop_mark_empty(zlo, zone);
445 clear_bit(ZLOOP_ZONE_SEQ_ERROR, &zone->flags);
446 spin_unlock(&zone->wp_lock);
447
448 unlock:
449 mutex_unlock(&zone->lock);
450
451 return ret;
452 }
453
zloop_reset_all_zones(struct zloop_device * zlo)454 static int zloop_reset_all_zones(struct zloop_device *zlo)
455 {
456 unsigned int i;
457 int ret;
458
459 for (i = zlo->nr_conv_zones; i < zlo->nr_zones; i++) {
460 ret = zloop_reset_zone(zlo, i);
461 if (ret)
462 return ret;
463 }
464
465 return 0;
466 }
467
zloop_finish_zone(struct zloop_device * zlo,unsigned int zone_no)468 static int zloop_finish_zone(struct zloop_device *zlo, unsigned int zone_no)
469 {
470 struct zloop_zone *zone = &zlo->zones[zone_no];
471 int ret = 0;
472
473 if (test_bit(ZLOOP_ZONE_CONV, &zone->flags))
474 return -EIO;
475
476 mutex_lock(&zone->lock);
477
478 if (!test_bit(ZLOOP_ZONE_SEQ_ERROR, &zone->flags) &&
479 zone->cond == BLK_ZONE_COND_FULL)
480 goto unlock;
481
482 if (vfs_truncate(&zone->file->f_path,
483 zlo->zone_capacity << SECTOR_SHIFT)) {
484 set_bit(ZLOOP_ZONE_SEQ_ERROR, &zone->flags);
485 ret = -EIO;
486 goto unlock;
487 }
488
489 spin_lock(&zone->wp_lock);
490 zloop_mark_full(zlo, zone);
491 clear_bit(ZLOOP_ZONE_SEQ_ERROR, &zone->flags);
492 spin_unlock(&zone->wp_lock);
493
494 unlock:
495 mutex_unlock(&zone->lock);
496
497 return ret;
498 }
499
zloop_put_cmd(struct zloop_cmd * cmd)500 static void zloop_put_cmd(struct zloop_cmd *cmd)
501 {
502 struct request *rq = blk_mq_rq_from_pdu(cmd);
503
504 if (!atomic_dec_and_test(&cmd->ref))
505 return;
506 kfree(cmd->bvec);
507 cmd->bvec = NULL;
508 if (likely(!blk_should_fake_timeout(rq->q)))
509 blk_mq_complete_request(rq);
510 }
511
zloop_rw_complete(struct kiocb * iocb,long ret)512 static void zloop_rw_complete(struct kiocb *iocb, long ret)
513 {
514 struct zloop_cmd *cmd = container_of(iocb, struct zloop_cmd, iocb);
515
516 cmd->ret = ret;
517 zloop_put_cmd(cmd);
518 }
519
zloop_do_rw(struct zloop_cmd * cmd)520 static int zloop_do_rw(struct zloop_cmd *cmd)
521 {
522 struct request *rq = blk_mq_rq_from_pdu(cmd);
523 int rw = req_op(rq) == REQ_OP_READ ? ITER_DEST : ITER_SOURCE;
524 unsigned int nr_bvec = blk_rq_nr_bvec(rq);
525 struct zloop_device *zlo = rq->q->queuedata;
526 struct zloop_zone *zone = &zlo->zones[rq_zone_no(rq)];
527 struct req_iterator rq_iter;
528 struct iov_iter iter;
529
530 if (rq->bio != rq->biotail) {
531 struct bio_vec tmp, *bvec;
532
533 cmd->bvec = kmalloc_objs(*cmd->bvec, nr_bvec, GFP_NOIO);
534 if (!cmd->bvec)
535 return -EIO;
536
537 /*
538 * The bios of the request may be started from the middle of
539 * the 'bvec' because of bio splitting, so we can't directly
540 * copy bio->bi_iov_vec to new bvec. The rq_for_each_bvec
541 * API will take care of all details for us.
542 */
543 bvec = cmd->bvec;
544 rq_for_each_bvec(tmp, rq, rq_iter) {
545 *bvec = tmp;
546 bvec++;
547 }
548 iov_iter_bvec(&iter, rw, cmd->bvec, nr_bvec, blk_rq_bytes(rq));
549 } else {
550 /*
551 * Same here, this bio may be started from the middle of the
552 * 'bvec' because of bio splitting, so offset from the bvec
553 * must be passed to iov iterator
554 */
555 iov_iter_bvec(&iter, rw,
556 __bvec_iter_bvec(rq->bio->bi_io_vec, rq->bio->bi_iter),
557 nr_bvec, blk_rq_bytes(rq));
558 iter.iov_offset = rq->bio->bi_iter.bi_offset;
559 }
560
561 cmd->iocb.ki_pos = (cmd->sector - zone->start) << SECTOR_SHIFT;
562 cmd->iocb.ki_filp = zone->file;
563 cmd->iocb.ki_complete = zloop_rw_complete;
564 if (!zlo->buffered_io)
565 cmd->iocb.ki_flags = IOCB_DIRECT;
566 cmd->iocb.ki_ioprio = IOPRIO_PRIO_VALUE(IOPRIO_CLASS_NONE, 0);
567
568 if (rw == ITER_SOURCE)
569 return zone->file->f_op->write_iter(&cmd->iocb, &iter);
570 return zone->file->f_op->read_iter(&cmd->iocb, &iter);
571 }
572
zloop_seq_write_prep(struct zloop_cmd * cmd)573 static int zloop_seq_write_prep(struct zloop_cmd *cmd)
574 {
575 struct request *rq = blk_mq_rq_from_pdu(cmd);
576 struct zloop_device *zlo = rq->q->queuedata;
577 unsigned int zone_no = rq_zone_no(rq);
578 sector_t nr_sectors = blk_rq_sectors(rq);
579 bool is_append = req_op(rq) == REQ_OP_ZONE_APPEND;
580 struct zloop_zone *zone = &zlo->zones[zone_no];
581 sector_t zone_end = zone->start + zlo->zone_capacity;
582 int ret = 0;
583
584 spin_lock(&zone->wp_lock);
585
586 /*
587 * Zone append operations always go at the current write pointer, but
588 * regular write operations must already be aligned to the write pointer
589 * when submitted.
590 */
591 if (is_append) {
592 /*
593 * If ordered zone append is in use, we already checked and set
594 * the target sector in zloop_queue_rq().
595 */
596 if (!zlo->ordered_zone_append) {
597 if (zone->cond == BLK_ZONE_COND_FULL ||
598 zone->wp + nr_sectors > zone_end) {
599 ret = -EIO;
600 goto out_unlock;
601 }
602 cmd->sector = zone->wp;
603 }
604 } else {
605 if (cmd->sector != zone->wp) {
606 pr_err("Zone %u: unaligned write: sect %llu, wp %llu\n",
607 zone_no, cmd->sector, zone->wp);
608 ret = -EIO;
609 goto out_unlock;
610 }
611 }
612
613 /* Implicitly open the target zone. */
614 if (!zloop_do_open_zone(zlo, zone, false)) {
615 ret = -EIO;
616 goto out_unlock;
617 }
618
619 /*
620 * Advance the write pointer, unless ordered zone append is in use. If
621 * the write fails, the write pointer position will be corrected when
622 * the next I/O starts execution.
623 */
624 if (!is_append || !zlo->ordered_zone_append) {
625 zone->wp += nr_sectors;
626 if (zone->wp == zone_end)
627 zloop_mark_full(zlo, zone);
628 }
629 out_unlock:
630 spin_unlock(&zone->wp_lock);
631 return ret;
632 }
633
zloop_rw(struct zloop_cmd * cmd)634 static void zloop_rw(struct zloop_cmd *cmd)
635 {
636 struct request *rq = blk_mq_rq_from_pdu(cmd);
637 struct zloop_device *zlo = rq->q->queuedata;
638 unsigned int zone_no = rq_zone_no(rq);
639 sector_t nr_sectors = blk_rq_sectors(rq);
640 bool is_append = req_op(rq) == REQ_OP_ZONE_APPEND;
641 bool is_write = req_op(rq) == REQ_OP_WRITE || is_append;
642 struct zloop_zone *zone;
643 int ret = -EIO;
644
645 atomic_set(&cmd->ref, 2);
646 cmd->sector = blk_rq_pos(rq);
647 cmd->nr_sectors = nr_sectors;
648 cmd->ret = 0;
649
650 if (WARN_ON_ONCE(is_append && !zlo->zone_append))
651 goto out;
652
653 /* We should never get an I/O beyond the device capacity. */
654 if (WARN_ON_ONCE(zone_no >= zlo->nr_zones))
655 goto out;
656
657 zone = &zlo->zones[zone_no];
658
659 /*
660 * The block layer should never send requests that are not fully
661 * contained within the zone.
662 */
663 if (WARN_ON_ONCE(cmd->sector + nr_sectors >
664 zone->start + zlo->zone_size))
665 goto out;
666
667 if (test_and_clear_bit(ZLOOP_ZONE_SEQ_ERROR, &zone->flags)) {
668 mutex_lock(&zone->lock);
669 ret = zloop_update_seq_zone(zlo, zone_no);
670 mutex_unlock(&zone->lock);
671 if (ret)
672 goto out;
673 }
674
675 if (!test_bit(ZLOOP_ZONE_CONV, &zone->flags) && is_write) {
676 mutex_lock(&zone->lock);
677 ret = zloop_seq_write_prep(cmd);
678 if (!ret)
679 ret = zloop_do_rw(cmd);
680 mutex_unlock(&zone->lock);
681 } else {
682 ret = zloop_do_rw(cmd);
683 }
684 out:
685 if (ret != -EIOCBQUEUED)
686 zloop_rw_complete(&cmd->iocb, ret);
687 zloop_put_cmd(cmd);
688 }
689
zloop_zone_is_active(struct zloop_zone * zone)690 static inline bool zloop_zone_is_active(struct zloop_zone *zone)
691 {
692 switch (zone->cond) {
693 case BLK_ZONE_COND_EXP_OPEN:
694 case BLK_ZONE_COND_IMP_OPEN:
695 case BLK_ZONE_COND_CLOSED:
696 return true;
697 default:
698 return false;
699 }
700 }
701
zloop_record_safe_wps(struct zloop_device * zlo)702 static int zloop_record_safe_wps(struct zloop_device *zlo)
703 {
704 unsigned int i;
705 int ret;
706
707 for (i = 0; i < zlo->nr_zones; i++) {
708 struct zloop_zone *zone = &zlo->zones[i];
709 struct file *file = zone->file;
710
711 if (!zloop_zone_is_active(zone))
712 continue;
713 ret = vfs_setxattr(file_mnt_idmap(file), file_dentry(file),
714 "user.zloop.wp", &zone->wp, sizeof(zone->wp), 0);
715 if (ret) {
716 pr_err("%pg: failed to record write pointer (%d)\n",
717 zlo->disk->part0, ret);
718 return ret;
719 }
720 }
721
722 return 0;
723 }
724
725 /*
726 * Sync the entire FS containing the zone files instead of walking all files.
727 */
zloop_flush(struct zloop_device * zlo)728 static int zloop_flush(struct zloop_device *zlo)
729 {
730 struct super_block *sb = file_inode(zlo->data_dir)->i_sb;
731 int ret;
732
733 if (zlo->discard_write_cache) {
734 ret = zloop_record_safe_wps(zlo);
735 if (ret)
736 return ret;
737 }
738
739 down_read(&sb->s_umount);
740 ret = sync_filesystem(sb);
741 up_read(&sb->s_umount);
742
743 return ret;
744 }
745
zloop_handle_cmd(struct zloop_cmd * cmd)746 static void zloop_handle_cmd(struct zloop_cmd *cmd)
747 {
748 struct request *rq = blk_mq_rq_from_pdu(cmd);
749 struct zloop_device *zlo = rq->q->queuedata;
750
751 /* We can block in this context, so ignore REQ_NOWAIT. */
752 if (rq->cmd_flags & REQ_NOWAIT)
753 rq->cmd_flags &= ~REQ_NOWAIT;
754
755 switch (req_op(rq)) {
756 case REQ_OP_READ:
757 case REQ_OP_WRITE:
758 case REQ_OP_ZONE_APPEND:
759 /*
760 * zloop_rw() always executes asynchronously or completes
761 * directly.
762 */
763 zloop_rw(cmd);
764 return;
765 case REQ_OP_FLUSH:
766 cmd->ret = zloop_flush(zlo);
767 break;
768 case REQ_OP_ZONE_RESET:
769 cmd->ret = zloop_reset_zone(zlo, rq_zone_no(rq));
770 break;
771 case REQ_OP_ZONE_RESET_ALL:
772 cmd->ret = zloop_reset_all_zones(zlo);
773 break;
774 case REQ_OP_ZONE_FINISH:
775 cmd->ret = zloop_finish_zone(zlo, rq_zone_no(rq));
776 break;
777 case REQ_OP_ZONE_OPEN:
778 cmd->ret = zloop_open_zone(zlo, rq_zone_no(rq));
779 break;
780 case REQ_OP_ZONE_CLOSE:
781 cmd->ret = zloop_close_zone(zlo, rq_zone_no(rq));
782 break;
783 default:
784 WARN_ON_ONCE(1);
785 pr_err("Unsupported operation %d\n", req_op(rq));
786 cmd->ret = -EOPNOTSUPP;
787 break;
788 }
789
790 blk_mq_complete_request(rq);
791 }
792
zloop_cmd_workfn(struct work_struct * work)793 static void zloop_cmd_workfn(struct work_struct *work)
794 {
795 struct zloop_cmd *cmd = container_of(work, struct zloop_cmd, work);
796 int orig_flags = current->flags;
797
798 current->flags |= PF_LOCAL_THROTTLE | PF_MEMALLOC_NOIO;
799 zloop_handle_cmd(cmd);
800 current->flags = orig_flags;
801 }
802
zloop_complete_rq(struct request * rq)803 static void zloop_complete_rq(struct request *rq)
804 {
805 struct zloop_cmd *cmd = blk_mq_rq_to_pdu(rq);
806 struct zloop_device *zlo = rq->q->queuedata;
807 unsigned int zone_no = cmd->sector >> zlo->zone_shift;
808 struct zloop_zone *zone = &zlo->zones[zone_no];
809 blk_status_t sts = BLK_STS_OK;
810
811 switch (req_op(rq)) {
812 case REQ_OP_READ:
813 if (cmd->ret < 0)
814 pr_err("Zone %u: failed read sector %llu, %llu sectors\n",
815 zone_no, cmd->sector, cmd->nr_sectors);
816
817 if (cmd->ret >= 0 && cmd->ret != blk_rq_bytes(rq)) {
818 /* short read */
819 struct bio *bio;
820
821 __rq_for_each_bio(bio, rq)
822 zero_fill_bio(bio);
823 }
824 break;
825 case REQ_OP_WRITE:
826 case REQ_OP_ZONE_APPEND:
827 if (cmd->ret < 0)
828 pr_err("Zone %u: failed %swrite sector %llu, %llu sectors\n",
829 zone_no,
830 req_op(rq) == REQ_OP_WRITE ? "" : "append ",
831 cmd->sector, cmd->nr_sectors);
832
833 if (cmd->ret >= 0 && cmd->ret != blk_rq_bytes(rq)) {
834 pr_err("Zone %u: partial write %ld/%u B\n",
835 zone_no, cmd->ret, blk_rq_bytes(rq));
836 cmd->ret = -EIO;
837 }
838
839 if (cmd->ret < 0 && !test_bit(ZLOOP_ZONE_CONV, &zone->flags)) {
840 /*
841 * A write to a sequential zone file failed: mark the
842 * zone as having an error. This will be corrected and
843 * cleared when the next IO is submitted.
844 */
845 set_bit(ZLOOP_ZONE_SEQ_ERROR, &zone->flags);
846 break;
847 }
848 if (req_op(rq) == REQ_OP_ZONE_APPEND)
849 rq->__sector = cmd->sector;
850
851 break;
852 default:
853 break;
854 }
855
856 if (cmd->ret < 0)
857 sts = errno_to_blk_status(cmd->ret);
858 blk_mq_end_request(rq, sts);
859 }
860
zloop_set_zone_append_sector(struct request * rq)861 static bool zloop_set_zone_append_sector(struct request *rq)
862 {
863 struct zloop_device *zlo = rq->q->queuedata;
864 unsigned int zone_no = rq_zone_no(rq);
865 struct zloop_zone *zone = &zlo->zones[zone_no];
866 sector_t zone_end = zone->start + zlo->zone_capacity;
867 sector_t nr_sectors = blk_rq_sectors(rq);
868
869 spin_lock(&zone->wp_lock);
870
871 if (zone->cond == BLK_ZONE_COND_FULL ||
872 zone->wp + nr_sectors > zone_end) {
873 spin_unlock(&zone->wp_lock);
874 return false;
875 }
876
877 rq->__sector = zone->wp;
878 zone->wp += blk_rq_sectors(rq);
879 if (zone->wp >= zone_end)
880 zloop_mark_full(zlo, zone);
881
882 spin_unlock(&zone->wp_lock);
883
884 return true;
885 }
886
zloop_queue_rq(struct blk_mq_hw_ctx * hctx,const struct blk_mq_queue_data * bd)887 static blk_status_t zloop_queue_rq(struct blk_mq_hw_ctx *hctx,
888 const struct blk_mq_queue_data *bd)
889 {
890 struct request *rq = bd->rq;
891 struct zloop_cmd *cmd = blk_mq_rq_to_pdu(rq);
892 struct zloop_device *zlo = rq->q->queuedata;
893
894 if (data_race(READ_ONCE(zlo->state)) == Zlo_deleting) {
895 rq->rq_flags |= RQF_QUIET;
896 return BLK_STS_IOERR;
897 }
898
899 /*
900 * If we need to strongly order zone append operations, set the request
901 * sector to the zone write pointer location now instead of when the
902 * command work runs.
903 */
904 if (zlo->ordered_zone_append && req_op(rq) == REQ_OP_ZONE_APPEND) {
905 if (!zloop_set_zone_append_sector(rq))
906 return BLK_STS_IOERR;
907 }
908
909 blk_mq_start_request(rq);
910
911 INIT_WORK(&cmd->work, zloop_cmd_workfn);
912 queue_work(zlo->workqueue, &cmd->work);
913
914 return BLK_STS_OK;
915 }
916
917 static const struct blk_mq_ops zloop_mq_ops = {
918 .queue_rq = zloop_queue_rq,
919 .complete = zloop_complete_rq,
920 };
921
zloop_open(struct gendisk * disk,blk_mode_t mode)922 static int zloop_open(struct gendisk *disk, blk_mode_t mode)
923 {
924 struct zloop_device *zlo = disk->private_data;
925 int ret;
926
927 ret = mutex_lock_killable(&zloop_ctl_mutex);
928 if (ret)
929 return ret;
930
931 if (zlo->state != Zlo_live)
932 ret = -ENXIO;
933 mutex_unlock(&zloop_ctl_mutex);
934 return ret;
935 }
936
zloop_report_zones(struct gendisk * disk,sector_t sector,unsigned int nr_zones,struct blk_report_zones_args * args)937 static int zloop_report_zones(struct gendisk *disk, sector_t sector,
938 unsigned int nr_zones, struct blk_report_zones_args *args)
939 {
940 struct zloop_device *zlo = disk->private_data;
941 struct blk_zone blkz = {};
942 unsigned int first, i;
943 int ret;
944
945 first = disk_zone_no(disk, sector);
946 if (first >= zlo->nr_zones)
947 return 0;
948 nr_zones = min(nr_zones, zlo->nr_zones - first);
949
950 for (i = 0; i < nr_zones; i++) {
951 unsigned int zone_no = first + i;
952 struct zloop_zone *zone = &zlo->zones[zone_no];
953
954 mutex_lock(&zone->lock);
955
956 if (test_and_clear_bit(ZLOOP_ZONE_SEQ_ERROR, &zone->flags)) {
957 ret = zloop_update_seq_zone(zlo, zone_no);
958 if (ret) {
959 mutex_unlock(&zone->lock);
960 return ret;
961 }
962 }
963
964 blkz.start = zone->start;
965 blkz.len = zlo->zone_size;
966 spin_lock(&zone->wp_lock);
967 blkz.wp = zone->wp;
968 spin_unlock(&zone->wp_lock);
969 blkz.cond = zone->cond;
970 if (test_bit(ZLOOP_ZONE_CONV, &zone->flags)) {
971 blkz.type = BLK_ZONE_TYPE_CONVENTIONAL;
972 blkz.capacity = zlo->zone_size;
973 } else {
974 blkz.type = BLK_ZONE_TYPE_SEQWRITE_REQ;
975 blkz.capacity = zlo->zone_capacity;
976 }
977
978 mutex_unlock(&zone->lock);
979
980 ret = disk_report_zone(disk, &blkz, i, args);
981 if (ret)
982 return ret;
983 }
984
985 return nr_zones;
986 }
987
zloop_free_disk(struct gendisk * disk)988 static void zloop_free_disk(struct gendisk *disk)
989 {
990 struct zloop_device *zlo = disk->private_data;
991 unsigned int i;
992
993 blk_mq_free_tag_set(&zlo->tag_set);
994
995 for (i = 0; i < zlo->nr_zones; i++) {
996 struct zloop_zone *zone = &zlo->zones[i];
997
998 mapping_set_gfp_mask(zone->file->f_mapping,
999 zone->old_gfp_mask);
1000 fput(zone->file);
1001 }
1002
1003 fput(zlo->data_dir);
1004 destroy_workqueue(zlo->workqueue);
1005 kfree(zlo->base_dir);
1006 kvfree(zlo);
1007 }
1008
1009 static const struct block_device_operations zloop_fops = {
1010 .owner = THIS_MODULE,
1011 .open = zloop_open,
1012 .report_zones = zloop_report_zones,
1013 .free_disk = zloop_free_disk,
1014 };
1015
1016 __printf(3, 4)
zloop_filp_open_fmt(int oflags,umode_t mode,const char * fmt,...)1017 static struct file *zloop_filp_open_fmt(int oflags, umode_t mode,
1018 const char *fmt, ...)
1019 {
1020 struct file *file;
1021 va_list ap;
1022 char *p;
1023
1024 va_start(ap, fmt);
1025 p = kvasprintf(GFP_KERNEL, fmt, ap);
1026 va_end(ap);
1027
1028 if (!p)
1029 return ERR_PTR(-ENOMEM);
1030 file = filp_open(p, oflags, mode);
1031 kfree(p);
1032 return file;
1033 }
1034
zloop_get_block_size(struct zloop_device * zlo,struct zloop_zone * zone)1035 static int zloop_get_block_size(struct zloop_device *zlo,
1036 struct zloop_zone *zone)
1037 {
1038 struct block_device *sb_bdev = zone->file->f_mapping->host->i_sb->s_bdev;
1039 struct kstat st;
1040
1041 /*
1042 * Use the dio alignment of the file system if provided. The incoming
1043 * request's bio_vec is forwarded to the backing file unchanged, so its
1044 * required memory alignment becomes the device's dma_alignment when
1045 * used for direct-io.
1046 */
1047 if (!vfs_getattr(&zone->file->f_path, &st, STATX_DIOALIGN, 0) &&
1048 (st.result_mask & STATX_DIOALIGN)) {
1049 zlo->block_size = st.dio_offset_align;
1050 zlo->dio_mem_align = st.dio_mem_align - 1;
1051 } else if (sb_bdev) {
1052 zlo->block_size = bdev_physical_block_size(sb_bdev);
1053 zlo->dio_mem_align = bdev_dma_alignment(sb_bdev);
1054 } else {
1055 zlo->block_size = SECTOR_SIZE;
1056 zlo->dio_mem_align = SECTOR_SIZE - 1;
1057 }
1058
1059 /*
1060 * Prefer the FS block size for the device block size when it is no
1061 * larger than 4K; otherwise keep the direct I/O / physical block size
1062 * selected above.
1063 */
1064 if (file_inode(zone->file)->i_sb->s_blocksize <= SZ_4K)
1065 zlo->block_size = file_inode(zone->file)->i_sb->s_blocksize;
1066
1067 if (zlo->zone_capacity & ((zlo->block_size >> SECTOR_SHIFT) - 1)) {
1068 pr_err("Zone capacity is not aligned to block size %u\n",
1069 zlo->block_size);
1070 return -EINVAL;
1071 }
1072
1073 return 0;
1074 }
1075
zloop_init_zone(struct zloop_device * zlo,struct zloop_options * opts,unsigned int zone_no,bool restore)1076 static int zloop_init_zone(struct zloop_device *zlo, struct zloop_options *opts,
1077 unsigned int zone_no, bool restore)
1078 {
1079 struct zloop_zone *zone = &zlo->zones[zone_no];
1080 int oflags = O_RDWR;
1081 struct kstat stat;
1082 sector_t file_sectors;
1083 int ret;
1084
1085 mutex_init(&zone->lock);
1086 INIT_LIST_HEAD(&zone->open_zone_entry);
1087 spin_lock_init(&zone->wp_lock);
1088 zone->start = (sector_t)zone_no << zlo->zone_shift;
1089
1090 if (!restore)
1091 oflags |= O_CREAT;
1092
1093 if (!opts->buffered_io)
1094 oflags |= O_DIRECT;
1095
1096 if (zone_no < zlo->nr_conv_zones) {
1097 /* Conventional zone file. */
1098 set_bit(ZLOOP_ZONE_CONV, &zone->flags);
1099 zone->cond = BLK_ZONE_COND_NOT_WP;
1100 zone->wp = U64_MAX;
1101
1102 zone->file = zloop_filp_open_fmt(oflags, 0600, "%s/%u/cnv-%06u",
1103 zlo->base_dir, zlo->id, zone_no);
1104 if (IS_ERR(zone->file)) {
1105 pr_err("Failed to open zone %u file %s/%u/cnv-%06u (err=%ld)",
1106 zone_no, zlo->base_dir, zlo->id, zone_no,
1107 PTR_ERR(zone->file));
1108 return PTR_ERR(zone->file);
1109 }
1110
1111 if (!zlo->block_size) {
1112 ret = zloop_get_block_size(zlo, zone);
1113 if (ret)
1114 return ret;
1115 }
1116
1117 ret = vfs_getattr(&zone->file->f_path, &stat, STATX_SIZE, 0);
1118 if (ret < 0) {
1119 pr_err("Failed to get zone %u file stat\n", zone_no);
1120 return ret;
1121 }
1122 file_sectors = stat.size >> SECTOR_SHIFT;
1123
1124 if (restore && file_sectors != zlo->zone_size) {
1125 pr_err("Invalid conventional zone %u file size (%llu sectors != %llu)\n",
1126 zone_no, file_sectors, zlo->zone_capacity);
1127 return ret;
1128 }
1129
1130 ret = vfs_truncate(&zone->file->f_path,
1131 zlo->zone_size << SECTOR_SHIFT);
1132 if (ret < 0) {
1133 pr_err("Failed to truncate zone %u file (err=%d)\n",
1134 zone_no, ret);
1135 return ret;
1136 }
1137
1138 return 0;
1139 }
1140
1141 /* Sequential zone file. */
1142 zone->file = zloop_filp_open_fmt(oflags, 0600, "%s/%u/seq-%06u",
1143 zlo->base_dir, zlo->id, zone_no);
1144 if (IS_ERR(zone->file)) {
1145 pr_err("Failed to open zone %u file %s/%u/seq-%06u (err=%ld)",
1146 zone_no, zlo->base_dir, zlo->id, zone_no,
1147 PTR_ERR(zone->file));
1148 return PTR_ERR(zone->file);
1149 }
1150
1151 if (!zlo->block_size) {
1152 ret = zloop_get_block_size(zlo, zone);
1153 if (ret)
1154 return ret;
1155 }
1156
1157 zloop_get_block_size(zlo, zone);
1158
1159 mutex_lock(&zone->lock);
1160 ret = zloop_update_seq_zone(zlo, zone_no);
1161 mutex_unlock(&zone->lock);
1162
1163 return ret;
1164 }
1165
zloop_dev_exists(struct zloop_device * zlo)1166 static bool zloop_dev_exists(struct zloop_device *zlo)
1167 {
1168 struct file *cnv, *seq;
1169 bool exists;
1170
1171 cnv = zloop_filp_open_fmt(O_RDONLY, 0600, "%s/%u/cnv-%06u",
1172 zlo->base_dir, zlo->id, 0);
1173 seq = zloop_filp_open_fmt(O_RDONLY, 0600, "%s/%u/seq-%06u",
1174 zlo->base_dir, zlo->id, 0);
1175 exists = !IS_ERR(cnv) || !IS_ERR(seq);
1176
1177 if (!IS_ERR(cnv))
1178 fput(cnv);
1179 if (!IS_ERR(seq))
1180 fput(seq);
1181
1182 return exists;
1183 }
1184
zloop_ctl_add(struct zloop_options * opts)1185 static int zloop_ctl_add(struct zloop_options *opts)
1186 {
1187 struct queue_limits lim = {
1188 .max_hw_sectors = SZ_1M >> SECTOR_SHIFT,
1189 .chunk_sectors = opts->zone_size,
1190 .features = BLK_FEAT_ZONED | BLK_FEAT_WRITE_CACHE,
1191
1192 };
1193 unsigned int nr_zones, i, j;
1194 struct zloop_device *zlo;
1195 int ret = -EINVAL;
1196 bool restore;
1197
1198 __module_get(THIS_MODULE);
1199
1200 nr_zones = opts->capacity >> ilog2(opts->zone_size);
1201 if (opts->nr_conv_zones >= nr_zones) {
1202 pr_err("Invalid number of conventional zones %u\n",
1203 opts->nr_conv_zones);
1204 goto out;
1205 }
1206
1207 if (opts->max_open_zones > nr_zones - opts->nr_conv_zones) {
1208 pr_err("Invalid maximum number of open zones %u\n",
1209 opts->max_open_zones);
1210 goto out;
1211 }
1212
1213 zlo = kvzalloc_flex(*zlo, zones, nr_zones);
1214 if (!zlo) {
1215 ret = -ENOMEM;
1216 goto out;
1217 }
1218 WRITE_ONCE(zlo->state, Zlo_creating);
1219 spin_lock_init(&zlo->open_zones_lock);
1220 INIT_LIST_HEAD(&zlo->open_zones_lru_list);
1221
1222 ret = mutex_lock_killable(&zloop_ctl_mutex);
1223 if (ret)
1224 goto out_free_dev;
1225
1226 /* Allocate id, if @opts->id >= 0, we're requesting that specific id */
1227 if (opts->id >= 0) {
1228 ret = idr_alloc(&zloop_index_idr, zlo,
1229 opts->id, opts->id + 1, GFP_KERNEL);
1230 if (ret == -ENOSPC)
1231 ret = -EEXIST;
1232 } else {
1233 ret = idr_alloc(&zloop_index_idr, zlo, 0, 0, GFP_KERNEL);
1234 }
1235 mutex_unlock(&zloop_ctl_mutex);
1236 if (ret < 0)
1237 goto out_free_dev;
1238
1239 zlo->id = ret;
1240 zlo->zone_shift = ilog2(opts->zone_size);
1241 zlo->zone_size = opts->zone_size;
1242 if (opts->zone_capacity)
1243 zlo->zone_capacity = opts->zone_capacity;
1244 else
1245 zlo->zone_capacity = zlo->zone_size;
1246 zlo->nr_zones = nr_zones;
1247 zlo->nr_conv_zones = opts->nr_conv_zones;
1248 zlo->max_open_zones = opts->max_open_zones;
1249 zlo->buffered_io = opts->buffered_io;
1250 zlo->zone_append = opts->zone_append;
1251 if (zlo->zone_append)
1252 zlo->ordered_zone_append = opts->ordered_zone_append;
1253 zlo->discard_write_cache = opts->discard_write_cache;
1254
1255 zlo->workqueue = alloc_workqueue("zloop%d", WQ_UNBOUND | WQ_FREEZABLE,
1256 opts->nr_queues * opts->queue_depth, zlo->id);
1257 if (!zlo->workqueue) {
1258 ret = -ENOMEM;
1259 goto out_free_idr;
1260 }
1261
1262 if (opts->base_dir)
1263 zlo->base_dir = kstrdup(opts->base_dir, GFP_KERNEL);
1264 else
1265 zlo->base_dir = kstrdup(ZLOOP_DEF_BASE_DIR, GFP_KERNEL);
1266 if (!zlo->base_dir) {
1267 ret = -ENOMEM;
1268 goto out_destroy_workqueue;
1269 }
1270
1271 zlo->data_dir = zloop_filp_open_fmt(O_RDONLY | O_DIRECTORY, 0, "%s/%u",
1272 zlo->base_dir, zlo->id);
1273 if (IS_ERR(zlo->data_dir)) {
1274 ret = PTR_ERR(zlo->data_dir);
1275 pr_warn("Failed to open directory %s/%u (err=%d)\n",
1276 zlo->base_dir, zlo->id, ret);
1277 goto out_free_base_dir;
1278 }
1279
1280 /*
1281 * If we already have zone files, we are restoring a device created by a
1282 * previous add operation. In this case, zloop_init_zone() will check
1283 * that the zone files are consistent with the zone configuration given.
1284 */
1285 restore = zloop_dev_exists(zlo);
1286 for (i = 0; i < nr_zones; i++) {
1287 ret = zloop_init_zone(zlo, opts, i, restore);
1288 if (ret)
1289 goto out_close_files;
1290 }
1291
1292 lim.physical_block_size = zlo->block_size;
1293 lim.logical_block_size = zlo->block_size;
1294 /* Direct I/O forwards the request pages to the backing files as-is. */
1295 if (!opts->buffered_io)
1296 lim.dma_alignment = max_t(unsigned int, zlo->dio_mem_align,
1297 SECTOR_SIZE - 1);
1298 if (zlo->zone_append)
1299 lim.max_hw_zone_append_sectors = lim.max_hw_sectors;
1300 lim.max_open_zones = zlo->max_open_zones;
1301
1302 zlo->tag_set.ops = &zloop_mq_ops;
1303 zlo->tag_set.nr_hw_queues = opts->nr_queues;
1304 zlo->tag_set.queue_depth = opts->queue_depth;
1305 zlo->tag_set.numa_node = NUMA_NO_NODE;
1306 zlo->tag_set.cmd_size = sizeof(struct zloop_cmd);
1307 zlo->tag_set.driver_data = zlo;
1308
1309 ret = blk_mq_alloc_tag_set(&zlo->tag_set);
1310 if (ret) {
1311 pr_err("blk_mq_alloc_tag_set failed (err=%d)\n", ret);
1312 goto out_close_files;
1313 }
1314
1315 zlo->disk = blk_mq_alloc_disk(&zlo->tag_set, &lim, zlo);
1316 if (IS_ERR(zlo->disk)) {
1317 pr_err("blk_mq_alloc_disk failed (err=%d)\n", ret);
1318 ret = PTR_ERR(zlo->disk);
1319 goto out_cleanup_tags;
1320 }
1321 zlo->disk->flags = GENHD_FL_NO_PART;
1322 zlo->disk->fops = &zloop_fops;
1323 zlo->disk->private_data = zlo;
1324 sprintf(zlo->disk->disk_name, "zloop%d", zlo->id);
1325 set_capacity(zlo->disk, (u64)lim.chunk_sectors * zlo->nr_zones);
1326
1327 ret = blk_revalidate_disk_zones(zlo->disk);
1328 if (ret)
1329 goto out_cleanup_disk;
1330
1331 ret = add_disk(zlo->disk);
1332 if (ret) {
1333 pr_err("add_disk failed (err=%d)\n", ret);
1334 goto out_cleanup_disk;
1335 }
1336
1337 mutex_lock(&zloop_ctl_mutex);
1338 WRITE_ONCE(zlo->state, Zlo_live);
1339 mutex_unlock(&zloop_ctl_mutex);
1340
1341 pr_info("zloop: device %d, %u zones of %llu MiB, %u B block size\n",
1342 zlo->id, zlo->nr_zones,
1343 ((sector_t)zlo->zone_size << SECTOR_SHIFT) >> 20,
1344 zlo->block_size);
1345 pr_info("zloop%d: using %s%s zone append\n",
1346 zlo->id,
1347 zlo->ordered_zone_append ? "ordered " : "",
1348 zlo->zone_append ? "native" : "emulated");
1349
1350 return 0;
1351
1352 out_cleanup_disk:
1353 put_disk(zlo->disk);
1354 out_cleanup_tags:
1355 blk_mq_free_tag_set(&zlo->tag_set);
1356 out_close_files:
1357 for (j = 0; j < i; j++) {
1358 struct zloop_zone *zone = &zlo->zones[j];
1359
1360 if (!IS_ERR_OR_NULL(zone->file))
1361 fput(zone->file);
1362 }
1363 fput(zlo->data_dir);
1364 out_free_base_dir:
1365 kfree(zlo->base_dir);
1366 out_destroy_workqueue:
1367 destroy_workqueue(zlo->workqueue);
1368 out_free_idr:
1369 mutex_lock(&zloop_ctl_mutex);
1370 idr_remove(&zloop_index_idr, zlo->id);
1371 mutex_unlock(&zloop_ctl_mutex);
1372 out_free_dev:
1373 kvfree(zlo);
1374 out:
1375 module_put(THIS_MODULE);
1376 if (ret == -ENOENT)
1377 ret = -EINVAL;
1378 return ret;
1379 }
1380
zloop_forget_cache(struct zloop_device * zlo)1381 static void zloop_forget_cache(struct zloop_device *zlo)
1382 {
1383 unsigned int i;
1384 int ret;
1385
1386 pr_info("%pg: discarding volatile write cache\n", zlo->disk->part0);
1387
1388 for (i = 0; i < zlo->nr_zones; i++) {
1389 struct zloop_zone *zone = &zlo->zones[i];
1390 struct file *file = zone->file;
1391 sector_t old_wp;
1392
1393 if (!zloop_zone_is_active(zone))
1394 continue;
1395
1396 ret = vfs_getxattr(file_mnt_idmap(file), file_dentry(file),
1397 "user.zloop.wp", &old_wp, sizeof(old_wp));
1398 if (ret == -ENODATA) {
1399 old_wp = 0;
1400 } else if (ret != sizeof(old_wp)) {
1401 pr_err("%pg: failed to retrieve write pointer (%d)\n",
1402 zlo->disk->part0, ret);
1403 continue;
1404 }
1405
1406 if (old_wp > zone->wp)
1407 continue;
1408 /*
1409 * This should not happen, if we recored a full zone, it can't
1410 * be active.
1411 */
1412 if (WARN_ON_ONCE(old_wp == ULLONG_MAX))
1413 continue;
1414
1415 vfs_truncate(&file->f_path,
1416 (old_wp - zone->start) << SECTOR_SHIFT);
1417 }
1418 }
1419
zloop_ctl_remove(struct zloop_options * opts)1420 static int zloop_ctl_remove(struct zloop_options *opts)
1421 {
1422 struct zloop_device *zlo;
1423 int ret;
1424
1425 if (!(opts->mask & ZLOOP_OPT_ID)) {
1426 pr_err("No ID specified for remove\n");
1427 return -EINVAL;
1428 }
1429
1430 if (opts->mask & ~ZLOOP_OPT_ID) {
1431 pr_err("Invalid option specified for remove\n");
1432 return -EINVAL;
1433 }
1434
1435 ret = mutex_lock_killable(&zloop_ctl_mutex);
1436 if (ret)
1437 return ret;
1438
1439 zlo = idr_find(&zloop_index_idr, opts->id);
1440 if (!zlo || zlo->state == Zlo_creating) {
1441 ret = -ENODEV;
1442 } else if (zlo->state == Zlo_deleting) {
1443 ret = -EINVAL;
1444 } else {
1445 idr_remove(&zloop_index_idr, zlo->id);
1446 WRITE_ONCE(zlo->state, Zlo_deleting);
1447 }
1448
1449 mutex_unlock(&zloop_ctl_mutex);
1450 if (ret)
1451 return ret;
1452
1453 del_gendisk(zlo->disk);
1454
1455 if (zlo->discard_write_cache)
1456 zloop_forget_cache(zlo);
1457
1458 put_disk(zlo->disk);
1459
1460 pr_info("Removed device %d\n", opts->id);
1461
1462 module_put(THIS_MODULE);
1463
1464 return 0;
1465 }
1466
zloop_parse_options(struct zloop_options * opts,const char * buf)1467 static int zloop_parse_options(struct zloop_options *opts, const char *buf)
1468 {
1469 substring_t args[MAX_OPT_ARGS];
1470 char *options, *o, *p;
1471 unsigned int token;
1472 int ret = 0;
1473
1474 /* Set defaults. */
1475 opts->mask = 0;
1476 opts->id = ZLOOP_DEF_ID;
1477 opts->capacity = ZLOOP_DEF_ZONE_SIZE * ZLOOP_DEF_NR_ZONES;
1478 opts->zone_size = ZLOOP_DEF_ZONE_SIZE;
1479 opts->nr_conv_zones = ZLOOP_DEF_NR_CONV_ZONES;
1480 opts->max_open_zones = ZLOOP_DEF_MAX_OPEN_ZONES;
1481 opts->nr_queues = ZLOOP_DEF_NR_QUEUES;
1482 opts->queue_depth = ZLOOP_DEF_QUEUE_DEPTH;
1483 opts->buffered_io = ZLOOP_DEF_BUFFERED_IO;
1484 opts->zone_append = ZLOOP_DEF_ZONE_APPEND;
1485 opts->ordered_zone_append = ZLOOP_DEF_ORDERED_ZONE_APPEND;
1486
1487 if (!buf)
1488 return 0;
1489
1490 /* Skip leading spaces before the options. */
1491 while (isspace(*buf))
1492 buf++;
1493
1494 options = o = kstrdup(buf, GFP_KERNEL);
1495 if (!options)
1496 return -ENOMEM;
1497
1498 /* Parse the options, doing only some light invalid value checks. */
1499 while ((p = strsep(&o, ",\n")) != NULL) {
1500 if (!*p)
1501 continue;
1502
1503 token = match_token(p, zloop_opt_tokens, args);
1504 opts->mask |= token;
1505 switch (token) {
1506 case ZLOOP_OPT_ID:
1507 if (match_int(args, &opts->id)) {
1508 ret = -EINVAL;
1509 goto out;
1510 }
1511 break;
1512 case ZLOOP_OPT_CAPACITY:
1513 if (match_uint(args, &token)) {
1514 ret = -EINVAL;
1515 goto out;
1516 }
1517 if (!token) {
1518 pr_err("Invalid capacity\n");
1519 ret = -EINVAL;
1520 goto out;
1521 }
1522 opts->capacity =
1523 ((sector_t)token * SZ_1M) >> SECTOR_SHIFT;
1524 break;
1525 case ZLOOP_OPT_ZONE_SIZE:
1526 if (match_uint(args, &token)) {
1527 ret = -EINVAL;
1528 goto out;
1529 }
1530 if (!token || token > ZLOOP_MAX_ZONE_SIZE_MB ||
1531 !is_power_of_2(token)) {
1532 pr_err("Invalid zone size %u\n", token);
1533 ret = -EINVAL;
1534 goto out;
1535 }
1536 opts->zone_size =
1537 ((sector_t)token * SZ_1M) >> SECTOR_SHIFT;
1538 break;
1539 case ZLOOP_OPT_ZONE_CAPACITY:
1540 if (match_uint(args, &token)) {
1541 ret = -EINVAL;
1542 goto out;
1543 }
1544 if (!token) {
1545 pr_err("Invalid zone capacity\n");
1546 ret = -EINVAL;
1547 goto out;
1548 }
1549 opts->zone_capacity =
1550 ((sector_t)token * SZ_1M) >> SECTOR_SHIFT;
1551 break;
1552 case ZLOOP_OPT_NR_CONV_ZONES:
1553 if (match_uint(args, &token)) {
1554 ret = -EINVAL;
1555 goto out;
1556 }
1557 opts->nr_conv_zones = token;
1558 break;
1559 case ZLOOP_OPT_MAX_OPEN_ZONES:
1560 if (match_uint(args, &token)) {
1561 ret = -EINVAL;
1562 goto out;
1563 }
1564 opts->max_open_zones = token;
1565 break;
1566 case ZLOOP_OPT_BASE_DIR:
1567 p = match_strdup(args);
1568 if (!p) {
1569 ret = -ENOMEM;
1570 goto out;
1571 }
1572 kfree(opts->base_dir);
1573 opts->base_dir = p;
1574 break;
1575 case ZLOOP_OPT_NR_QUEUES:
1576 if (match_uint(args, &token)) {
1577 ret = -EINVAL;
1578 goto out;
1579 }
1580 if (!token) {
1581 pr_err("Invalid number of queues\n");
1582 ret = -EINVAL;
1583 goto out;
1584 }
1585 opts->nr_queues = min(token, num_online_cpus());
1586 break;
1587 case ZLOOP_OPT_QUEUE_DEPTH:
1588 if (match_uint(args, &token)) {
1589 ret = -EINVAL;
1590 goto out;
1591 }
1592 if (!token) {
1593 pr_err("Invalid queue depth\n");
1594 ret = -EINVAL;
1595 goto out;
1596 }
1597 opts->queue_depth = token;
1598 break;
1599 case ZLOOP_OPT_BUFFERED_IO:
1600 opts->buffered_io = true;
1601 break;
1602 case ZLOOP_OPT_ZONE_APPEND:
1603 if (match_uint(args, &token)) {
1604 ret = -EINVAL;
1605 goto out;
1606 }
1607 if (token != 0 && token != 1) {
1608 pr_err("Invalid zone_append value\n");
1609 ret = -EINVAL;
1610 goto out;
1611 }
1612 opts->zone_append = token;
1613 break;
1614 case ZLOOP_OPT_ORDERED_ZONE_APPEND:
1615 opts->ordered_zone_append = true;
1616 break;
1617 case ZLOOP_OPT_DISCARD_WRITE_CACHE:
1618 opts->discard_write_cache = true;
1619 break;
1620 case ZLOOP_OPT_ERR:
1621 default:
1622 pr_warn("unknown parameter or missing value '%s'\n", p);
1623 ret = -EINVAL;
1624 goto out;
1625 }
1626 }
1627
1628 ret = -EINVAL;
1629 if (opts->capacity <= opts->zone_size) {
1630 pr_err("Invalid capacity\n");
1631 goto out;
1632 }
1633
1634 if (opts->zone_capacity > opts->zone_size) {
1635 pr_err("Invalid zone capacity\n");
1636 goto out;
1637 }
1638
1639 ret = 0;
1640 out:
1641 kfree(options);
1642 return ret;
1643 }
1644
1645 enum {
1646 ZLOOP_CTL_ADD,
1647 ZLOOP_CTL_REMOVE,
1648 };
1649
1650 static struct zloop_ctl_op {
1651 int code;
1652 const char *name;
1653 } zloop_ctl_ops[] = {
1654 { ZLOOP_CTL_ADD, "add" },
1655 { ZLOOP_CTL_REMOVE, "remove" },
1656 { -1, NULL },
1657 };
1658
zloop_ctl_write(struct file * file,const char __user * ubuf,size_t count,loff_t * pos)1659 static ssize_t zloop_ctl_write(struct file *file, const char __user *ubuf,
1660 size_t count, loff_t *pos)
1661 {
1662 struct zloop_options opts = { };
1663 struct zloop_ctl_op *op;
1664 const char *buf, *opts_buf;
1665 int i, ret;
1666
1667 if (count > PAGE_SIZE)
1668 return -ENOMEM;
1669
1670 buf = memdup_user_nul(ubuf, count);
1671 if (IS_ERR(buf))
1672 return PTR_ERR(buf);
1673
1674 for (i = 0; i < ARRAY_SIZE(zloop_ctl_ops); i++) {
1675 op = &zloop_ctl_ops[i];
1676 if (!op->name) {
1677 pr_err("Invalid operation\n");
1678 ret = -EINVAL;
1679 goto out;
1680 }
1681 if (!strncmp(buf, op->name, strlen(op->name)))
1682 break;
1683 }
1684
1685 if (count <= strlen(op->name))
1686 opts_buf = NULL;
1687 else
1688 opts_buf = buf + strlen(op->name);
1689
1690 ret = zloop_parse_options(&opts, opts_buf);
1691 if (ret) {
1692 pr_err("Failed to parse options\n");
1693 goto out;
1694 }
1695
1696 switch (op->code) {
1697 case ZLOOP_CTL_ADD:
1698 ret = zloop_ctl_add(&opts);
1699 break;
1700 case ZLOOP_CTL_REMOVE:
1701 ret = zloop_ctl_remove(&opts);
1702 break;
1703 default:
1704 pr_err("Invalid operation\n");
1705 ret = -EINVAL;
1706 goto out;
1707 }
1708
1709 out:
1710 kfree(opts.base_dir);
1711 kfree(buf);
1712 return ret ? ret : count;
1713 }
1714
zloop_ctl_show(struct seq_file * seq_file,void * private)1715 static int zloop_ctl_show(struct seq_file *seq_file, void *private)
1716 {
1717 const struct match_token *tok;
1718 int i;
1719
1720 /* Add operation */
1721 seq_printf(seq_file, "%s ", zloop_ctl_ops[0].name);
1722 for (i = 0; i < ARRAY_SIZE(zloop_opt_tokens); i++) {
1723 tok = &zloop_opt_tokens[i];
1724 if (!tok->pattern)
1725 break;
1726 if (i)
1727 seq_putc(seq_file, ',');
1728 seq_puts(seq_file, tok->pattern);
1729 }
1730 seq_putc(seq_file, '\n');
1731
1732 /* Remove operation */
1733 seq_puts(seq_file, zloop_ctl_ops[1].name);
1734 seq_puts(seq_file, " id=%d\n");
1735
1736 return 0;
1737 }
1738
zloop_ctl_open(struct inode * inode,struct file * file)1739 static int zloop_ctl_open(struct inode *inode, struct file *file)
1740 {
1741 file->private_data = NULL;
1742 return single_open(file, zloop_ctl_show, NULL);
1743 }
1744
zloop_ctl_release(struct inode * inode,struct file * file)1745 static int zloop_ctl_release(struct inode *inode, struct file *file)
1746 {
1747 return single_release(inode, file);
1748 }
1749
1750 static const struct file_operations zloop_ctl_fops = {
1751 .owner = THIS_MODULE,
1752 .open = zloop_ctl_open,
1753 .release = zloop_ctl_release,
1754 .write = zloop_ctl_write,
1755 .read = seq_read,
1756 };
1757
1758 static struct miscdevice zloop_misc = {
1759 .minor = MISC_DYNAMIC_MINOR,
1760 .name = "zloop-control",
1761 .fops = &zloop_ctl_fops,
1762 };
1763
zloop_init(void)1764 static int __init zloop_init(void)
1765 {
1766 int ret;
1767
1768 ret = misc_register(&zloop_misc);
1769 if (ret) {
1770 pr_err("Failed to register misc device: %d\n", ret);
1771 return ret;
1772 }
1773 pr_info("Module loaded\n");
1774
1775 return 0;
1776 }
1777
zloop_exit(void)1778 static void __exit zloop_exit(void)
1779 {
1780 misc_deregister(&zloop_misc);
1781 idr_destroy(&zloop_index_idr);
1782 }
1783
1784 module_init(zloop_init);
1785 module_exit(zloop_exit);
1786
1787 MODULE_DESCRIPTION("Zoned loopback device");
1788 MODULE_LICENSE("GPL");
1789