xref: /linux/tools/testing/selftests/mm/khugepaged.c (revision 67f8bc848ee31831336bd478e57d2f993551902e)
1 #define _GNU_SOURCE
2 #include <ctype.h>
3 #include <errno.h>
4 #include <fcntl.h>
5 #include <limits.h>
6 #include <dirent.h>
7 #include <signal.h>
8 #include <stdio.h>
9 #include <stdlib.h>
10 #include <stdbool.h>
11 #include <string.h>
12 #include <unistd.h>
13 
14 #include <linux/mman.h>
15 #include <sys/mman.h>
16 #include <sys/wait.h>
17 #include <sys/types.h>
18 #include <sys/stat.h>
19 #include <sys/sysmacros.h>
20 #include <sys/vfs.h>
21 
22 #include "linux/magic.h"
23 
24 #include "vm_util.h"
25 #include "hugepage_settings.h"
26 
27 #define BASE_ADDR ((void *)(1UL << 30))
28 static unsigned long hpage_pmd_size;
29 static int hpage_pmd_order;
30 static unsigned long page_size;
31 static int hpage_pmd_nr;
32 static int anon_order;
33 static int collapse_order;
34 
35 #define PID_SMAPS "/proc/self/smaps"
36 #define TEST_FILE "collapse_test_file"
37 
38 #define MAX_LINE_LENGTH 500
39 
40 enum vma_type {
41 	VMA_ANON,
42 	VMA_FILE,
43 	VMA_SHMEM,
44 };
45 
46 enum file_setup_ops {
47 	FILE_SETUP_READ_ONLY_FS,
48 	FILE_SETUP_READ_WRITE_FS_READ_DATA,
49 	FILE_SETUP_READ_WRITE_FS_WRITE_DATA,
50 };
51 
52 struct mem_ops {
53 	void *(*setup_area)(int nr_hpages);
54 	void (*cleanup_area)(void *p, unsigned long size);
55 	void (*fault)(void *p, unsigned long start, unsigned long end);
56 	bool (*check_huge)(void *addr, size_t len, int nr_hpages, unsigned long hpage_size);
57 	const char *name;
58 };
59 
60 static struct mem_ops *read_only_file_ops;
61 static struct mem_ops *read_write_file_read_ops;
62 static struct mem_ops *read_write_file_write_ops;
63 static struct mem_ops *anon_ops;
64 static struct mem_ops *shmem_ops;
65 
66 struct collapse_context {
67 	void (*collapse)(const char *msg, char *p, int nr_hpages,
68 			 struct mem_ops *ops, bool expect);
69 	bool enforce_pte_scan_limits;
70 	const char *name;
71 };
72 
73 static struct collapse_context *khugepaged_context;
74 static struct collapse_context *mthp_khugepaged_context;
75 static struct collapse_context *madvise_context;
76 
77 struct file_info {
78 	const char *dir;
79 	char path[PATH_MAX];
80 	enum vma_type type;
81 	int fd;
82 	char dev_queue_read_ahead_path[PATH_MAX];
83 };
84 
85 static struct file_info finfo;
86 static int exit_status;
87 
88 static void success(const char *msg)
89 {
90 	printf(" \e[32m%s\e[0m\n", msg);
91 	exit_status = KSFT_PASS;
92 }
93 
94 static void fail(const char *msg)
95 {
96 	printf(" \e[31m%s\e[0m\n", msg);
97 	exit_status = KSFT_FAIL;
98 }
99 
100 static void skip(const char *msg)
101 {
102 	printf(" \e[33m%s\e[0m\n", msg);
103 	exit_status = KSFT_SKIP;
104 }
105 
106 static void save_settings(void)
107 {
108 	ksft_print_msg("Save THP and khugepaged settings...");
109 	if ((read_only_file_ops || read_write_file_read_ops ||
110 	     read_write_file_write_ops) &&
111 	    finfo.type == VMA_FILE)
112 		thp_set_read_ahead_path(finfo.dev_queue_read_ahead_path);
113 	thp_save_settings();
114 
115 	success("OK");
116 }
117 
118 static void get_finfo(const char *dir)
119 {
120 	struct stat path_stat;
121 	struct statfs fs;
122 	char buf[1 << 10];
123 	char path[PATH_MAX];
124 	char *str, *end;
125 
126 	finfo.dir = dir;
127 	if (stat(finfo.dir, &path_stat))
128 		ksft_exit_fail_perror("stat()");
129 	if (!S_ISDIR(path_stat.st_mode))
130 		ksft_exit_fail_msg("%s: Not a directory (%s)\n", __func__, finfo.dir);
131 	if (snprintf(finfo.path, sizeof(finfo.path), "%s/" TEST_FILE,
132 		     finfo.dir) >= sizeof(finfo.path))
133 		ksft_exit_fail_msg("%s: Pathname is too long\n", __func__);
134 	if (statfs(finfo.dir, &fs))
135 		ksft_exit_fail_perror("statfs()");
136 	finfo.type = fs.f_type == TMPFS_MAGIC ? VMA_SHMEM : VMA_FILE;
137 	if (finfo.type == VMA_SHMEM)
138 		return;
139 
140 	/* Find owning device's queue/read_ahead_kb control */
141 	if (snprintf(path, sizeof(path), "/sys/dev/block/%d:%d/uevent",
142 		     major(path_stat.st_dev), minor(path_stat.st_dev))
143 	    >= sizeof(path))
144 		ksft_exit_fail_msg("%s: Pathname is too long\n", __func__);
145 	if (!read_file(path, buf, sizeof(buf)))
146 		ksft_exit_fail_perror("read_file(uevent)");
147 	if (strstr(buf, "DEVTYPE=disk")) {
148 		/* Found it */
149 		if (snprintf(finfo.dev_queue_read_ahead_path,
150 			     sizeof(finfo.dev_queue_read_ahead_path),
151 			     "/sys/dev/block/%d:%d/queue/read_ahead_kb",
152 			     major(path_stat.st_dev), minor(path_stat.st_dev))
153 		    >= sizeof(finfo.dev_queue_read_ahead_path))
154 			ksft_exit_fail_msg("%s: Pathname is too long\n", __func__);
155 		return;
156 	}
157 	if (!strstr(buf, "DEVTYPE=partition"))
158 		ksft_exit_fail_msg("%s: Unknown device type: %s\n", __func__, path);
159 	/*
160 	 * Partition of block device - need to find actual device.
161 	 * Using naming convention that devnameN is partition of
162 	 * device devname.
163 	 */
164 	str = strstr(buf, "DEVNAME=");
165 	if (!str)
166 		ksft_exit_fail_msg("%s: Could not read: %s", __func__, path);
167 	str += 8;
168 	end = str;
169 	while (*end) {
170 		if (isdigit(*end)) {
171 			*end = '\0';
172 			if (snprintf(finfo.dev_queue_read_ahead_path,
173 				     sizeof(finfo.dev_queue_read_ahead_path),
174 				     "/sys/block/%s/queue/read_ahead_kb",
175 				     str) >= sizeof(finfo.dev_queue_read_ahead_path))
176 				ksft_exit_fail_msg("%s: Pathname is too long\n", __func__);
177 			return;
178 		}
179 		++end;
180 	}
181 	ksft_exit_fail_msg("%s: Could not read: %s\n", __func__, path);
182 }
183 
184 static bool check_swap(void *addr, unsigned long size)
185 {
186 	bool swap = false;
187 	int ret;
188 	FILE *fp;
189 	char buffer[MAX_LINE_LENGTH];
190 	char addr_pattern[MAX_LINE_LENGTH];
191 
192 	ret = snprintf(addr_pattern, MAX_LINE_LENGTH, "%08lx-",
193 		       (unsigned long) addr);
194 	if (ret >= MAX_LINE_LENGTH)
195 		ksft_exit_fail_msg("%s: Pattern is too long\n", __func__);
196 
197 	fp = fopen(PID_SMAPS, "r");
198 	if (!fp)
199 		ksft_exit_fail_msg("%s: Failed to open file %s\n", __func__, PID_SMAPS);
200 	if (!check_for_pattern(fp, addr_pattern, buffer, sizeof(buffer)))
201 		goto err_out;
202 
203 	ret = snprintf(addr_pattern, MAX_LINE_LENGTH, "Swap:%19ld kB",
204 		       size >> 10);
205 	if (ret >= MAX_LINE_LENGTH)
206 		ksft_exit_fail_msg("%s: Pattern is too long\n", __func__);
207 	/*
208 	 * Fetch the Swap: in the same block and check whether it got
209 	 * the expected number of hugeepages next.
210 	 */
211 	if (!check_for_pattern(fp, "Swap:", buffer, sizeof(buffer)))
212 		goto err_out;
213 
214 	if (strncmp(buffer, addr_pattern, strlen(addr_pattern)))
215 		goto err_out;
216 
217 	swap = true;
218 err_out:
219 	fclose(fp);
220 	return swap;
221 }
222 
223 static void *alloc_mapping(int nr)
224 {
225 	void *p;
226 
227 	p = mmap(BASE_ADDR, nr * hpage_pmd_size, PROT_READ | PROT_WRITE,
228 		 MAP_ANONYMOUS | MAP_PRIVATE, -1, 0);
229 	if (p != BASE_ADDR)
230 		ksft_exit_fail_msg("Failed to allocate VMA at %p\n", BASE_ADDR);
231 
232 	return p;
233 }
234 
235 static void fill_memory(int *p, unsigned long start, unsigned long end)
236 {
237 	int i;
238 
239 	for (i = start / page_size; i < end / page_size; i++)
240 		p[i * page_size / sizeof(*p)] = i + 0xdead0000;
241 }
242 
243 /*
244  * MADV_COLLAPSE is a best-effort request and may fail if an internal
245  * resource is temporarily unavailable, in which case it will set errno to
246  * EAGAIN.  In such a case, immediately reattempt the operation one more
247  * time.
248  */
249 static int madvise_collapse_retry(void *p, unsigned long size)
250 {
251 	bool retry = true;
252 	int ret;
253 
254 retry:
255 	ret = madvise(p, size, MADV_COLLAPSE);
256 	if (ret && errno == EAGAIN && retry) {
257 		retry = false;
258 		goto retry;
259 	}
260 	return ret;
261 }
262 
263 /*
264  * Returns pmd-mapped hugepage in VMA marked VM_HUGEPAGE, filled with
265  * validate_memory()'able contents.
266  */
267 static void *alloc_hpage(struct mem_ops *ops)
268 {
269 	void *p = ops->setup_area(1);
270 
271 	ops->fault(p, 0, hpage_pmd_size);
272 
273 	/*
274 	 * VMA should be neither VM_HUGEPAGE nor VM_NOHUGEPAGE.
275 	 * The latter is ineligible for collapse by MADV_COLLAPSE
276 	 * while the former might cause MADV_COLLAPSE to race with
277 	 * khugepaged on low-load system (like a test machine), which
278 	 * would cause MADV_COLLAPSE to fail with EAGAIN.
279 	 */
280 	ksft_print_msg("Allocate huge page...");
281 	if (madvise_collapse_retry(p, hpage_pmd_size))
282 		ksft_exit_fail_perror("madvise(MADV_COLLAPSE)");
283 	if (!ops->check_huge(p, hpage_pmd_size, 1, hpage_pmd_size))
284 		ksft_exit_fail_perror("madvise(MADV_COLLAPSE)");
285 	if (madvise(p, hpage_pmd_size, MADV_HUGEPAGE))
286 		ksft_exit_fail_perror("madvise(MADV_HUGEPAGE)");
287 	success("OK");
288 	return p;
289 }
290 
291 static void validate_memory(int *p, unsigned long start, unsigned long end)
292 {
293 	int i;
294 
295 	for (i = start / page_size; i < end / page_size; i++) {
296 		if (p[i * page_size / sizeof(*p)] != i + 0xdead0000)
297 			ksft_exit_fail_msg("Page %d is corrupted: %#x\n",
298 					   i, p[i * page_size / sizeof(*p)]);
299 	}
300 }
301 
302 static void *anon_setup_area(int nr_hpages)
303 {
304 	return alloc_mapping(nr_hpages);
305 }
306 
307 static void anon_cleanup_area(void *p, unsigned long size)
308 {
309 	munmap(p, size);
310 }
311 
312 static void anon_fault(void *p, unsigned long start, unsigned long end)
313 {
314 	fill_memory(p, start, end);
315 }
316 
317 static bool anon_check_huge(void *addr, size_t len, int nr_hpages,
318 		unsigned long hpage_size)
319 {
320 	return check_huge_anon(addr, len, nr_hpages, hpage_size);
321 }
322 
323 static void *file_setup_area_common(int nr_hpages, enum file_setup_ops setup)
324 {
325 	const int open_opt = setup == FILE_SETUP_READ_ONLY_FS ? O_RDONLY : O_RDWR;
326 	const int mmap_prot = setup == FILE_SETUP_READ_ONLY_FS ? PROT_READ : (PROT_READ | PROT_WRITE);
327 	int fd;
328 	void *p;
329 	unsigned long size;
330 
331 	unlink(finfo.path);  /* Cleanup from previous failed tests */
332 	ksft_print_msg("Creating %s for collapse%s...", finfo.path,
333 		       finfo.type == VMA_SHMEM ? " (tmpfs)" : "");
334 	fd = open(finfo.path, O_CREAT | O_RDWR | O_TRUNC | O_EXCL,
335 		  777);
336 	if (fd < 0)
337 		ksft_exit_fail_perror("open()");
338 
339 	size = nr_hpages * hpage_pmd_size;
340 	if (ftruncate(fd, size)) {
341 		perror("ftruncate()");
342 		exit(EXIT_FAILURE);
343 	}
344 	p = mmap(BASE_ADDR, size, PROT_READ | PROT_WRITE,
345 		MAP_SHARED, fd, 0);
346 	if (p != BASE_ADDR) {
347 		perror("mmap()");
348 		exit(EXIT_FAILURE);
349 	}
350 	fill_memory(p, 0, size);
351 	if (msync(p, size, MS_SYNC)) {
352 		perror("msync()");
353 		exit(EXIT_FAILURE);
354 	}
355 	close(fd);
356 	munmap(p, size);
357 	success("OK");
358 	ksft_print_msg("Opening %s %s for collapse...", finfo.path,
359 	       setup == FILE_SETUP_READ_ONLY_FS ? "read-only" :
360 	       setup == FILE_SETUP_READ_WRITE_FS_READ_DATA ?
361 						  "read-write (read)" :
362 						  "read-write (write)");
363 	finfo.fd = open(finfo.path, open_opt, 777);
364 	if (finfo.fd < 0)
365 		ksft_exit_fail_perror("open()");
366 	p = mmap(BASE_ADDR, size, mmap_prot, MAP_SHARED, finfo.fd, 0);
367 	if (p == MAP_FAILED || p != BASE_ADDR)
368 		ksft_exit_fail_perror("mmap()");
369 
370 	/* Drop page cache */
371 	write_file("/proc/sys/vm/drop_caches", "3", 2);
372 	success("OK");
373 	return p;
374 }
375 
376 static void *file_setup_read_only_area(int nr_hpages)
377 {
378 	return file_setup_area_common(nr_hpages, FILE_SETUP_READ_ONLY_FS);
379 }
380 
381 static void *file_setup_read_write_fs_read_area(int nr_hpages)
382 {
383 	return file_setup_area_common(nr_hpages, FILE_SETUP_READ_WRITE_FS_READ_DATA);
384 }
385 
386 static void *file_setup_read_write_fs_write_area(int nr_hpages)
387 {
388 	return file_setup_area_common(nr_hpages, FILE_SETUP_READ_WRITE_FS_WRITE_DATA);
389 }
390 
391 static void file_cleanup_area(void *p, unsigned long size)
392 {
393 	munmap(p, size);
394 	close(finfo.fd);
395 	unlink(finfo.path);
396 }
397 
398 static void file_fault_read(void *p, unsigned long start, unsigned long end)
399 {
400 	if (madvise(((char *)p) + start, end - start, MADV_POPULATE_READ))
401 		ksft_exit_fail_perror("madvise(MADV_POPULATE_READ)");
402 }
403 
404 static void file_fault_read_and_flush(void *p, unsigned long start, unsigned long end)
405 {
406 	file_fault_read(p, start, end);
407 	/*
408 	 * make folio clean, since dirty folios from read&write file are
409 	 * rejected and not flushed
410 	 */
411 	msync((char *)p + start, end - start, MS_SYNC);
412 }
413 
414 static void file_fault_write(void *p, unsigned long start, unsigned long end)
415 {
416 	if (madvise(((char *)p) + start, end - start, MADV_POPULATE_WRITE))
417 		ksft_exit_fail_perror("madvise(MADV_POPULATE_WRITE)");
418 }
419 
420 static bool file_check_huge(void *addr, size_t len, int nr_hpages,
421 		unsigned long hpage_size)
422 {
423 	switch (finfo.type) {
424 	case VMA_FILE:
425 		return check_huge_file(addr, len, nr_hpages, hpage_size);
426 	case VMA_SHMEM:
427 		return check_huge_shmem(addr, len, nr_hpages, hpage_size);
428 	default:
429 		exit(EXIT_FAILURE);
430 		return false;
431 	}
432 }
433 
434 static void *shmem_setup_area(int nr_hpages)
435 {
436 	void *p;
437 	unsigned long size = nr_hpages * hpage_pmd_size;
438 
439 	finfo.fd = memfd_create("khugepaged-selftest-collapse-shmem", 0);
440 	if (finfo.fd < 0)
441 		ksft_exit_fail_perror("memfd_create()");
442 	if (ftruncate(finfo.fd, size))
443 		ksft_exit_fail_perror("ftruncate()");
444 	p = mmap(BASE_ADDR, size, PROT_READ | PROT_WRITE, MAP_SHARED, finfo.fd,
445 		 0);
446 	if (p != BASE_ADDR)
447 		ksft_exit_fail_perror("mmap()");
448 	return p;
449 }
450 
451 static void shmem_cleanup_area(void *p, unsigned long size)
452 {
453 	munmap(p, size);
454 	close(finfo.fd);
455 }
456 
457 static bool shmem_check_huge(void *addr, size_t len, int nr_hpages,
458 		unsigned long hpage_size)
459 {
460 	return check_huge_shmem(addr, len, nr_hpages, hpage_size);
461 }
462 
463 static struct mem_ops __anon_ops = {
464 	.setup_area = &anon_setup_area,
465 	.cleanup_area = &anon_cleanup_area,
466 	.fault = &anon_fault,
467 	.check_huge = &anon_check_huge,
468 	.name = "anon",
469 };
470 
471 static struct mem_ops __read_only_file_ops = {
472 	.setup_area = &file_setup_read_only_area,
473 	.cleanup_area = &file_cleanup_area,
474 	.fault = &file_fault_read,
475 	.check_huge = &file_check_huge,
476 	.name = "file",
477 };
478 
479 static struct mem_ops __read_write_file_read_ops = {
480 	.setup_area = &file_setup_read_write_fs_read_area,
481 	.cleanup_area = &file_cleanup_area,
482 	.fault = &file_fault_read_and_flush,
483 	.check_huge = &file_check_huge,
484 	.name = "file",
485 };
486 
487 static struct mem_ops __read_write_file_write_ops = {
488 	.setup_area = &file_setup_read_write_fs_write_area,
489 	.cleanup_area = &file_cleanup_area,
490 	.fault = &file_fault_write,
491 	.check_huge = &file_check_huge,
492 	.name = "file",
493 };
494 
495 static struct mem_ops __shmem_ops = {
496 	.setup_area = &shmem_setup_area,
497 	.cleanup_area = &shmem_cleanup_area,
498 	.fault = &anon_fault,
499 	.check_huge = &shmem_check_huge,
500 	.name = "shmem",
501 };
502 
503 static bool is_tmpfs(struct mem_ops *ops)
504 {
505 	return (ops == &__read_only_file_ops ||
506 		ops == &__read_write_file_read_ops ||
507 		ops == &__read_write_file_write_ops) &&
508 	       finfo.type == VMA_SHMEM;
509 }
510 
511 static bool is_anon(struct mem_ops *ops)
512 {
513 	return ops == &__anon_ops;
514 }
515 
516 static void __madvise_collapse(const char *msg, char *p, int nr_hpages,
517 			       struct mem_ops *ops, bool expect)
518 {
519 	int ret;
520 	struct thp_settings settings = *thp_current_settings();
521 
522 	ksft_print_msg("%s...", msg);
523 
524 	/*
525 	 * read&write file collapse succeeds for MADV_COLLAPSE because dirty
526 	 * folios are written back after collapse fails for dirty folios and
527 	 * another collapse is attempted.
528 	 */
529 
530 	/*
531 	 * Prevent khugepaged interference and tests that MADV_COLLAPSE
532 	 * ignores /sys/kernel/mm/transparent_hugepage/enabled
533 	 */
534 	settings.thp_enabled = THP_NEVER;
535 	settings.shmem_enabled = SHMEM_NEVER;
536 	thp_push_settings(&settings);
537 
538 	/* Clear VM_NOHUGEPAGE */
539 	madvise(p, nr_hpages * hpage_pmd_size, MADV_HUGEPAGE);
540 	ret = madvise_collapse_retry(p, nr_hpages * hpage_pmd_size);
541 	if (((bool)ret) == expect)
542 		fail("Fail: Bad return value");
543 	else if (!ops->check_huge(p, nr_hpages * hpage_pmd_size, expect ? nr_hpages : 0, hpage_pmd_size))
544 		fail("Fail: check_huge()");
545 	else
546 		success("OK");
547 
548 	thp_pop_settings();
549 }
550 
551 static void madvise_collapse(const char *msg, char *p, int nr_hpages,
552 			     struct mem_ops *ops, bool expect)
553 {
554 	/* Sanity check */
555 	if (!ops->check_huge(p, nr_hpages * hpage_pmd_size, 0, hpage_pmd_size))
556 		ksft_exit_fail_msg("Unexpected huge page\n");
557 	__madvise_collapse(msg, p, nr_hpages, ops, expect);
558 }
559 
560 #define TICK 500000
561 static bool wait_for_scan(const char *msg, char *p, size_t len,
562 		int nr_hpages, int collap_order, struct mem_ops *ops)
563 {
564 	unsigned long hpage_size = page_size << collap_order;
565 	int full_scans;
566 	int timeout = 6; /* 3 seconds */
567 
568 	/* Sanity check */
569 	if (!ops->check_huge(p, len, 0, hpage_size))
570 		ksft_exit_fail_msg("Unexpected huge page\n");
571 
572 	madvise(p, len, MADV_HUGEPAGE);
573 
574 	/* Wait until the second full_scan completed */
575 	full_scans = thp_read_num("khugepaged/full_scans") + 2;
576 
577 	ksft_print_msg("%s...", msg);
578 	while (timeout--) {
579 		if (ops->check_huge(p, len, nr_hpages, hpage_size))
580 			break;
581 		if (thp_read_num("khugepaged/full_scans") >= full_scans)
582 			break;
583 		printf(".");
584 		usleep(TICK);
585 	}
586 
587 	return timeout == -1;
588 }
589 
590 static void khugepaged_collapse(const char *msg, char *p, int nr_hpages,
591 				struct mem_ops *ops, bool expect)
592 {
593 	size_t len = nr_hpages * hpage_pmd_size;
594 
595 	/*
596 	 * read&write file collapse fails since khugepaged does not flush
597 	 * the target dirty folios
598 	 */
599 	if (!is_tmpfs(ops) && ops == &__read_write_file_write_ops)
600 		expect = false;
601 
602 	if (wait_for_scan(msg, p, len, nr_hpages, hpage_pmd_order, ops)) {
603 		if (expect)
604 			fail("Timeout");
605 		else
606 			success("OK");
607 		return;
608 	}
609 
610 	/*
611 	 * For file and shmem memory, khugepaged only retracts pte entries after
612 	 * putting the new hugepage in the page cache. The hugepage must be
613 	 * subsequently refaulted to install the pmd mapping for the mm.
614 	 */
615 	if (ops != &__anon_ops)
616 		ops->fault(p, 0, nr_hpages * hpage_pmd_size);
617 
618 	if (ops->check_huge(p, len, expect ? nr_hpages : 0, hpage_pmd_size))
619 		success("OK");
620 	else
621 		fail("Fail");
622 }
623 
624 static void mthp_khugepaged_collapse(const char *msg, char *p, int nr_hpages,
625 				struct mem_ops *ops, bool expect)
626 {
627 	unsigned long hpage_size = page_size << collapse_order;
628 	struct thp_settings settings = *thp_current_settings();
629 	/* mTHP collpase only allocates PMD sized memory */
630 	size_t len = hpage_pmd_size;
631 
632 	/* Set mTHP setting for mTHP collapse */
633 	if (ops == &__anon_ops) {
634 		settings.thp_enabled = THP_NEVER;
635 		settings.hugepages[collapse_order].enabled = THP_MADVISE;
636 	}
637 
638 	thp_push_settings(&settings);
639 
640 	if (wait_for_scan(msg, p, len, nr_hpages, collapse_order, ops)) {
641 		if (expect)
642 			fail("Timeout");
643 		else
644 			success("OK");
645 
646 		/* Restore THP settings for mTHP collapse. */
647 		thp_pop_settings();
648 		return;
649 	}
650 
651 	/*
652 	 * For file and shmem memory, khugepaged only retracts pte entries after
653 	 * putting the new hugepage in the page cache. The hugepage must be
654 	 * subsequently refaulted to install the pmd mapping for the mm.
655 	 */
656 	if (ops != &__anon_ops)
657 		ops->fault(p, 0, nr_hpages * hpage_size);
658 
659 	if (ops->check_huge(p, len, expect ? nr_hpages : 0, hpage_size))
660 		success("OK");
661 	else
662 		fail("Fail");
663 
664 	/* Restore THP settings for mTHP collapse. */
665 	thp_pop_settings();
666 }
667 
668 static struct collapse_context __khugepaged_context = {
669 	.collapse = &khugepaged_collapse,
670 	.enforce_pte_scan_limits = true,
671 	.name = "khugepaged",
672 };
673 
674 static struct collapse_context __mthp_khugepaged_context = {
675 	.collapse = &mthp_khugepaged_collapse,
676 	.enforce_pte_scan_limits = true,
677 	.name = "mthp_khugepaged",
678 };
679 
680 static struct collapse_context __madvise_context = {
681 	.collapse = &madvise_collapse,
682 	.enforce_pte_scan_limits = false,
683 	.name = "madvise",
684 };
685 
686 static void alloc_at_fault(void)
687 {
688 	struct thp_settings settings = *thp_current_settings();
689 	char *p;
690 
691 	settings.thp_enabled = THP_ALWAYS;
692 	thp_push_settings(&settings);
693 
694 	p = alloc_mapping(1);
695 	*p = 1;
696 	ksft_print_msg("Allocate huge page on fault...");
697 	if (check_huge_anon(p, hpage_pmd_size, 1, hpage_pmd_size))
698 		success("OK");
699 	else
700 		fail("Fail");
701 
702 	thp_pop_settings();
703 
704 	madvise(p, page_size, MADV_DONTNEED);
705 	ksft_print_msg("Split huge PMD on MADV_DONTNEED...");
706 	if (check_huge_anon(p, hpage_pmd_size, 0, hpage_pmd_size))
707 		success("OK");
708 	else
709 		fail("Fail");
710 	munmap(p, hpage_pmd_size);
711 
712 	ksft_test_result_report(exit_status, "allocate on fault and split\n");
713 }
714 
715 static void collapse_full(struct collapse_context *c, struct mem_ops *ops)
716 {
717 	void *p;
718 	int nr_pmds = 4, nr_hpages = 4;
719 	unsigned long size = nr_hpages * hpage_pmd_size;
720 
721 	/* Only try 1 PMD sized range for mTHP collapse. */
722 	if (c == &__mthp_khugepaged_context) {
723 		nr_pmds = 1;
724 		nr_hpages = 1 << (hpage_pmd_order - collapse_order);
725 		size = hpage_pmd_size;
726 	}
727 
728 	p = ops->setup_area(nr_pmds);
729 	ops->fault(p, 0, size);
730 	c->collapse("Collapse multiple fully populated PTE table", p, nr_hpages,
731 		    ops, true);
732 	validate_memory(p, 0, size);
733 	ops->cleanup_area(p, size);
734 
735 	ksft_test_result_report(exit_status, "%s\n", __func__);
736 }
737 
738 static void collapse_empty(struct collapse_context *c, struct mem_ops *ops)
739 {
740 	int nr_hpages = 1;
741 	void *p;
742 
743 	if (c == &__mthp_khugepaged_context)
744 		nr_hpages = 1 << (hpage_pmd_order - collapse_order);
745 
746 	p = ops->setup_area(1);
747 	c->collapse("Do not collapse empty PTE table", p, nr_hpages, ops, false);
748 	ops->cleanup_area(p, hpage_pmd_size);
749 	ksft_test_result_report(exit_status, "%s\n", __func__);
750 }
751 
752 static void collapse_single_mthp(struct collapse_context *c, struct mem_ops *ops)
753 {
754 	unsigned long hpage_size = page_size << collapse_order;
755 	void *p;
756 
757 	p = ops->setup_area(1);
758 	/*
759 	 * Only fault collapse_order sized ranges, and only check 1
760 	 * collapse_order sized huge page.
761 	 */
762 	ops->fault(p, 0, hpage_size);
763 	c->collapse("Collapse PTE table with half PTE entries present",
764 		p, 1, ops, true);
765 	ops->cleanup_area(p, hpage_pmd_size);
766 	ksft_test_result_report(exit_status, "%s\n", __func__);
767 }
768 
769 static void collapse_single_pte_entry(struct collapse_context *c, struct mem_ops *ops)
770 {
771 	void *p;
772 
773 	p = ops->setup_area(1);
774 	ops->fault(p, 0, page_size);
775 	c->collapse("Collapse PTE table with single PTE entry present", p,
776 		    1, ops, true);
777 	ops->cleanup_area(p, hpage_pmd_size);
778 	ksft_test_result_report(exit_status, "%s\n", __func__);
779 }
780 
781 static void collapse_max_ptes_none(struct collapse_context *c, struct mem_ops *ops)
782 {
783 	int max_ptes_none = hpage_pmd_nr / 2;
784 	struct thp_settings settings = *thp_current_settings();
785 	void *p;
786 	int fault_nr_pages = is_anon(ops) ? 1 << anon_order : 1;
787 
788 	settings.khugepaged.max_ptes_none = max_ptes_none;
789 	thp_push_settings(&settings);
790 
791 	p = ops->setup_area(1);
792 
793 	if (is_tmpfs(ops)) {
794 		/* shmem pages always in the page cache */
795 		printf("tmpfs...");
796 		skip("Skip");
797 		goto skip;
798 	}
799 
800 	ops->fault(p, 0, (hpage_pmd_nr - max_ptes_none - fault_nr_pages) * page_size);
801 	c->collapse("Maybe collapse with max_ptes_none exceeded", p, 1,
802 		    ops, !c->enforce_pte_scan_limits);
803 	validate_memory(p, 0, (hpage_pmd_nr - max_ptes_none - fault_nr_pages) * page_size);
804 
805 	if (c->enforce_pte_scan_limits) {
806 		ops->cleanup_area(p, hpage_pmd_size);
807 		p = ops->setup_area(1);
808 
809 		ops->fault(p, 0, (hpage_pmd_nr - max_ptes_none) * page_size);
810 		c->collapse("Collapse with max_ptes_none PTEs empty", p, 1, ops,
811 			    true);
812 		validate_memory(p, 0,
813 				(hpage_pmd_nr - max_ptes_none) * page_size);
814 	}
815 skip:
816 	ops->cleanup_area(p, hpage_pmd_size);
817 	thp_pop_settings();
818 	ksft_test_result_report(exit_status, "%s\n", __func__);
819 }
820 
821 static void collapse_swapin_single_pte(struct collapse_context *c, struct mem_ops *ops)
822 {
823 	void *p;
824 
825 	p = ops->setup_area(1);
826 	ops->fault(p, 0, hpage_pmd_size);
827 
828 	ksft_print_msg("Swapout one page...");
829 	if (madvise(p, page_size, MADV_PAGEOUT))
830 		ksft_exit_fail_perror("madvise(MADV_PAGEOUT)");
831 	if (check_swap(p, page_size)) {
832 		success("OK");
833 	} else {
834 		fail("Fail");
835 		goto out;
836 	}
837 
838 	c->collapse("Collapse with swapping in single PTE entry", p, 1, ops,
839 		    true);
840 	validate_memory(p, 0, hpage_pmd_size);
841 out:
842 	ops->cleanup_area(p, hpage_pmd_size);
843 	ksft_test_result_report(exit_status, "%s\n", __func__);
844 }
845 
846 static void collapse_max_ptes_swap(struct collapse_context *c, struct mem_ops *ops)
847 {
848 	int max_ptes_swap = thp_read_num("khugepaged/max_ptes_swap");
849 	void *p;
850 
851 	p = ops->setup_area(1);
852 	ops->fault(p, 0, hpage_pmd_size);
853 
854 	ksft_print_msg("Swapout %d of %d pages...", max_ptes_swap + 1, hpage_pmd_nr);
855 	if (madvise(p, (max_ptes_swap + 1) * page_size, MADV_PAGEOUT))
856 		ksft_exit_fail_perror("madvise(MADV_PAGEOUT)");
857 	if (check_swap(p, (max_ptes_swap + 1) * page_size)) {
858 		success("OK");
859 	} else {
860 		fail("Fail");
861 		goto out;
862 	}
863 
864 	c->collapse("Maybe collapse with max_ptes_swap exceeded", p, 1, ops,
865 		    !c->enforce_pte_scan_limits);
866 	validate_memory(p, 0, hpage_pmd_size);
867 
868 	if (c->enforce_pte_scan_limits) {
869 		ops->fault(p, 0, hpage_pmd_size);
870 		ksft_print_msg("Swapout %d of %d pages...", max_ptes_swap,
871 		       hpage_pmd_nr);
872 		if (madvise(p, max_ptes_swap * page_size, MADV_PAGEOUT))
873 			ksft_exit_fail_perror("madvise(MADV_PAGEOUT)");
874 		if (check_swap(p, max_ptes_swap * page_size)) {
875 			success("OK");
876 		} else {
877 			fail("Fail");
878 			goto out;
879 		}
880 
881 		c->collapse("Collapse with max_ptes_swap pages swapped out", p,
882 			    1, ops, true);
883 		validate_memory(p, 0, hpage_pmd_size);
884 	}
885 out:
886 	ops->cleanup_area(p, hpage_pmd_size);
887 	ksft_test_result_report(exit_status, "%s\n", __func__);
888 }
889 
890 static void collapse_single_pte_entry_compound(struct collapse_context *c, struct mem_ops *ops)
891 {
892 	void *p;
893 
894 	p = alloc_hpage(ops);
895 
896 	if (is_tmpfs(ops)) {
897 		/* MADV_DONTNEED won't evict tmpfs pages */
898 		printf("tmpfs...");
899 		skip("Skip");
900 		goto skip;
901 	}
902 
903 	madvise(p, hpage_pmd_size, MADV_NOHUGEPAGE);
904 	ksft_print_msg("Split huge page leaving single PTE mapping compound page...");
905 	madvise(p + page_size, hpage_pmd_size - page_size, MADV_DONTNEED);
906 	if (ops->check_huge(p, hpage_pmd_size, 0, hpage_pmd_size))
907 		success("OK");
908 	else
909 		fail("Fail");
910 
911 	c->collapse("Collapse PTE table with single PTE mapping compound page",
912 		    p, 1, ops, true);
913 	validate_memory(p, 0, page_size);
914 skip:
915 	ops->cleanup_area(p, hpage_pmd_size);
916 	ksft_test_result_report(exit_status, "%s\n", __func__);
917 }
918 
919 static void collapse_full_of_compound(struct collapse_context *c, struct mem_ops *ops)
920 {
921 	void *p;
922 
923 	p = alloc_hpage(ops);
924 	ksft_print_msg("Split huge page leaving single PTE page table full of compound pages...");
925 	madvise(p, page_size, MADV_NOHUGEPAGE);
926 	madvise(p, hpage_pmd_size, MADV_NOHUGEPAGE);
927 	if (ops->check_huge(p, hpage_pmd_size, 0, hpage_pmd_size))
928 		success("OK");
929 	else
930 		fail("Fail");
931 
932 	c->collapse("Collapse PTE table full of compound pages", p, 1, ops,
933 		    true);
934 	validate_memory(p, 0, hpage_pmd_size);
935 	ops->cleanup_area(p, hpage_pmd_size);
936 	ksft_test_result_report(exit_status, "%s\n", __func__);
937 }
938 
939 static void collapse_compound_extreme(struct collapse_context *c, struct mem_ops *ops)
940 {
941 	void *p;
942 	int i;
943 
944 	p = ops->setup_area(1);
945 	ksft_print_msg("Construct PTE page table full of different PTE-mapped compound pages\n");
946 	for (i = 0; i < hpage_pmd_nr; i++) {
947 		madvise(BASE_ADDR, hpage_pmd_size, MADV_HUGEPAGE);
948 		ops->fault(BASE_ADDR, 0, hpage_pmd_size);
949 		if (!ops->check_huge(BASE_ADDR, hpage_pmd_size, 1, hpage_pmd_size))
950 			ksft_exit_fail_msg("Failed to allocate huge page\n");
951 		madvise(BASE_ADDR, hpage_pmd_size, MADV_NOHUGEPAGE);
952 
953 		p = mremap(BASE_ADDR - i * page_size,
954 				i * page_size + hpage_pmd_size,
955 				(i + 1) * page_size,
956 				MREMAP_MAYMOVE | MREMAP_FIXED,
957 				BASE_ADDR + 2 * hpage_pmd_size);
958 		if (p == MAP_FAILED)
959 			ksft_exit_fail_perror("mremap+unmap");
960 
961 		p = mremap(BASE_ADDR + 2 * hpage_pmd_size,
962 				(i + 1) * page_size,
963 				(i + 1) * page_size + hpage_pmd_size,
964 				MREMAP_MAYMOVE | MREMAP_FIXED,
965 				BASE_ADDR - (i + 1) * page_size);
966 		if (p == MAP_FAILED)
967 			ksft_exit_fail_perror("mremap+alloc");
968 	}
969 
970 	ops->cleanup_area(BASE_ADDR, hpage_pmd_size);
971 	ops->fault(p, 0, hpage_pmd_size);
972 	if (!ops->check_huge(p, hpage_pmd_size, 1, hpage_pmd_size))
973 		success("OK");
974 	else
975 		fail("Fail");
976 
977 	c->collapse("Collapse PTE table full of different compound pages", p, 1,
978 		    ops, true);
979 
980 	validate_memory(p, 0, hpage_pmd_size);
981 	ops->cleanup_area(p, hpage_pmd_size);
982 	ksft_test_result_report(exit_status, "%s\n", __func__);
983 }
984 
985 static void collapse_fork(struct collapse_context *c, struct mem_ops *ops)
986 {
987 	int wstatus;
988 	void *p;
989 
990 	p = ops->setup_area(1);
991 
992 	ksft_print_msg("Allocate small page...");
993 	ops->fault(p, 0, page_size);
994 	if (ops->check_huge(p, hpage_pmd_size, 0, hpage_pmd_size))
995 		success("OK");
996 	else
997 		fail("Fail");
998 
999 	ksft_print_msg("Share small page over fork()...");
1000 	if (!fork()) {
1001 		/* Do not touch settings on child exit */
1002 		if (ops->check_huge(p, hpage_pmd_size, 0, hpage_pmd_size))
1003 			success("OK");
1004 		else
1005 			fail("Fail");
1006 
1007 		ops->fault(p, page_size, 2 * page_size);
1008 		c->collapse("Collapse PTE table with single page shared with parent process",
1009 			    p, 1, ops, true);
1010 
1011 		validate_memory(p, 0, page_size);
1012 		ops->cleanup_area(p, hpage_pmd_size);
1013 		_exit(exit_status);
1014 	}
1015 
1016 	wait(&wstatus);
1017 	exit_status = WEXITSTATUS(wstatus);
1018 
1019 	ksft_print_msg("Check if parent still has small page...");
1020 	if (ops->check_huge(p, hpage_pmd_size, 0, hpage_pmd_size))
1021 		success("OK");
1022 	else
1023 		fail("Fail");
1024 	validate_memory(p, 0, page_size);
1025 	ops->cleanup_area(p, hpage_pmd_size);
1026 	ksft_test_result_report(exit_status, "%s\n", __func__);
1027 }
1028 
1029 static void collapse_fork_compound(struct collapse_context *c, struct mem_ops *ops)
1030 {
1031 	int wstatus;
1032 	void *p;
1033 
1034 	p = alloc_hpage(ops);
1035 	ksft_print_msg("Share huge page over fork()...");
1036 	if (!fork()) {
1037 		/* Do not touch settings on child exit */
1038 		if (ops->check_huge(p, hpage_pmd_size, 1, hpage_pmd_size))
1039 			success("OK");
1040 		else
1041 			fail("Fail");
1042 
1043 		ksft_print_msg("Split huge page PMD in child process...");
1044 		madvise(p, page_size, MADV_NOHUGEPAGE);
1045 		madvise(p, hpage_pmd_size, MADV_NOHUGEPAGE);
1046 		if (ops->check_huge(p, hpage_pmd_size, 0, hpage_pmd_size))
1047 			success("OK");
1048 		else
1049 			fail("Fail");
1050 		ops->fault(p, 0, page_size);
1051 
1052 		thp_write_num("khugepaged/max_ptes_shared", hpage_pmd_nr - 1);
1053 		c->collapse("Collapse PTE table full of compound pages in child",
1054 			    p, 1, ops, true);
1055 		thp_write_num("khugepaged/max_ptes_shared",
1056 			  thp_current_settings()->khugepaged.max_ptes_shared);
1057 
1058 		validate_memory(p, 0, hpage_pmd_size);
1059 		ops->cleanup_area(p, hpage_pmd_size);
1060 		_exit(exit_status);
1061 	}
1062 
1063 	wait(&wstatus);
1064 	exit_status = WEXITSTATUS(wstatus);
1065 
1066 	ksft_print_msg("Check if parent still has huge page...");
1067 	if (ops->check_huge(p, hpage_pmd_size, 1, hpage_pmd_size))
1068 		success("OK");
1069 	else
1070 		fail("Fail");
1071 	validate_memory(p, 0, hpage_pmd_size);
1072 	ops->cleanup_area(p, hpage_pmd_size);
1073 	ksft_test_result_report(exit_status, "%s\n", __func__);
1074 }
1075 
1076 static void collapse_max_ptes_shared(struct collapse_context *c, struct mem_ops *ops)
1077 {
1078 	int max_ptes_shared = thp_read_num("khugepaged/max_ptes_shared");
1079 	int wstatus;
1080 	void *p;
1081 
1082 	p = alloc_hpage(ops);
1083 	ksft_print_msg("Share huge page over fork()...");
1084 	if (!fork()) {
1085 		/* Do not touch settings on child exit */
1086 		if (ops->check_huge(p, hpage_pmd_size, 1, hpage_pmd_size))
1087 			success("OK");
1088 		else
1089 			fail("Fail");
1090 
1091 		ksft_print_msg("Trigger CoW on page %d of %d...",
1092 				hpage_pmd_nr - max_ptes_shared - 1, hpage_pmd_nr);
1093 		ops->fault(p, 0, (hpage_pmd_nr - max_ptes_shared - 1) * page_size);
1094 		if (ops->check_huge(p, hpage_pmd_size, 0, hpage_pmd_size))
1095 			success("OK");
1096 		else
1097 			fail("Fail");
1098 
1099 		c->collapse("Maybe collapse with max_ptes_shared exceeded", p,
1100 			    1, ops, !c->enforce_pte_scan_limits);
1101 
1102 		if (c->enforce_pte_scan_limits) {
1103 			ksft_print_msg("Trigger CoW on page %d of %d...",
1104 			       hpage_pmd_nr - max_ptes_shared, hpage_pmd_nr);
1105 			ops->fault(p, 0, (hpage_pmd_nr - max_ptes_shared) *
1106 				    page_size);
1107 			if (ops->check_huge(p, hpage_pmd_size, 0, hpage_pmd_size))
1108 				success("OK");
1109 			else
1110 				fail("Fail");
1111 
1112 			c->collapse("Collapse with max_ptes_shared PTEs shared",
1113 				    p, 1, ops, true);
1114 		}
1115 
1116 		validate_memory(p, 0, hpage_pmd_size);
1117 		ops->cleanup_area(p, hpage_pmd_size);
1118 		_exit(exit_status);
1119 	}
1120 
1121 	wait(&wstatus);
1122 	exit_status = WEXITSTATUS(wstatus);
1123 
1124 	ksft_print_msg("Check if parent still has huge page...");
1125 	if (ops->check_huge(p, hpage_pmd_size, 1, hpage_pmd_size))
1126 		success("OK");
1127 	else
1128 		fail("Fail");
1129 	validate_memory(p, 0, hpage_pmd_size);
1130 	ops->cleanup_area(p, hpage_pmd_size);
1131 	ksft_test_result_report(exit_status, "%s\n", __func__);
1132 }
1133 
1134 static void madvise_collapse_existing_thps(struct collapse_context *c,
1135 					   struct mem_ops *ops)
1136 {
1137 	void *p;
1138 
1139 	p = ops->setup_area(1);
1140 	ops->fault(p, 0, hpage_pmd_size);
1141 	c->collapse("Collapse fully populated PTE table...", p, 1, ops, true);
1142 	validate_memory(p, 0, hpage_pmd_size);
1143 
1144 	/* c->collapse() will find a hugepage and complain - call directly. */
1145 	__madvise_collapse("Re-collapse PMD-mapped hugepage", p, 1, ops, true);
1146 	validate_memory(p, 0, hpage_pmd_size);
1147 	ops->cleanup_area(p, hpage_pmd_size);
1148 	ksft_test_result_report(exit_status, "%s\n", __func__);
1149 }
1150 
1151 /*
1152  * Test race with khugepaged where page tables have been retracted and
1153  * pmd cleared.
1154  */
1155 static void madvise_retracted_page_tables(struct collapse_context *c,
1156 					  struct mem_ops *ops)
1157 {
1158 	void *p;
1159 	int nr_hpages = 1;
1160 	unsigned long size = nr_hpages * hpage_pmd_size;
1161 
1162 	p = ops->setup_area(nr_hpages);
1163 	ops->fault(p, 0, size);
1164 
1165 	/* Let khugepaged collapse and leave pmd cleared */
1166 	if (wait_for_scan("Collapse and leave PMD cleared", p, size, nr_hpages,
1167 			  hpage_pmd_order, ops)) {
1168 		fail("Timeout");
1169 		return;
1170 	}
1171 	success("OK");
1172 	c->collapse("Install huge PMD from page cache", p, nr_hpages, ops,
1173 		    true);
1174 	validate_memory(p, 0, size);
1175 	ops->cleanup_area(p, size);
1176 	ksft_test_result_report(exit_status, "%s\n", __func__);
1177 }
1178 
1179 static void usage(void)
1180 {
1181 	fprintf(stderr, "\nUsage: ./khugepaged [OPTIONS] <test type> [dir]\n\n");
1182 	fprintf(stderr, "\t<test type>\t: <context>:<mem_type>\n");
1183 	fprintf(stderr, "\t<context>\t: [all|khugepaged|mthp_khugepaged|madvise]\n");
1184 	fprintf(stderr, "\t<mem_type>\t: [all|anon|file|shmem]\n");
1185 	fprintf(stderr, "\n\t\"file,all\" mem_type requires [dir] argument\n");
1186 	fprintf(stderr, "\n\t\"file,all\" mem_type requires a file system\n");
1187 	fprintf(stderr,	"\twith PMD-sized large folio support\n");
1188 	fprintf(stderr, "\n\tif [dir] is a (sub)directory of a tmpfs mount, tmpfs must be\n");
1189 	fprintf(stderr,	"\tmounted with huge=advise option for khugepaged tests to work\n");
1190 	fprintf(stderr, "\n\tmthp_khugepaged only supports anon mem_type now.\n");
1191 	fprintf(stderr,	"\n\tSupported Options:\n");
1192 	fprintf(stderr,	"\t\t-h: This help message.\n");
1193 	fprintf(stderr,	"\t\t-s: mTHP size, expressed as page order.\n");
1194 	fprintf(stderr,	"\t\t    Defaults to 0. Use this size for anon or shmem allocations.\n");
1195 	fprintf(stderr,	"\t\t-c: collapse order for mTHP collapse, expressed as page order.\n");
1196 	exit(1);
1197 }
1198 
1199 static void parse_test_type(int argc, char **argv)
1200 {
1201 	int opt;
1202 	char *buf;
1203 	const char *token;
1204 
1205 	while ((opt = getopt(argc, argv, "s:c:h")) != -1) {
1206 		switch (opt) {
1207 		case 's':
1208 			anon_order = atoi(optarg);
1209 			break;
1210 		case 'c':
1211 			collapse_order = atoi(optarg);
1212 			break;
1213 		case 'h':
1214 		default:
1215 			usage();
1216 		}
1217 	}
1218 
1219 	argv += optind;
1220 	argc -= optind;
1221 
1222 	if (argc == 0) {
1223 		/* Backwards compatibility */
1224 		khugepaged_context =  &__khugepaged_context;
1225 		madvise_context =  &__madvise_context;
1226 		anon_ops = &__anon_ops;
1227 		return;
1228 	}
1229 
1230 	buf = strdup(argv[0]);
1231 	token = strsep(&buf, ":");
1232 
1233 	if (!strcmp(token, "all")) {
1234 		khugepaged_context =  &__khugepaged_context;
1235 		madvise_context =  &__madvise_context;
1236 	} else if (!strcmp(token, "khugepaged")) {
1237 		khugepaged_context =  &__khugepaged_context;
1238 	} else if (!strcmp(token, "mthp_khugepaged")) {
1239 		mthp_khugepaged_context =  &__mthp_khugepaged_context;
1240 		if (collapse_order <= 0 || collapse_order >= hpage_pmd_order)
1241 			usage();
1242 	} else if (!strcmp(token, "madvise")) {
1243 		madvise_context =  &__madvise_context;
1244 	} else {
1245 		usage();
1246 	}
1247 
1248 	if (!buf)
1249 		usage();
1250 
1251 	if (!strcmp(buf, "all")) {
1252 		read_only_file_ops =  &__read_only_file_ops;
1253 		read_write_file_read_ops =  &__read_write_file_read_ops;
1254 		read_write_file_write_ops =  &__read_write_file_write_ops;
1255 		anon_ops = &__anon_ops;
1256 		shmem_ops = &__shmem_ops;
1257 		if (mthp_khugepaged_context)
1258 			usage();
1259 	} else if (!strcmp(buf, "anon")) {
1260 		anon_ops = &__anon_ops;
1261 	} else if (!strcmp(buf, "file")) {
1262 		read_only_file_ops =  &__read_only_file_ops;
1263 		read_write_file_read_ops =  &__read_write_file_read_ops;
1264 		read_write_file_write_ops =  &__read_write_file_write_ops;
1265 		if (mthp_khugepaged_context)
1266 			usage();
1267 	} else if (!strcmp(buf, "shmem")) {
1268 		shmem_ops = &__shmem_ops;
1269 		if (mthp_khugepaged_context)
1270 			usage();
1271 	} else {
1272 		usage();
1273 	}
1274 
1275 	if (!read_only_file_ops && !read_write_file_read_ops &&
1276 	    !read_write_file_write_ops)
1277 		return;
1278 
1279 	if (argc != 2)
1280 		usage();
1281 
1282 	get_finfo(argv[1]);
1283 }
1284 
1285 typedef void (*test_fn)(struct collapse_context *c, struct mem_ops *ops);
1286 
1287 struct test_case {
1288 	struct collapse_context *ctx;
1289 	struct mem_ops *ops;
1290 	const char *desc;
1291 	test_fn fn;
1292 };
1293 
1294 #define MAX_TEST_CASES 64
1295 static struct test_case test_cases[MAX_TEST_CASES];
1296 static int nr_test_cases;
1297 
1298 #define TEST(t, c, o) do {						\
1299 	if (c && o) {							\
1300 		if (nr_test_cases >= MAX_TEST_CASES)			\
1301 			ksft_exit_fail_msg("MAX_TEST_CASES is too small\n"); \
1302 		test_cases[nr_test_cases++] = (struct test_case){	\
1303 			.ctx	= c,					\
1304 			.ops	= o,					\
1305 			.desc	= #t,					\
1306 			.fn	= t,					\
1307 		};							\
1308 	}								\
1309 	} while (0)
1310 
1311 int main(int argc, char **argv)
1312 {
1313 	struct thp_settings default_settings = {
1314 		.thp_enabled = THP_MADVISE,
1315 		.thp_defrag = THP_DEFRAG_ALWAYS,
1316 		.shmem_enabled = SHMEM_ADVISE,
1317 		.use_zero_page = 0,
1318 		.khugepaged = {
1319 			.defrag = 1,
1320 			.alloc_sleep_millisecs = 10,
1321 			.scan_sleep_millisecs = 10,
1322 		},
1323 		/*
1324 		 * When testing file-backed memory, the collapse path
1325 		 * looks at how many pages are found in the page cache, not
1326 		 * what pages are mapped. Disable read ahead optimization so
1327 		 * pages don't find their way into the page cache unless
1328 		 * we mem_ops->fault() them in.
1329 		 */
1330 		.read_ahead_kb = 0,
1331 	};
1332 
1333 	ksft_print_header();
1334 
1335 	if (!thp_is_enabled())
1336 		ksft_exit_skip("Transparent Hugepages not available\n");
1337 
1338 	page_size = getpagesize();
1339 	hpage_pmd_size = read_pmd_pagesize();
1340 	if (!hpage_pmd_size)
1341 		ksft_exit_fail_msg("Reading PMD pagesize failed\n");
1342 	hpage_pmd_nr = hpage_pmd_size / page_size;
1343 	hpage_pmd_order = __builtin_ctz(hpage_pmd_nr);
1344 
1345 	parse_test_type(argc, argv);
1346 
1347 	setbuf(stdout, NULL);
1348 
1349 	default_settings.khugepaged.max_ptes_none = hpage_pmd_nr - 1;
1350 	default_settings.khugepaged.max_ptes_swap = hpage_pmd_nr / 8;
1351 	default_settings.khugepaged.max_ptes_shared = hpage_pmd_nr / 2;
1352 	default_settings.khugepaged.pages_to_scan = hpage_pmd_nr * 8;
1353 	default_settings.hugepages[hpage_pmd_order].enabled = THP_INHERIT;
1354 	default_settings.hugepages[anon_order].enabled = THP_ALWAYS;
1355 	default_settings.shmem_hugepages[hpage_pmd_order].enabled = SHMEM_INHERIT;
1356 	default_settings.shmem_hugepages[anon_order].enabled = SHMEM_ALWAYS;
1357 
1358 	save_settings();
1359 	thp_push_settings(&default_settings);
1360 
1361 	TEST(collapse_full, khugepaged_context, anon_ops);
1362 	TEST(collapse_full, khugepaged_context, read_only_file_ops);
1363 	TEST(collapse_full, khugepaged_context, read_write_file_read_ops);
1364 	TEST(collapse_full, khugepaged_context, read_write_file_write_ops);
1365 	TEST(collapse_full, khugepaged_context, shmem_ops);
1366 	TEST(collapse_full, mthp_khugepaged_context, anon_ops);
1367 	TEST(collapse_full, madvise_context, anon_ops);
1368 	TEST(collapse_full, madvise_context, read_only_file_ops);
1369 	TEST(collapse_full, madvise_context, read_write_file_read_ops);
1370 	TEST(collapse_full, madvise_context, read_write_file_write_ops);
1371 	TEST(collapse_full, madvise_context, shmem_ops);
1372 
1373 	TEST(collapse_empty, khugepaged_context, anon_ops);
1374 	TEST(collapse_empty, mthp_khugepaged_context, anon_ops);
1375 	TEST(collapse_empty, madvise_context, anon_ops);
1376 
1377 	TEST(collapse_single_mthp, mthp_khugepaged_context, anon_ops);
1378 
1379 	TEST(collapse_single_pte_entry, khugepaged_context, anon_ops);
1380 	TEST(collapse_single_pte_entry, khugepaged_context, read_only_file_ops);
1381 	TEST(collapse_single_pte_entry, khugepaged_context, read_write_file_read_ops);
1382 	TEST(collapse_single_pte_entry, khugepaged_context, read_write_file_write_ops);
1383 	TEST(collapse_single_pte_entry, khugepaged_context, shmem_ops);
1384 	TEST(collapse_single_pte_entry, madvise_context, anon_ops);
1385 	TEST(collapse_single_pte_entry, madvise_context, read_only_file_ops);
1386 	TEST(collapse_single_pte_entry, madvise_context, read_write_file_read_ops);
1387 	TEST(collapse_single_pte_entry, madvise_context, read_write_file_write_ops);
1388 	TEST(collapse_single_pte_entry, madvise_context, shmem_ops);
1389 
1390 	TEST(collapse_max_ptes_none, khugepaged_context, anon_ops);
1391 	TEST(collapse_max_ptes_none, khugepaged_context, read_only_file_ops);
1392 	TEST(collapse_max_ptes_none, khugepaged_context, read_write_file_read_ops);
1393 	TEST(collapse_max_ptes_none, khugepaged_context, read_write_file_write_ops);
1394 	TEST(collapse_max_ptes_none, madvise_context, anon_ops);
1395 	TEST(collapse_max_ptes_none, madvise_context, read_only_file_ops);
1396 	TEST(collapse_max_ptes_none, madvise_context, read_write_file_read_ops);
1397 	TEST(collapse_max_ptes_none, madvise_context, read_write_file_write_ops);
1398 
1399 	TEST(collapse_single_pte_entry_compound, khugepaged_context, anon_ops);
1400 	TEST(collapse_single_pte_entry_compound, khugepaged_context, read_only_file_ops);
1401 	TEST(collapse_single_pte_entry_compound, khugepaged_context, read_write_file_read_ops);
1402 	TEST(collapse_single_pte_entry_compound, madvise_context, anon_ops);
1403 	TEST(collapse_single_pte_entry_compound, madvise_context, read_only_file_ops);
1404 	TEST(collapse_single_pte_entry_compound, madvise_context, read_write_file_read_ops);
1405 
1406 	TEST(collapse_full_of_compound, khugepaged_context, anon_ops);
1407 	TEST(collapse_full_of_compound, khugepaged_context, read_only_file_ops);
1408 	TEST(collapse_full_of_compound, khugepaged_context, read_write_file_read_ops);
1409 	TEST(collapse_full_of_compound, khugepaged_context, shmem_ops);
1410 	TEST(collapse_full_of_compound, madvise_context, anon_ops);
1411 	TEST(collapse_full_of_compound, madvise_context, read_only_file_ops);
1412 	TEST(collapse_full_of_compound, madvise_context, read_write_file_read_ops);
1413 	TEST(collapse_full_of_compound, madvise_context, shmem_ops);
1414 
1415 	TEST(collapse_compound_extreme, khugepaged_context, anon_ops);
1416 	TEST(collapse_compound_extreme, madvise_context, anon_ops);
1417 
1418 	TEST(collapse_swapin_single_pte, khugepaged_context, anon_ops);
1419 	TEST(collapse_swapin_single_pte, madvise_context, anon_ops);
1420 
1421 	TEST(collapse_max_ptes_swap, khugepaged_context, anon_ops);
1422 	TEST(collapse_max_ptes_swap, madvise_context, anon_ops);
1423 
1424 	TEST(collapse_fork, khugepaged_context, anon_ops);
1425 	TEST(collapse_fork, madvise_context, anon_ops);
1426 
1427 	TEST(collapse_fork_compound, khugepaged_context, anon_ops);
1428 	TEST(collapse_fork_compound, madvise_context, anon_ops);
1429 
1430 	TEST(collapse_max_ptes_shared, khugepaged_context, anon_ops);
1431 	TEST(collapse_max_ptes_shared, madvise_context, anon_ops);
1432 
1433 	TEST(madvise_collapse_existing_thps, madvise_context, anon_ops);
1434 	TEST(madvise_collapse_existing_thps, madvise_context, read_only_file_ops);
1435 	TEST(madvise_collapse_existing_thps, madvise_context, read_write_file_read_ops);
1436 	TEST(madvise_collapse_existing_thps, madvise_context, shmem_ops);
1437 
1438 	TEST(madvise_retracted_page_tables, madvise_context, read_only_file_ops);
1439 	TEST(madvise_retracted_page_tables, madvise_context, read_write_file_read_ops);
1440 	TEST(madvise_retracted_page_tables, madvise_context, shmem_ops);
1441 
1442 	ksft_set_plan(nr_test_cases + 1);
1443 
1444 	alloc_at_fault();
1445 	for (int i = 0; i < nr_test_cases; i++) {
1446 		struct test_case *t = &test_cases[i];
1447 
1448 		ksft_print_msg("\n# Run test: %s (%s:%s)\n", t->desc, t->ctx->name, t->ops->name);
1449 		t->fn(t->ctx, t->ops);
1450 	}
1451 
1452 	ksft_finished();
1453 }
1454