xref: /linux/kernel/liveupdate/kexec_handover.c (revision 2f0f6b0773be0a1ec475097ae54848eea42adc7d)
1 // SPDX-License-Identifier: GPL-2.0-only
2 /*
3  * kexec_handover.c - kexec handover metadata processing
4  * Copyright (C) 2023 Alexander Graf <graf@amazon.com>
5  * Copyright (C) 2025 Microsoft Corporation, Mike Rapoport <rppt@kernel.org>
6  * Copyright (C) 2025 Google LLC, Changyuan Lyu <changyuanl@google.com>
7  * Copyright (C) 2025 Pasha Tatashin <pasha.tatashin@soleen.com>
8  * Copyright (C) 2026 Google LLC, Jason Miu <jasonmiu@google.com>
9  */
10 
11 #define pr_fmt(fmt) "KHO: " fmt
12 
13 #include <linux/cleanup.h>
14 #include <linux/cma.h>
15 #include <linux/kmemleak.h>
16 #include <linux/count_zeros.h>
17 #include <linux/kasan.h>
18 #include <linux/kexec.h>
19 #include <linux/kexec_handover.h>
20 #include <linux/kho_radix_tree.h>
21 #include <linux/utsname.h>
22 #include <linux/kho/abi/kexec_handover.h>
23 #include <linux/kho/abi/kexec_metadata.h>
24 #include <linux/libfdt.h>
25 #include <linux/list.h>
26 #include <linux/memblock.h>
27 #include <linux/page-isolation.h>
28 #include <linux/unaligned.h>
29 #include <linux/vmalloc.h>
30 
31 #include <asm/early_ioremap.h>
32 
33 /*
34  * KHO is tightly coupled with mm init and needs access to some of mm
35  * internal APIs.
36  */
37 #include "../../mm/mm_init.h"
38 #include "../../mm/vmalloc.h"
39 #include "../kexec_internal.h"
40 #include "kexec_handover_internal.h"
41 
42 /*
43  * This is the minimal alignment required by deferred struct page init.
44  * deferred_init_memmap_chunk frees memory to the buddy allocator, which looks
45  * at the neighboring pages (up to MAX_PAGE_ORDER) to merge them.
46  * If KHO scratch is not aligned to that value, buddy can access uninitialized
47  * struct pages, which can cause a crash.
48  */
49 #define SCRATCH_ALIGNMENT_BYTES (PAGE_SIZE * MAX_ORDER_NR_PAGES)
50 static_assert(SCRATCH_ALIGNMENT_BYTES >= CMA_MIN_ALIGNMENT_BYTES);
51 
52 /* The magic token for preserved pages */
53 #define KHO_PAGE_MAGIC 0x4b484f50U /* ASCII for 'KHOP' */
54 
55 /*
56  * KHO uses page->private, which is an unsigned long, to store page metadata.
57  * Use it to store both the magic and the order.
58  */
59 union kho_page_info {
60 	unsigned long page_private;
61 	struct {
62 		unsigned int order;
63 		unsigned int magic;
64 	};
65 };
66 
67 static_assert(sizeof(union kho_page_info) == sizeof(((struct page *)0)->private));
68 
69 static bool kho_enable __ro_after_init = IS_ENABLED(CONFIG_KEXEC_HANDOVER_ENABLE_DEFAULT);
70 
71 bool kho_is_enabled(void)
72 {
73 	return kho_enable;
74 }
75 EXPORT_SYMBOL_GPL(kho_is_enabled);
76 
77 static int __init kho_parse_enable(char *p)
78 {
79 	return kstrtobool(p, &kho_enable);
80 }
81 early_param("kho", kho_parse_enable);
82 
83 struct kho_out {
84 	void *fdt;
85 	struct mutex lock; /* protects KHO FDT */
86 
87 	struct kho_radix_tree radix_tree;
88 	struct kho_debugfs dbg;
89 };
90 
91 static struct kho_out kho_out = {
92 	.lock = __MUTEX_INITIALIZER(kho_out.lock),
93 	.radix_tree = {
94 		.lock = __MUTEX_INITIALIZER(kho_out.radix_tree.lock),
95 	},
96 };
97 
98 /**
99  * kho_radix_encode_key - Encodes a physical address and order into a radix key.
100  * @phys: The physical address of the page.
101  * @order: The order of the page.
102  *
103  * This function combines a page's physical address and its order into a
104  * single unsigned long, which is used as a key for all radix tree
105  * operations.
106  *
107  * Return: The encoded unsigned long radix key.
108  */
109 static unsigned long kho_radix_encode_key(phys_addr_t phys, unsigned int order)
110 {
111 	/* Order bits part */
112 	unsigned long h = 1UL << (KHO_ORDER_0_LOG2 - order);
113 	/* Shifted physical address part */
114 	unsigned long l = phys >> (PAGE_SHIFT + order);
115 
116 	return h | l;
117 }
118 
119 /**
120  * kho_radix_decode_key - Decodes a radix key back into a physical address and order.
121  * @key: The unsigned long key to decode.
122  * @order: An output parameter, a pointer to an unsigned int where the decoded
123  *         page order will be stored.
124  *
125  * This function reverses the encoding performed by kho_radix_encode_key(),
126  * extracting the original physical address and page order from a given key.
127  *
128  * Return: The decoded physical address.
129  */
130 static phys_addr_t kho_radix_decode_key(unsigned long key, unsigned int *order)
131 {
132 	unsigned int order_bit = fls64(key);
133 	phys_addr_t phys;
134 
135 	/* order_bit is numbered starting at 1 from fls64 */
136 	*order = KHO_ORDER_0_LOG2 - order_bit + 1;
137 	/* The order is discarded by the shift */
138 	phys = key << (PAGE_SHIFT + *order);
139 
140 	return phys;
141 }
142 
143 static unsigned long kho_radix_get_bitmap_index(unsigned long key)
144 {
145 	return key % (1 << KHO_BITMAP_SIZE_LOG2);
146 }
147 
148 static unsigned long kho_radix_get_table_index(unsigned long key,
149 					       unsigned int level)
150 {
151 	int s;
152 
153 	s = ((level - 1) * KHO_TABLE_SIZE_LOG2) + KHO_BITMAP_SIZE_LOG2;
154 	return (key >> s) % (1 << KHO_TABLE_SIZE_LOG2);
155 }
156 
157 /**
158  * kho_radix_add_page - Marks a page as preserved in the radix tree.
159  * @tree: The KHO radix tree.
160  * @pfn: The page frame number of the page to preserve.
161  * @order: The order of the page.
162  *
163  * This function traverses the radix tree based on the key derived from @pfn
164  * and @order. It sets the corresponding bit in the leaf bitmap to mark the
165  * page for preservation. If intermediate nodes do not exist along the path,
166  * they are allocated and added to the tree.
167  *
168  * Return: 0 on success, or a negative error code on failure.
169  */
170 int kho_radix_add_page(struct kho_radix_tree *tree,
171 		       unsigned long pfn, unsigned int order)
172 {
173 	/* Newly allocated nodes for error cleanup */
174 	struct kho_radix_node *intermediate_nodes[KHO_TREE_MAX_DEPTH] = { 0 };
175 	unsigned long key = kho_radix_encode_key(PFN_PHYS(pfn), order);
176 	struct kho_radix_node *anchor_node = NULL;
177 	struct kho_radix_node *node = tree->root;
178 	struct kho_radix_node *new_node;
179 	unsigned int i, idx, anchor_idx;
180 	struct kho_radix_leaf *leaf;
181 	int err = 0;
182 
183 	if (WARN_ON_ONCE(!tree->root))
184 		return -EINVAL;
185 
186 	might_sleep();
187 
188 	guard(mutex)(&tree->lock);
189 
190 	/* Go from high levels to low levels */
191 	for (i = KHO_TREE_MAX_DEPTH - 1; i > 0; i--) {
192 		idx = kho_radix_get_table_index(key, i);
193 
194 		if (node->table[idx]) {
195 			node = phys_to_virt(node->table[idx]);
196 			continue;
197 		}
198 
199 		/* Next node is empty, create a new node for it */
200 		new_node = (struct kho_radix_node *)get_zeroed_page(GFP_KERNEL);
201 		if (!new_node) {
202 			err = -ENOMEM;
203 			goto err_free_nodes;
204 		}
205 
206 		node->table[idx] = virt_to_phys(new_node);
207 
208 		/*
209 		 * Capture the node where the new branch starts for cleanup
210 		 * if allocation fails.
211 		 */
212 		if (!anchor_node) {
213 			anchor_node = node;
214 			anchor_idx = idx;
215 		}
216 		intermediate_nodes[i] = new_node;
217 
218 		node = new_node;
219 	}
220 
221 	/* Handle the leaf level bitmap (level 0) */
222 	idx = kho_radix_get_bitmap_index(key);
223 	leaf = (struct kho_radix_leaf *)node;
224 	__set_bit(idx, leaf->bitmap);
225 
226 	return 0;
227 
228 err_free_nodes:
229 	for (i = KHO_TREE_MAX_DEPTH - 1; i > 0; i--) {
230 		if (intermediate_nodes[i])
231 			free_page((unsigned long)intermediate_nodes[i]);
232 	}
233 	if (anchor_node)
234 		anchor_node->table[anchor_idx] = 0;
235 
236 	return err;
237 }
238 EXPORT_SYMBOL_GPL(kho_radix_add_page);
239 
240 /**
241  * kho_radix_del_page - Removes a page's preservation status from the radix tree.
242  * @tree: The KHO radix tree.
243  * @pfn: The page frame number of the page to unpreserve.
244  * @order: The order of the page.
245  *
246  * This function traverses the radix tree and clears the bit corresponding to
247  * the page, effectively removing its "preserved" status. It does not free
248  * the tree's intermediate nodes, even if they become empty.
249  */
250 void kho_radix_del_page(struct kho_radix_tree *tree, unsigned long pfn,
251 			unsigned int order)
252 {
253 	unsigned long key = kho_radix_encode_key(PFN_PHYS(pfn), order);
254 	struct kho_radix_node *node = tree->root;
255 	struct kho_radix_leaf *leaf;
256 	unsigned int i, idx;
257 
258 	if (WARN_ON_ONCE(!tree->root))
259 		return;
260 
261 	might_sleep();
262 
263 	guard(mutex)(&tree->lock);
264 
265 	/* Go from high levels to low levels */
266 	for (i = KHO_TREE_MAX_DEPTH - 1; i > 0; i--) {
267 		idx = kho_radix_get_table_index(key, i);
268 
269 		/*
270 		 * Attempting to delete a page that has not been preserved,
271 		 * return with a warning.
272 		 */
273 		if (WARN_ON(!node->table[idx]))
274 			return;
275 
276 		node = phys_to_virt(node->table[idx]);
277 	}
278 
279 	/* Handle the leaf level bitmap (level 0) */
280 	leaf = (struct kho_radix_leaf *)node;
281 	idx = kho_radix_get_bitmap_index(key);
282 	__clear_bit(idx, leaf->bitmap);
283 }
284 EXPORT_SYMBOL_GPL(kho_radix_del_page);
285 
286 static int kho_radix_walk_leaf(struct kho_radix_leaf *leaf,
287 			       unsigned long key,
288 			       kho_radix_tree_walk_callback_t cb)
289 {
290 	unsigned long *bitmap = (unsigned long *)leaf;
291 	unsigned int order;
292 	phys_addr_t phys;
293 	unsigned int i;
294 	int err;
295 
296 	for_each_set_bit(i, bitmap, PAGE_SIZE * BITS_PER_BYTE) {
297 		phys = kho_radix_decode_key(key | i, &order);
298 		err = cb(phys, order);
299 		if (err)
300 			return err;
301 	}
302 
303 	return 0;
304 }
305 
306 static int __kho_radix_walk_tree(struct kho_radix_node *root,
307 				 unsigned int level, unsigned long start,
308 				 kho_radix_tree_walk_callback_t cb)
309 {
310 	struct kho_radix_node *node;
311 	struct kho_radix_leaf *leaf;
312 	unsigned long key, i;
313 	unsigned int shift;
314 	int err;
315 
316 	for (i = 0; i < PAGE_SIZE / sizeof(phys_addr_t); i++) {
317 		if (!root->table[i])
318 			continue;
319 
320 		shift = ((level - 1) * KHO_TABLE_SIZE_LOG2) +
321 			KHO_BITMAP_SIZE_LOG2;
322 		key = start | (i << shift);
323 
324 		node = phys_to_virt(root->table[i]);
325 
326 		if (level == 1) {
327 			/*
328 			 * we are at level 1,
329 			 * node is pointing to the level 0 bitmap.
330 			 */
331 			leaf = (struct kho_radix_leaf *)node;
332 			err = kho_radix_walk_leaf(leaf, key, cb);
333 		} else {
334 			err  = __kho_radix_walk_tree(node, level - 1,
335 						     key, cb);
336 		}
337 
338 		if (err)
339 			return err;
340 	}
341 
342 	return 0;
343 }
344 
345 /**
346  * kho_radix_walk_tree - Traverses the radix tree and calls a callback for each preserved page.
347  * @tree: A pointer to the KHO radix tree to walk.
348  * @cb: A callback function of type kho_radix_tree_walk_callback_t that will be
349  *      invoked for each preserved page found in the tree. The callback receives
350  *      the physical address and order of the preserved page.
351  *
352  * This function walks the radix tree, searching from the specified top level
353  * down to the lowest level (level 0). For each preserved page found, it invokes
354  * the provided callback, passing the page's physical address and order.
355  *
356  * Return: 0 if the walk completed the specified tree, or the non-zero return
357  *         value from the callback that stopped the walk.
358  */
359 int kho_radix_walk_tree(struct kho_radix_tree *tree,
360 			kho_radix_tree_walk_callback_t cb)
361 {
362 	if (WARN_ON_ONCE(!tree->root))
363 		return -EINVAL;
364 
365 	guard(mutex)(&tree->lock);
366 
367 	return __kho_radix_walk_tree(tree->root, KHO_TREE_MAX_DEPTH - 1, 0, cb);
368 }
369 EXPORT_SYMBOL_GPL(kho_radix_walk_tree);
370 
371 /* For physically contiguous 0-order pages. */
372 static void kho_init_pages(struct page *page, unsigned long nr_pages)
373 {
374 	for (unsigned long i = 0; i < nr_pages; i++) {
375 		set_page_count(page + i, 1);
376 		/* Clear each page's codetag to avoid accounting mismatch. */
377 		clear_page_tag_ref(page + i);
378 	}
379 }
380 
381 static void kho_init_folio(struct page *page, unsigned int order)
382 {
383 	unsigned long nr_pages = (1 << order);
384 
385 	/* Head page gets refcount of 1. */
386 	set_page_count(page, 1);
387 	/* Clear head page's codetag to avoid accounting mismatch. */
388 	clear_page_tag_ref(page);
389 
390 	/* For higher order folios, tail pages get a page count of zero. */
391 	for (unsigned long i = 1; i < nr_pages; i++)
392 		set_page_count(page + i, 0);
393 
394 	if (order > 0)
395 		prep_compound_page(page, order);
396 }
397 
398 static struct page *kho_restore_page(phys_addr_t phys, bool is_folio)
399 {
400 	struct page *page = pfn_to_online_page(PHYS_PFN(phys));
401 	unsigned long nr_pages;
402 	union kho_page_info info;
403 
404 	if (!page)
405 		return NULL;
406 
407 	info.page_private = page->private;
408 	/*
409 	 * deserialize_bitmap() only sets the magic on the head page. This magic
410 	 * check also implicitly makes sure phys is order-aligned since for
411 	 * non-order-aligned phys addresses, magic will never be set.
412 	 */
413 	if (WARN_ON_ONCE(info.magic != KHO_PAGE_MAGIC))
414 		return NULL;
415 	nr_pages = (1 << info.order);
416 
417 	/* Clear private to make sure later restores on this page error out. */
418 	page->private = 0;
419 
420 	if (is_folio)
421 		kho_init_folio(page, info.order);
422 	else
423 		kho_init_pages(page, nr_pages);
424 
425 	adjust_managed_page_count(page, nr_pages);
426 	return page;
427 }
428 
429 /**
430  * kho_restore_folio - recreates the folio from the preserved memory.
431  * @phys: physical address of the folio.
432  *
433  * Return: pointer to the struct folio on success, NULL on failure.
434  */
435 struct folio *kho_restore_folio(phys_addr_t phys)
436 {
437 	struct page *page = kho_restore_page(phys, true);
438 
439 	return page ? page_folio(page) : NULL;
440 }
441 EXPORT_SYMBOL_GPL(kho_restore_folio);
442 
443 /**
444  * kho_restore_pages - restore list of contiguous order 0 pages.
445  * @phys: physical address of the first page.
446  * @nr_pages: number of pages.
447  *
448  * Restore a contiguous list of order 0 pages that was preserved with
449  * kho_preserve_pages().
450  *
451  * Return: the first page on success, NULL on failure.
452  */
453 struct page *kho_restore_pages(phys_addr_t phys, unsigned long nr_pages)
454 {
455 	const unsigned long start_pfn = PHYS_PFN(phys);
456 	const unsigned long end_pfn = start_pfn + nr_pages;
457 	unsigned long pfn = start_pfn;
458 
459 	while (pfn < end_pfn) {
460 		const unsigned int order =
461 			min(count_trailing_zeros(pfn), ilog2(end_pfn - pfn));
462 		struct page *page = kho_restore_page(PFN_PHYS(pfn), false);
463 
464 		if (!page)
465 			return NULL;
466 		pfn += 1 << order;
467 	}
468 
469 	return pfn_to_page(start_pfn);
470 }
471 EXPORT_SYMBOL_GPL(kho_restore_pages);
472 
473 /*
474  * With CONFIG_DEFERRED_STRUCT_PAGE_INIT, struct pages in higher memory regions
475  * may not be initialized yet at the time KHO deserializes preserved memory.
476  * KHO uses the struct page to store metadata and a later initialization would
477  * overwrite it.
478  * Ensure all the struct pages in the preservation are
479  * initialized. kho_preserved_memory_reserve() marks the reservation as noinit
480  * to make sure they don't get re-initialized later.
481  */
482 static struct page *__init kho_get_preserved_page(phys_addr_t phys,
483 						  unsigned int order)
484 {
485 	unsigned long pfn = PHYS_PFN(phys);
486 	int nid;
487 
488 	if (!IS_ENABLED(CONFIG_DEFERRED_STRUCT_PAGE_INIT))
489 		return pfn_to_page(pfn);
490 
491 	nid = early_pfn_to_nid(pfn);
492 	for (unsigned long i = 0; i < (1UL << order); i++)
493 		init_deferred_page(pfn + i, nid);
494 
495 	return pfn_to_page(pfn);
496 }
497 
498 static int __init kho_preserved_memory_reserve(phys_addr_t phys,
499 					       unsigned int order)
500 {
501 	union kho_page_info info;
502 	struct page *page;
503 	u64 sz;
504 
505 	sz = 1UL << (order + PAGE_SHIFT);
506 	page = kho_get_preserved_page(phys, order);
507 
508 	/* Reserve the memory preserved in KHO in memblock */
509 	memblock_reserve(phys, sz);
510 	memblock_reserved_mark_noinit(phys, sz);
511 	info.magic = KHO_PAGE_MAGIC;
512 	info.order = order;
513 	page->private = info.page_private;
514 
515 	return 0;
516 }
517 
518 /* Returns physical address of the preserved memory map from FDT */
519 static phys_addr_t __init kho_get_mem_map_phys(const void *fdt)
520 {
521 	const void *mem_ptr;
522 	int len;
523 
524 	mem_ptr = fdt_getprop(fdt, 0, KHO_FDT_MEMORY_MAP_PROP_NAME, &len);
525 	if (!mem_ptr || len != sizeof(u64)) {
526 		pr_err("failed to get preserved memory map\n");
527 		return 0;
528 	}
529 
530 	return get_unaligned((const u64 *)mem_ptr);
531 }
532 
533 /*
534  * With KHO enabled, memory can become fragmented because KHO regions may
535  * be anywhere in physical address space. The scratch regions give us a
536  * safe zones that we will never see KHO allocations from. This is where we
537  * can later safely load our new kexec images into and then use the scratch
538  * area for early allocations that happen before page allocator is
539  * initialized.
540  */
541 struct kho_scratch *kho_scratch;
542 unsigned int kho_scratch_cnt;
543 
544 /*
545  * The scratch areas are scaled by default as percent of memory allocated from
546  * memblock. A user can override the scale with command line parameter:
547  *
548  * kho_scratch=N%
549  *
550  * It is also possible to explicitly define size for a lowmem, a global and
551  * per-node scratch areas:
552  *
553  * kho_scratch=l[KMG],n[KMG],m[KMG]
554  *
555  * The explicit size definition takes precedence over scale definition.
556  */
557 static unsigned int scratch_scale __initdata = 200;
558 static phys_addr_t scratch_size_global __initdata;
559 static phys_addr_t scratch_size_pernode __initdata;
560 static phys_addr_t scratch_size_lowmem __initdata;
561 
562 static int __init kho_parse_scratch_size(char *p)
563 {
564 	size_t len;
565 	unsigned long sizes[3];
566 	size_t total_size = 0;
567 	int i;
568 
569 	if (!p)
570 		return -EINVAL;
571 
572 	len = strlen(p);
573 	if (!len)
574 		return -EINVAL;
575 
576 	/* parse nn% */
577 	if (p[len - 1] == '%') {
578 		/* unsigned int max is 4,294,967,295, 10 chars */
579 		char s_scale[11] = {};
580 		int ret = 0;
581 
582 		if (len > ARRAY_SIZE(s_scale))
583 			return -EINVAL;
584 
585 		memcpy(s_scale, p, len - 1);
586 		ret = kstrtouint(s_scale, 10, &scratch_scale);
587 		if (!ret)
588 			pr_notice("scratch scale is %d%%\n", scratch_scale);
589 		return ret;
590 	}
591 
592 	/* parse ll[KMG],mm[KMG],nn[KMG] */
593 	for (i = 0; i < ARRAY_SIZE(sizes); i++) {
594 		char *endp = p;
595 
596 		if (i > 0) {
597 			if (*p != ',')
598 				return -EINVAL;
599 			p += 1;
600 		}
601 
602 		sizes[i] = memparse(p, &endp);
603 		if (endp == p)
604 			return -EINVAL;
605 		p = endp;
606 		total_size += sizes[i];
607 	}
608 
609 	if (!total_size)
610 		return -EINVAL;
611 
612 	/* The string should be fully consumed by now. */
613 	if (*p)
614 		return -EINVAL;
615 
616 	scratch_size_lowmem = sizes[0];
617 	scratch_size_global = sizes[1];
618 	scratch_size_pernode = sizes[2];
619 	scratch_scale = 0;
620 
621 	pr_notice("scratch areas: lowmem: %lluMiB global: %lluMiB pernode: %lldMiB\n",
622 		  (u64)(scratch_size_lowmem >> 20),
623 		  (u64)(scratch_size_global >> 20),
624 		  (u64)(scratch_size_pernode >> 20));
625 
626 	return 0;
627 }
628 early_param("kho_scratch", kho_parse_scratch_size);
629 
630 static void __init scratch_size_update(void)
631 {
632 	/*
633 	 * If fixed sizes are not provided via command line, calculate them
634 	 * now.
635 	 */
636 	if (scratch_scale) {
637 		phys_addr_t size;
638 
639 		size = memblock_reserved_kern_size(ARCH_LOW_ADDRESS_LIMIT,
640 						   NUMA_NO_NODE);
641 		size = size * scratch_scale / 100;
642 		scratch_size_lowmem = size;
643 
644 		size = memblock_reserved_kern_size(MEMBLOCK_ALLOC_ANYWHERE,
645 						   NUMA_NO_NODE);
646 		size = size * scratch_scale / 100 - scratch_size_lowmem;
647 		scratch_size_global = size;
648 	}
649 
650 	/*
651 	 * Scratch areas are released as MIGRATE_CMA. Round them up to the right
652 	 * size.
653 	 */
654 	scratch_size_lowmem = round_up(scratch_size_lowmem, SCRATCH_ALIGNMENT_BYTES);
655 	scratch_size_global = round_up(scratch_size_global, SCRATCH_ALIGNMENT_BYTES);
656 }
657 
658 static phys_addr_t __init scratch_size_node(int nid)
659 {
660 	phys_addr_t size;
661 
662 	if (scratch_scale) {
663 		size = memblock_reserved_kern_size(MEMBLOCK_ALLOC_ANYWHERE,
664 						   nid);
665 		size = size * scratch_scale / 100;
666 	} else {
667 		size = scratch_size_pernode;
668 	}
669 
670 	return round_up(size, SCRATCH_ALIGNMENT_BYTES);
671 }
672 
673 /**
674  * kho_reserve_scratch - Reserve a contiguous chunk of memory for kexec
675  *
676  * With KHO we can preserve arbitrary pages in the system. To ensure we still
677  * have a large contiguous region of memory when we search the physical address
678  * space for target memory, let's make sure we always have a large CMA region
679  * active. This CMA region will only be used for movable pages which are not a
680  * problem for us during KHO because we can just move them somewhere else.
681  */
682 static void __init kho_reserve_scratch(void)
683 {
684 	phys_addr_t addr, size;
685 	int nid, i = 0;
686 
687 	if (!kho_enable)
688 		return;
689 
690 	scratch_size_update();
691 
692 	/* FIXME: deal with node hot-plug/remove */
693 	kho_scratch_cnt = nodes_weight(node_states[N_MEMORY]) + 2;
694 	size = kho_scratch_cnt * sizeof(*kho_scratch);
695 	kho_scratch = memblock_alloc(size, PAGE_SIZE);
696 	if (!kho_scratch) {
697 		pr_err("Failed to reserve scratch array\n");
698 		goto err_disable_kho;
699 	}
700 
701 	/*
702 	 * reserve scratch area in low memory for lowmem allocations in the
703 	 * next kernel
704 	 */
705 	size = scratch_size_lowmem;
706 	addr = memblock_phys_alloc_range(size, SCRATCH_ALIGNMENT_BYTES, 0,
707 					 ARCH_LOW_ADDRESS_LIMIT);
708 	if (!addr) {
709 		pr_err("Failed to reserve lowmem scratch buffer\n");
710 		goto err_free_scratch_desc;
711 	}
712 
713 	kho_scratch[i].addr = addr;
714 	kho_scratch[i].size = size;
715 	i++;
716 
717 	/* reserve large contiguous area for allocations without nid */
718 	size = scratch_size_global;
719 	addr = memblock_phys_alloc(size, SCRATCH_ALIGNMENT_BYTES);
720 	if (!addr) {
721 		pr_err("Failed to reserve global scratch buffer\n");
722 		goto err_free_scratch_areas;
723 	}
724 
725 	kho_scratch[i].addr = addr;
726 	kho_scratch[i].size = size;
727 	i++;
728 
729 	/*
730 	 * Loop over nodes that have both memory and are online. Skip
731 	 * memoryless nodes, as we can not allocate scratch areas there.
732 	 */
733 	for_each_node_state(nid, N_MEMORY) {
734 		size = scratch_size_node(nid);
735 		addr = memblock_alloc_range_nid(size, SCRATCH_ALIGNMENT_BYTES,
736 						0, MEMBLOCK_ALLOC_ACCESSIBLE,
737 						nid, true);
738 		if (!addr) {
739 			pr_err("Failed to reserve nid %d scratch buffer\n", nid);
740 			goto err_free_scratch_areas;
741 		}
742 
743 		kho_scratch[i].addr = addr;
744 		kho_scratch[i].size = size;
745 		i++;
746 	}
747 
748 	return;
749 
750 err_free_scratch_areas:
751 	for (i--; i >= 0; i--)
752 		memblock_phys_free(kho_scratch[i].addr, kho_scratch[i].size);
753 err_free_scratch_desc:
754 	memblock_free(kho_scratch, kho_scratch_cnt * sizeof(*kho_scratch));
755 err_disable_kho:
756 	pr_warn("Failed to reserve scratch area, disabling kexec handover\n");
757 	kho_enable = false;
758 }
759 
760 /**
761  * kho_add_subtree - record the physical address of a sub blob in KHO root tree.
762  * @name: name of the sub tree.
763  * @blob: the sub tree blob.
764  * @size: size of the blob in bytes.
765  *
766  * Creates a new child node named @name in KHO root FDT and records
767  * the physical address of @blob. The pages of @blob must also be preserved
768  * by KHO for the new kernel to retrieve it after kexec.
769  *
770  * A debugfs blob entry is also created at
771  * ``/sys/kernel/debug/kho/out/sub_fdts/@name`` when kernel is configured with
772  * CONFIG_KEXEC_HANDOVER_DEBUGFS
773  *
774  * Return: 0 on success, error code on failure
775  */
776 int kho_add_subtree(const char *name, void *blob, size_t size)
777 {
778 	phys_addr_t phys = virt_to_phys(blob);
779 	void *root_fdt = kho_out.fdt;
780 	u64 size_u64 = size;
781 	int err = -ENOMEM;
782 	int off, fdt_err;
783 
784 	guard(mutex)(&kho_out.lock);
785 
786 	fdt_err = fdt_open_into(root_fdt, root_fdt, PAGE_SIZE);
787 	if (fdt_err < 0)
788 		return err;
789 
790 	off = fdt_add_subnode(root_fdt, 0, name);
791 	if (off < 0) {
792 		if (off == -FDT_ERR_EXISTS)
793 			err = -EEXIST;
794 		goto out_pack;
795 	}
796 
797 	fdt_err = fdt_setprop(root_fdt, off, KHO_SUB_TREE_PROP_NAME,
798 			      &phys, sizeof(phys));
799 	if (fdt_err < 0)
800 		goto out_del_node;
801 
802 	fdt_err = fdt_setprop(root_fdt, off, KHO_SUB_TREE_SIZE_PROP_NAME,
803 			      &size_u64, sizeof(size_u64));
804 	if (fdt_err < 0)
805 		goto out_del_node;
806 
807 	WARN_ON_ONCE(kho_debugfs_blob_add(&kho_out.dbg, name, blob,
808 					  size, false));
809 
810 	err = 0;
811 	goto out_pack;
812 
813 out_del_node:
814 	fdt_del_node(root_fdt, off);
815 out_pack:
816 	fdt_pack(root_fdt);
817 
818 	return err;
819 }
820 EXPORT_SYMBOL_GPL(kho_add_subtree);
821 
822 void kho_remove_subtree(void *blob)
823 {
824 	phys_addr_t target_phys = virt_to_phys(blob);
825 	void *root_fdt = kho_out.fdt;
826 	int off;
827 	int err;
828 
829 	guard(mutex)(&kho_out.lock);
830 
831 	err = fdt_open_into(root_fdt, root_fdt, PAGE_SIZE);
832 	if (err < 0)
833 		return;
834 
835 	for (off = fdt_first_subnode(root_fdt, 0); off >= 0;
836 	     off = fdt_next_subnode(root_fdt, off)) {
837 		const u64 *val;
838 		int len;
839 
840 		val = fdt_getprop(root_fdt, off, KHO_SUB_TREE_PROP_NAME, &len);
841 		if (!val || len != sizeof(phys_addr_t))
842 			continue;
843 
844 		if ((phys_addr_t)*val == target_phys) {
845 			fdt_del_node(root_fdt, off);
846 			kho_debugfs_blob_remove(&kho_out.dbg, blob);
847 			break;
848 		}
849 	}
850 
851 	fdt_pack(root_fdt);
852 }
853 EXPORT_SYMBOL_GPL(kho_remove_subtree);
854 
855 /**
856  * kho_preserve_folio - preserve a folio across kexec.
857  * @folio: folio to preserve.
858  *
859  * Instructs KHO to preserve the whole folio across kexec. The order
860  * will be preserved as well.
861  *
862  * Return: 0 on success, error code on failure
863  */
864 int kho_preserve_folio(struct folio *folio)
865 {
866 	struct kho_radix_tree *tree = &kho_out.radix_tree;
867 	const unsigned long pfn = folio_pfn(folio);
868 	const unsigned int order = folio_order(folio);
869 
870 	if (WARN_ON(kho_scratch_overlap(pfn << PAGE_SHIFT, PAGE_SIZE << order)))
871 		return -EINVAL;
872 
873 	return kho_radix_add_page(tree, pfn, order);
874 }
875 EXPORT_SYMBOL_GPL(kho_preserve_folio);
876 
877 /**
878  * kho_unpreserve_folio - unpreserve a folio.
879  * @folio: folio to unpreserve.
880  *
881  * Instructs KHO to unpreserve a folio that was preserved by
882  * kho_preserve_folio() before. The provided @folio (pfn and order)
883  * must exactly match a previously preserved folio.
884  */
885 void kho_unpreserve_folio(struct folio *folio)
886 {
887 	struct kho_radix_tree *tree = &kho_out.radix_tree;
888 	const unsigned long pfn = folio_pfn(folio);
889 	const unsigned int order = folio_order(folio);
890 
891 	kho_radix_del_page(tree, pfn, order);
892 }
893 EXPORT_SYMBOL_GPL(kho_unpreserve_folio);
894 
895 static unsigned int __kho_preserve_pages_order(unsigned long start_pfn,
896 					       unsigned long end_pfn)
897 {
898 	unsigned int order = min(count_trailing_zeros(start_pfn),
899 				 ilog2(end_pfn - start_pfn));
900 
901 	/*
902 	 * Make sure all the pages in a single preservation are in the same NUMA
903 	 * node. The restore machinery can not cope with a preservation spanning
904 	 * multiple NUMA nodes.
905 	 */
906 	while (pfn_to_nid(start_pfn) != pfn_to_nid(start_pfn + (1UL << order) - 1))
907 		order--;
908 
909 	return order;
910 }
911 
912 static void __kho_unpreserve(struct kho_radix_tree *tree,
913 			     unsigned long pfn, unsigned long end_pfn)
914 {
915 	unsigned int order;
916 
917 	while (pfn < end_pfn) {
918 		order = __kho_preserve_pages_order(pfn, end_pfn);
919 
920 		kho_radix_del_page(tree, pfn, order);
921 
922 		pfn += 1 << order;
923 	}
924 }
925 
926 /**
927  * kho_preserve_pages - preserve contiguous pages across kexec
928  * @page: first page in the list.
929  * @nr_pages: number of pages.
930  *
931  * Preserve a contiguous list of order 0 pages. Must be restored using
932  * kho_restore_pages() to ensure the pages are restored properly as order 0.
933  *
934  * Return: 0 on success, error code on failure
935  */
936 int kho_preserve_pages(struct page *page, unsigned long nr_pages)
937 {
938 	struct kho_radix_tree *tree = &kho_out.radix_tree;
939 	const unsigned long start_pfn = page_to_pfn(page);
940 	const unsigned long end_pfn = start_pfn + nr_pages;
941 	unsigned long pfn = start_pfn;
942 	unsigned long failed_pfn = 0;
943 	int err = 0;
944 
945 	if (WARN_ON(kho_scratch_overlap(start_pfn << PAGE_SHIFT,
946 					nr_pages << PAGE_SHIFT))) {
947 		return -EINVAL;
948 	}
949 
950 	while (pfn < end_pfn) {
951 		unsigned int order = __kho_preserve_pages_order(pfn, end_pfn);
952 
953 		err = kho_radix_add_page(tree, pfn, order);
954 		if (err) {
955 			failed_pfn = pfn;
956 			break;
957 		}
958 
959 		pfn += 1 << order;
960 	}
961 
962 	if (err)
963 		__kho_unpreserve(tree, start_pfn, failed_pfn);
964 
965 	return err;
966 }
967 EXPORT_SYMBOL_GPL(kho_preserve_pages);
968 
969 /**
970  * kho_unpreserve_pages - unpreserve contiguous pages.
971  * @page: first page in the list.
972  * @nr_pages: number of pages.
973  *
974  * Instructs KHO to unpreserve @nr_pages contiguous pages starting from @page.
975  * This must be called with the same @page and @nr_pages as the corresponding
976  * kho_preserve_pages() call. Unpreserving arbitrary sub-ranges of larger
977  * preserved blocks is not supported.
978  */
979 void kho_unpreserve_pages(struct page *page, unsigned long nr_pages)
980 {
981 	struct kho_radix_tree *tree = &kho_out.radix_tree;
982 	const unsigned long start_pfn = page_to_pfn(page);
983 	const unsigned long end_pfn = start_pfn + nr_pages;
984 
985 	__kho_unpreserve(tree, start_pfn, end_pfn);
986 }
987 EXPORT_SYMBOL_GPL(kho_unpreserve_pages);
988 
989 /* vmalloc flags KHO supports */
990 #define KHO_VMALLOC_SUPPORTED_FLAGS	(VM_ALLOC | VM_ALLOW_HUGE_VMAP)
991 
992 /* KHO internal flags for vmalloc preservations */
993 #define KHO_VMALLOC_ALLOC	0x0001
994 #define KHO_VMALLOC_HUGE_VMAP	0x0002
995 
996 static unsigned short vmalloc_flags_to_kho(unsigned int vm_flags)
997 {
998 	unsigned short kho_flags = 0;
999 
1000 	if (vm_flags & VM_ALLOC)
1001 		kho_flags |= KHO_VMALLOC_ALLOC;
1002 	if (vm_flags & VM_ALLOW_HUGE_VMAP)
1003 		kho_flags |= KHO_VMALLOC_HUGE_VMAP;
1004 
1005 	return kho_flags;
1006 }
1007 
1008 static unsigned int kho_flags_to_vmalloc(unsigned short kho_flags)
1009 {
1010 	unsigned int vm_flags = 0;
1011 
1012 	if (kho_flags & KHO_VMALLOC_ALLOC)
1013 		vm_flags |= VM_ALLOC;
1014 	if (kho_flags & KHO_VMALLOC_HUGE_VMAP)
1015 		vm_flags |= VM_ALLOW_HUGE_VMAP;
1016 
1017 	return vm_flags;
1018 }
1019 
1020 static struct kho_vmalloc_chunk *new_vmalloc_chunk(struct kho_vmalloc_chunk *cur)
1021 {
1022 	struct kho_vmalloc_chunk *chunk;
1023 	int err;
1024 
1025 	chunk = (struct kho_vmalloc_chunk *)get_zeroed_page(GFP_KERNEL);
1026 	if (!chunk)
1027 		return NULL;
1028 
1029 	err = kho_preserve_pages(virt_to_page(chunk), 1);
1030 	if (err)
1031 		goto err_free;
1032 	if (cur)
1033 		KHOSER_STORE_PTR(cur->hdr.next, chunk);
1034 	return chunk;
1035 
1036 err_free:
1037 	free_page((unsigned long)chunk);
1038 	return NULL;
1039 }
1040 
1041 static void kho_vmalloc_unpreserve_chunk(struct kho_vmalloc_chunk *chunk,
1042 					 unsigned short order)
1043 {
1044 	struct kho_radix_tree *tree = &kho_out.radix_tree;
1045 	unsigned long pfn = PHYS_PFN(virt_to_phys(chunk));
1046 
1047 	__kho_unpreserve(tree, pfn, pfn + 1);
1048 
1049 	for (int i = 0; i < ARRAY_SIZE(chunk->phys) && chunk->phys[i]; i++) {
1050 		pfn = PHYS_PFN(chunk->phys[i]);
1051 		__kho_unpreserve(tree, pfn, pfn + (1 << order));
1052 	}
1053 }
1054 
1055 /**
1056  * kho_preserve_vmalloc - preserve memory allocated with vmalloc() across kexec
1057  * @ptr: pointer to the area in vmalloc address space
1058  * @preservation: placeholder for preservation metadata
1059  *
1060  * Instructs KHO to preserve the area in vmalloc address space at @ptr. The
1061  * physical pages mapped at @ptr will be preserved and on successful return
1062  * @preservation will hold the physical address of a structure that describes
1063  * the preservation.
1064  *
1065  * NOTE: The memory allocated with vmalloc_node() variants cannot be reliably
1066  * restored on the same node
1067  *
1068  * Return: 0 on success, error code on failure
1069  */
1070 int kho_preserve_vmalloc(void *ptr, struct kho_vmalloc *preservation)
1071 {
1072 	struct kho_vmalloc_chunk *chunk;
1073 	struct vm_struct *vm = find_vm_area(ptr);
1074 	unsigned int order, flags, nr_contig_pages;
1075 	unsigned int idx = 0;
1076 	int err;
1077 
1078 	if (!vm)
1079 		return -EINVAL;
1080 
1081 	if (vm->flags & ~KHO_VMALLOC_SUPPORTED_FLAGS)
1082 		return -EOPNOTSUPP;
1083 
1084 	flags = vmalloc_flags_to_kho(vm->flags);
1085 	order = get_vm_area_page_order(vm);
1086 
1087 	chunk = new_vmalloc_chunk(NULL);
1088 	if (!chunk)
1089 		return -ENOMEM;
1090 	KHOSER_STORE_PTR(preservation->first, chunk);
1091 
1092 	nr_contig_pages = (1 << order);
1093 	for (int i = 0; i < vm->nr_pages; i += nr_contig_pages) {
1094 		phys_addr_t phys = page_to_phys(vm->pages[i]);
1095 
1096 		err = kho_preserve_pages(vm->pages[i], nr_contig_pages);
1097 		if (err)
1098 			goto err_free;
1099 
1100 		chunk->phys[idx++] = phys;
1101 		if (idx == ARRAY_SIZE(chunk->phys)) {
1102 			chunk = new_vmalloc_chunk(chunk);
1103 			if (!chunk) {
1104 				err = -ENOMEM;
1105 				goto err_free;
1106 			}
1107 			idx = 0;
1108 		}
1109 	}
1110 
1111 	preservation->total_pages = vm->nr_pages;
1112 	preservation->flags = flags;
1113 	preservation->order = order;
1114 
1115 	return 0;
1116 
1117 err_free:
1118 	kho_unpreserve_vmalloc(preservation);
1119 	return err;
1120 }
1121 EXPORT_SYMBOL_GPL(kho_preserve_vmalloc);
1122 
1123 /**
1124  * kho_unpreserve_vmalloc - unpreserve memory allocated with vmalloc()
1125  * @preservation: preservation metadata returned by kho_preserve_vmalloc()
1126  *
1127  * Instructs KHO to unpreserve the area in vmalloc address space that was
1128  * previously preserved with kho_preserve_vmalloc().
1129  */
1130 void kho_unpreserve_vmalloc(struct kho_vmalloc *preservation)
1131 {
1132 	struct kho_vmalloc_chunk *chunk = KHOSER_LOAD_PTR(preservation->first);
1133 
1134 	while (chunk) {
1135 		struct kho_vmalloc_chunk *tmp = chunk;
1136 
1137 		kho_vmalloc_unpreserve_chunk(chunk, preservation->order);
1138 
1139 		chunk = KHOSER_LOAD_PTR(chunk->hdr.next);
1140 		free_page((unsigned long)tmp);
1141 	}
1142 }
1143 EXPORT_SYMBOL_GPL(kho_unpreserve_vmalloc);
1144 
1145 /**
1146  * kho_restore_vmalloc - recreates and populates an area in vmalloc address
1147  * space from the preserved memory.
1148  * @preservation: preservation metadata.
1149  *
1150  * Recreates an area in vmalloc address space and populates it with memory that
1151  * was preserved using kho_preserve_vmalloc().
1152  *
1153  * Return: pointer to the area in the vmalloc address space, NULL on failure.
1154  */
1155 void *kho_restore_vmalloc(const struct kho_vmalloc *preservation)
1156 {
1157 	struct kho_vmalloc_chunk *chunk = KHOSER_LOAD_PTR(preservation->first);
1158 	kasan_vmalloc_flags_t kasan_flags = KASAN_VMALLOC_PROT_NORMAL;
1159 	unsigned int align, order, shift, vm_flags;
1160 	unsigned long total_pages, contig_pages;
1161 	unsigned long addr, size;
1162 	struct vm_struct *area;
1163 	struct page **pages;
1164 	unsigned int idx = 0;
1165 	int err;
1166 
1167 	vm_flags = kho_flags_to_vmalloc(preservation->flags);
1168 	if (vm_flags & ~KHO_VMALLOC_SUPPORTED_FLAGS)
1169 		return NULL;
1170 
1171 	total_pages = preservation->total_pages;
1172 	pages = kvmalloc_objs(*pages, total_pages);
1173 	if (!pages)
1174 		return NULL;
1175 	order = preservation->order;
1176 	contig_pages = (1 << order);
1177 	shift = PAGE_SHIFT + order;
1178 	align = 1 << shift;
1179 
1180 	while (chunk) {
1181 		struct page *page;
1182 
1183 		for (int i = 0; i < ARRAY_SIZE(chunk->phys) && chunk->phys[i]; i++) {
1184 			phys_addr_t phys = chunk->phys[i];
1185 
1186 			if (idx + contig_pages > total_pages)
1187 				goto err_free_pages_array;
1188 
1189 			page = kho_restore_pages(phys, contig_pages);
1190 			if (!page)
1191 				goto err_free_pages_array;
1192 
1193 			for (int j = 0; j < contig_pages; j++)
1194 				pages[idx++] = page + j;
1195 
1196 			phys += contig_pages * PAGE_SIZE;
1197 		}
1198 
1199 		page = kho_restore_pages(virt_to_phys(chunk), 1);
1200 		if (!page)
1201 			goto err_free_pages_array;
1202 		chunk = KHOSER_LOAD_PTR(chunk->hdr.next);
1203 		__free_page(page);
1204 	}
1205 
1206 	if (idx != total_pages)
1207 		goto err_free_pages_array;
1208 
1209 	area = __get_vm_area_node(total_pages * PAGE_SIZE, align, shift,
1210 				  vm_flags | VM_UNINITIALIZED,
1211 				  VMALLOC_START, VMALLOC_END,
1212 				  NUMA_NO_NODE, GFP_KERNEL,
1213 				  __builtin_return_address(0));
1214 	if (!area)
1215 		goto err_free_pages_array;
1216 
1217 	addr = (unsigned long)area->addr;
1218 	size = get_vm_area_size(area);
1219 	err = vmap_pages_range(addr, addr + size, PAGE_KERNEL, pages, shift);
1220 	if (err)
1221 		goto err_free_vm_area;
1222 
1223 	area->nr_pages = total_pages;
1224 	area->pages = pages;
1225 
1226 	if (vm_flags & VM_ALLOC)
1227 		kasan_flags |= KASAN_VMALLOC_VM_ALLOC;
1228 
1229 	area->addr = kasan_unpoison_vmalloc(area->addr, total_pages * PAGE_SIZE,
1230 					    kasan_flags);
1231 	clear_vm_uninitialized_flag(area);
1232 
1233 	return area->addr;
1234 
1235 err_free_vm_area:
1236 	free_vm_area(area);
1237 err_free_pages_array:
1238 	kvfree(pages);
1239 	return NULL;
1240 }
1241 EXPORT_SYMBOL_GPL(kho_restore_vmalloc);
1242 
1243 /**
1244  * kho_alloc_preserve - Allocate, zero, and preserve memory.
1245  * @size: The number of bytes to allocate.
1246  *
1247  * Allocates a physically contiguous block of zeroed pages that is large
1248  * enough to hold @size bytes. The allocated memory is then registered with
1249  * KHO for preservation across a kexec.
1250  *
1251  * Note: The actual allocated size will be rounded up to the nearest
1252  * power-of-two page boundary.
1253  *
1254  * @return A virtual pointer to the allocated and preserved memory on success,
1255  * or an ERR_PTR() encoded error on failure.
1256  */
1257 void *kho_alloc_preserve(size_t size)
1258 {
1259 	struct folio *folio;
1260 	int order, ret;
1261 
1262 	if (!size)
1263 		return ERR_PTR(-EINVAL);
1264 
1265 	order = get_order(size);
1266 	if (order > MAX_PAGE_ORDER)
1267 		return ERR_PTR(-E2BIG);
1268 
1269 	folio = folio_alloc(GFP_KERNEL | __GFP_ZERO, order);
1270 	if (!folio)
1271 		return ERR_PTR(-ENOMEM);
1272 
1273 	ret = kho_preserve_folio(folio);
1274 	if (ret) {
1275 		folio_put(folio);
1276 		return ERR_PTR(ret);
1277 	}
1278 
1279 	return folio_address(folio);
1280 }
1281 EXPORT_SYMBOL_GPL(kho_alloc_preserve);
1282 
1283 /**
1284  * kho_unpreserve_free - Unpreserve and free memory.
1285  * @mem:  Pointer to the memory allocated by kho_alloc_preserve().
1286  *
1287  * Unregisters the memory from KHO preservation and frees the underlying
1288  * pages back to the system. This function should be called to clean up
1289  * memory allocated with kho_alloc_preserve().
1290  */
1291 void kho_unpreserve_free(void *mem)
1292 {
1293 	struct folio *folio;
1294 
1295 	if (!mem)
1296 		return;
1297 
1298 	folio = virt_to_folio(mem);
1299 	kho_unpreserve_folio(folio);
1300 	folio_put(folio);
1301 }
1302 EXPORT_SYMBOL_GPL(kho_unpreserve_free);
1303 
1304 /**
1305  * kho_restore_free - Restore and free memory after kexec.
1306  * @mem:  Pointer to the memory (in the new kernel's address space)
1307  * that was allocated by the old kernel.
1308  *
1309  * This function is intended to be called in the new kernel (post-kexec)
1310  * to take ownership of and free a memory region that was preserved by the
1311  * old kernel using kho_alloc_preserve().
1312  *
1313  * It first restores the pages from KHO (using their physical address)
1314  * and then frees the pages back to the new kernel's page allocator.
1315  */
1316 void kho_restore_free(void *mem)
1317 {
1318 	struct folio *folio;
1319 
1320 	if (!mem)
1321 		return;
1322 
1323 	folio = kho_restore_folio(__pa(mem));
1324 	if (!WARN_ON(!folio))
1325 		folio_put(folio);
1326 }
1327 EXPORT_SYMBOL_GPL(kho_restore_free);
1328 
1329 struct kho_in {
1330 	phys_addr_t fdt_phys;
1331 	phys_addr_t scratch_phys;
1332 	char previous_release[__NEW_UTS_LEN + 1];
1333 	u32 kexec_count;
1334 	struct kho_debugfs dbg;
1335 };
1336 
1337 static struct kho_in kho_in = {
1338 };
1339 
1340 static const void *kho_get_fdt(void)
1341 {
1342 	return kho_in.fdt_phys ? phys_to_virt(kho_in.fdt_phys) : NULL;
1343 }
1344 
1345 /**
1346  * is_kho_boot - check if current kernel was booted via KHO-enabled
1347  * kexec
1348  *
1349  * This function checks if the current kernel was loaded through a kexec
1350  * operation with KHO enabled, by verifying that a valid KHO FDT
1351  * was passed.
1352  *
1353  * Note: This function returns reliable results only after
1354  * kho_populate() has been called during early boot. Before that,
1355  * it may return false even if KHO data is present.
1356  *
1357  * Return: true if booted via KHO-enabled kexec, false otherwise
1358  */
1359 bool is_kho_boot(void)
1360 {
1361 	return !!kho_get_fdt();
1362 }
1363 EXPORT_SYMBOL_GPL(is_kho_boot);
1364 
1365 /**
1366  * kho_retrieve_subtree - retrieve a preserved sub blob by its name.
1367  * @name: the name of the sub blob passed to kho_add_subtree().
1368  * @phys: if found, the physical address of the sub blob is stored in @phys.
1369  * @size: if not NULL and found, the size of the sub blob is stored in @size.
1370  *
1371  * Retrieve a preserved sub blob named @name and store its physical
1372  * address in @phys and optionally its size in @size.
1373  *
1374  * Return: 0 on success, error code on failure
1375  */
1376 int kho_retrieve_subtree(const char *name, phys_addr_t *phys, size_t *size)
1377 {
1378 	const void *fdt = kho_get_fdt();
1379 	const u64 *val;
1380 	int offset, len;
1381 
1382 	if (!fdt)
1383 		return -ENOENT;
1384 
1385 	if (!phys)
1386 		return -EINVAL;
1387 
1388 	offset = fdt_subnode_offset(fdt, 0, name);
1389 	if (offset < 0)
1390 		return -ENOENT;
1391 
1392 	val = fdt_getprop(fdt, offset, KHO_SUB_TREE_PROP_NAME, &len);
1393 	if (!val || len != sizeof(*val))
1394 		return -EINVAL;
1395 
1396 	*phys = (phys_addr_t)*val;
1397 
1398 	val = fdt_getprop(fdt, offset, KHO_SUB_TREE_SIZE_PROP_NAME, &len);
1399 	if (!val || len != sizeof(*val)) {
1400 		pr_warn("broken KHO subnode '%s': missing or invalid blob-size property\n",
1401 			name);
1402 		return -EINVAL;
1403 	}
1404 
1405 	if (size)
1406 		*size = (size_t)*val;
1407 
1408 	return 0;
1409 }
1410 EXPORT_SYMBOL_GPL(kho_retrieve_subtree);
1411 
1412 static int __init kho_mem_retrieve(const void *fdt)
1413 {
1414 	struct kho_radix_tree tree;
1415 	const phys_addr_t *mem;
1416 	int len;
1417 
1418 	/* Retrieve the KHO radix tree from passed-in FDT. */
1419 	mem = fdt_getprop(fdt, 0, KHO_FDT_MEMORY_MAP_PROP_NAME, &len);
1420 
1421 	if (!mem || len != sizeof(*mem)) {
1422 		pr_err("failed to get preserved KHO memory tree\n");
1423 		return -ENOENT;
1424 	}
1425 
1426 	if (!*mem)
1427 		return -EINVAL;
1428 
1429 	tree.root = phys_to_virt(*mem);
1430 	mutex_init(&tree.lock);
1431 	return kho_radix_walk_tree(&tree, kho_preserved_memory_reserve);
1432 }
1433 
1434 static __init int kho_out_fdt_setup(void)
1435 {
1436 	struct kho_radix_tree *tree = &kho_out.radix_tree;
1437 	void *root = kho_out.fdt;
1438 	u64 preserved_mem_tree_pa;
1439 	int err;
1440 
1441 	err = fdt_create(root, PAGE_SIZE);
1442 	err |= fdt_finish_reservemap(root);
1443 	err |= fdt_begin_node(root, "");
1444 	err |= fdt_property_string(root, "compatible", KHO_FDT_COMPATIBLE);
1445 
1446 	preserved_mem_tree_pa = virt_to_phys(tree->root);
1447 
1448 	err |= fdt_property(root, KHO_FDT_MEMORY_MAP_PROP_NAME,
1449 			    &preserved_mem_tree_pa,
1450 			    sizeof(preserved_mem_tree_pa));
1451 
1452 	err |= fdt_end_node(root);
1453 	err |= fdt_finish(root);
1454 
1455 	return err;
1456 }
1457 
1458 static void __init kho_in_kexec_metadata(void)
1459 {
1460 	struct kho_kexec_metadata *metadata;
1461 	phys_addr_t metadata_phys;
1462 	size_t blob_size;
1463 	int err;
1464 
1465 	err = kho_retrieve_subtree(KHO_METADATA_NODE_NAME, &metadata_phys,
1466 				   &blob_size);
1467 	if (err)
1468 		/* This is fine, previous kernel didn't export metadata */
1469 		return;
1470 
1471 	/* Check that, at least, "version" is present */
1472 	if (blob_size < sizeof(u32)) {
1473 		pr_warn("kexec-metadata blob too small (%zu bytes)\n",
1474 			blob_size);
1475 		return;
1476 	}
1477 
1478 	metadata = phys_to_virt(metadata_phys);
1479 
1480 	if (metadata->version != KHO_KEXEC_METADATA_VERSION) {
1481 		pr_warn("kexec-metadata version %u not supported (expected %u)\n",
1482 			metadata->version, KHO_KEXEC_METADATA_VERSION);
1483 		return;
1484 	}
1485 
1486 	if (blob_size < sizeof(*metadata)) {
1487 		pr_warn("kexec-metadata blob too small for v%u (%zu < %zu)\n",
1488 			metadata->version, blob_size, sizeof(*metadata));
1489 		return;
1490 	}
1491 
1492 	/*
1493 	 * Copy data to the kernel structure that will persist during
1494 	 * kernel lifetime.
1495 	 */
1496 	kho_in.kexec_count = metadata->kexec_count;
1497 	strscpy(kho_in.previous_release, metadata->previous_release,
1498 		sizeof(kho_in.previous_release));
1499 
1500 	pr_info("exec from: %s (count %u)\n",
1501 		kho_in.previous_release, kho_in.kexec_count);
1502 }
1503 
1504 /*
1505  * Create kexec metadata to pass kernel version and boot count to the
1506  * next kernel. This keeps the core KHO ABI minimal and allows the
1507  * metadata format to evolve independently.
1508  */
1509 static __init int kho_out_kexec_metadata(void)
1510 {
1511 	struct kho_kexec_metadata *metadata;
1512 	int err;
1513 
1514 	metadata = kho_alloc_preserve(sizeof(*metadata));
1515 	if (IS_ERR(metadata))
1516 		return PTR_ERR(metadata);
1517 
1518 	metadata->version = KHO_KEXEC_METADATA_VERSION;
1519 	strscpy(metadata->previous_release, init_uts_ns.name.release,
1520 		sizeof(metadata->previous_release));
1521 	/* kho_in.kexec_count is set to 0 on cold boot */
1522 	metadata->kexec_count = kho_in.kexec_count + 1;
1523 
1524 	err = kho_add_subtree(KHO_METADATA_NODE_NAME, metadata,
1525 			      sizeof(*metadata));
1526 	if (err)
1527 		kho_unpreserve_free(metadata);
1528 
1529 	return err;
1530 }
1531 
1532 static int __init kho_kexec_metadata_init(const void *fdt)
1533 {
1534 	int err;
1535 
1536 	if (fdt)
1537 		kho_in_kexec_metadata();
1538 
1539 	/* Populate kexec metadata for the possible next kexec */
1540 	err = kho_out_kexec_metadata();
1541 	if (err)
1542 		pr_warn("failed to initialize kexec-metadata subtree: %d\n",
1543 			err);
1544 
1545 	return err;
1546 }
1547 
1548 static __init int kho_init(void)
1549 {
1550 	struct kho_radix_tree *tree = &kho_out.radix_tree;
1551 	const void *fdt = kho_get_fdt();
1552 	int err = 0;
1553 
1554 	if (!kho_enable)
1555 		return 0;
1556 
1557 	tree->root = kzalloc(PAGE_SIZE, GFP_KERNEL);
1558 	if (!tree->root) {
1559 		err = -ENOMEM;
1560 		goto err_free_scratch;
1561 	}
1562 
1563 	kho_out.fdt = kho_alloc_preserve(PAGE_SIZE);
1564 	if (IS_ERR(kho_out.fdt)) {
1565 		err = PTR_ERR(kho_out.fdt);
1566 		goto err_free_kho_radix_tree_root;
1567 	}
1568 
1569 	err = kho_debugfs_init();
1570 	if (err)
1571 		goto err_free_fdt;
1572 
1573 	err = kho_out_debugfs_init(&kho_out.dbg);
1574 	if (err)
1575 		goto err_free_fdt;
1576 
1577 	err = kho_out_fdt_setup();
1578 	if (err)
1579 		goto err_free_fdt;
1580 
1581 	err = kho_kexec_metadata_init(fdt);
1582 	if (err)
1583 		goto err_free_fdt;
1584 
1585 	if (fdt) {
1586 		kho_in_debugfs_init(&kho_in.dbg, fdt);
1587 		return 0;
1588 	}
1589 
1590 	for (int i = 0; i < kho_scratch_cnt; i++) {
1591 		unsigned long base_pfn = PHYS_PFN(kho_scratch[i].addr);
1592 		unsigned long count = kho_scratch[i].size >> PAGE_SHIFT;
1593 		unsigned long pfn;
1594 
1595 		/*
1596 		 * When debug_pagealloc is enabled, __free_pages() clears the
1597 		 * corresponding PRESENT bit in the kernel page table.
1598 		 * Subsequent kmemleak scans of these pages cause the
1599 		 * non-PRESENT page faults.
1600 		 * Mark scratch areas with kmemleak_ignore_phys() to exclude
1601 		 * them from kmemleak scanning.
1602 		 */
1603 		kmemleak_ignore_phys(kho_scratch[i].addr);
1604 		for (pfn = base_pfn; pfn < base_pfn + count;
1605 		     pfn += pageblock_nr_pages)
1606 			init_cma_reserved_pageblock(pfn_to_page(pfn));
1607 	}
1608 
1609 	WARN_ON_ONCE(kho_debugfs_blob_add(&kho_out.dbg, "fdt",
1610 					  kho_out.fdt,
1611 					  fdt_totalsize(kho_out.fdt), true));
1612 
1613 	return 0;
1614 
1615 err_free_fdt:
1616 	kho_unpreserve_free(kho_out.fdt);
1617 err_free_kho_radix_tree_root:
1618 	kfree(tree->root);
1619 	tree->root = NULL;
1620 err_free_scratch:
1621 	kho_out.fdt = NULL;
1622 	for (int i = 0; i < kho_scratch_cnt; i++) {
1623 		void *start = __va(kho_scratch[i].addr);
1624 		void *end = start + kho_scratch[i].size;
1625 
1626 		free_reserved_area(start, end, -1, "");
1627 	}
1628 	kho_enable = false;
1629 	return err;
1630 }
1631 fs_initcall(kho_init);
1632 
1633 void __init kho_memory_init(void)
1634 {
1635 	if (kho_in.scratch_phys) {
1636 		kho_scratch = phys_to_virt(kho_in.scratch_phys);
1637 
1638 		if (kho_mem_retrieve(kho_get_fdt()))
1639 			kho_in.fdt_phys = 0;
1640 	} else {
1641 		kho_reserve_scratch();
1642 	}
1643 }
1644 
1645 void __init kho_populate(phys_addr_t fdt_phys, u64 fdt_len,
1646 			 phys_addr_t scratch_phys, u64 scratch_len)
1647 {
1648 	unsigned int scratch_cnt = scratch_len / sizeof(*kho_scratch);
1649 	struct kho_scratch *scratch = NULL;
1650 	phys_addr_t mem_map_phys;
1651 	void *fdt = NULL;
1652 	bool populated = false;
1653 	int err;
1654 
1655 	/* Validate the input FDT */
1656 	fdt = early_memremap(fdt_phys, fdt_len);
1657 	if (!fdt) {
1658 		pr_warn("setup: failed to memremap FDT (0x%llx)\n", fdt_phys);
1659 		goto report;
1660 	}
1661 	err = fdt_check_header(fdt);
1662 	if (err) {
1663 		pr_warn("setup: handover FDT (0x%llx) is invalid: %d\n",
1664 			fdt_phys, err);
1665 		goto unmap_fdt;
1666 	}
1667 	err = fdt_node_check_compatible(fdt, 0, KHO_FDT_COMPATIBLE);
1668 	if (err) {
1669 		pr_warn("setup: handover FDT (0x%llx) is incompatible with '%s': %d\n",
1670 			fdt_phys, KHO_FDT_COMPATIBLE, err);
1671 		goto unmap_fdt;
1672 	}
1673 
1674 	mem_map_phys = kho_get_mem_map_phys(fdt);
1675 	if (!mem_map_phys)
1676 		goto unmap_fdt;
1677 
1678 	scratch = early_memremap(scratch_phys, scratch_len);
1679 	if (!scratch) {
1680 		pr_warn("setup: failed to memremap scratch (phys=0x%llx, len=%lld)\n",
1681 			scratch_phys, scratch_len);
1682 		goto unmap_fdt;
1683 	}
1684 
1685 	/*
1686 	 * We pass a safe contiguous blocks of memory to use for early boot
1687 	 * purporses from the previous kernel so that we can resize the
1688 	 * memblock array as needed.
1689 	 */
1690 	for (int i = 0; i < scratch_cnt; i++) {
1691 		struct kho_scratch *area = &scratch[i];
1692 		u64 size = area->size;
1693 
1694 		memblock_add(area->addr, size);
1695 		err = memblock_mark_kho_scratch(area->addr, size);
1696 		if (err) {
1697 			pr_warn("failed to mark the scratch region 0x%pa+0x%pa: %pe",
1698 				&area->addr, &size, ERR_PTR(err));
1699 			goto unmap_scratch;
1700 		}
1701 		pr_debug("Marked 0x%pa+0x%pa as scratch", &area->addr, &size);
1702 	}
1703 
1704 	memblock_reserve(scratch_phys, scratch_len);
1705 
1706 	/*
1707 	 * Now that we have a viable region of scratch memory, let's tell
1708 	 * the memblocks allocator to only use that for any allocations.
1709 	 * That way we ensure that nothing scribbles over in use data while
1710 	 * we initialize the page tables which we will need to ingest all
1711 	 * memory reservations from the previous kernel.
1712 	 */
1713 	memblock_set_kho_scratch_only();
1714 
1715 	kho_in.fdt_phys = fdt_phys;
1716 	kho_in.scratch_phys = scratch_phys;
1717 	kho_scratch_cnt = scratch_cnt;
1718 
1719 	populated = true;
1720 	pr_info("found kexec handover data.\n");
1721 
1722 unmap_scratch:
1723 	early_memunmap(scratch, scratch_len);
1724 unmap_fdt:
1725 	early_memunmap(fdt, fdt_len);
1726 report:
1727 	if (!populated)
1728 		pr_warn("disabling KHO revival\n");
1729 }
1730 
1731 /* Helper functions for kexec_file_load */
1732 
1733 int kho_fill_kimage(struct kimage *image)
1734 {
1735 	ssize_t scratch_size;
1736 	int err = 0;
1737 	struct kexec_buf scratch;
1738 
1739 	if (!kho_enable || image->type == KEXEC_TYPE_CRASH)
1740 		return 0;
1741 
1742 	image->kho.fdt = virt_to_phys(kho_out.fdt);
1743 
1744 	scratch_size = sizeof(*kho_scratch) * kho_scratch_cnt;
1745 	scratch = (struct kexec_buf){
1746 		.image = image,
1747 		.buffer = kho_scratch,
1748 		.bufsz = scratch_size,
1749 		.mem = KEXEC_BUF_MEM_UNKNOWN,
1750 		.memsz = scratch_size,
1751 		.buf_align = SZ_64K, /* Makes it easier to map */
1752 		.buf_max = ULONG_MAX,
1753 		.top_down = true,
1754 	};
1755 	err = kexec_add_buffer(&scratch);
1756 	if (err)
1757 		return err;
1758 	image->kho.scratch = &image->segment[image->nr_segments - 1];
1759 
1760 	return 0;
1761 }
1762 
1763 static int kho_walk_scratch(struct kexec_buf *kbuf,
1764 			    int (*func)(struct resource *, void *))
1765 {
1766 	int ret = 0;
1767 	int i;
1768 
1769 	for (i = 0; i < kho_scratch_cnt; i++) {
1770 		struct resource res = {
1771 			.start = kho_scratch[i].addr,
1772 			.end = kho_scratch[i].addr + kho_scratch[i].size - 1,
1773 		};
1774 
1775 		/* Try to fit the kimage into our KHO scratch region */
1776 		ret = func(&res, kbuf);
1777 		if (ret)
1778 			break;
1779 	}
1780 
1781 	return ret;
1782 }
1783 
1784 int kho_locate_mem_hole(struct kexec_buf *kbuf,
1785 			int (*func)(struct resource *, void *))
1786 {
1787 	int ret;
1788 
1789 	if (!kho_enable || kbuf->image->type == KEXEC_TYPE_CRASH)
1790 		return 1;
1791 
1792 	ret = kho_walk_scratch(kbuf, func);
1793 
1794 	return ret == 1 ? 0 : -EADDRNOTAVAIL;
1795 }
1796