1 // SPDX-License-Identifier: GPL-2.0-only 2 /* 3 * kexec_handover.c - kexec handover metadata processing 4 * Copyright (C) 2023 Alexander Graf <graf@amazon.com> 5 * Copyright (C) 2025 Microsoft Corporation, Mike Rapoport <rppt@kernel.org> 6 * Copyright (C) 2025 Google LLC, Changyuan Lyu <changyuanl@google.com> 7 * Copyright (C) 2025 Pasha Tatashin <pasha.tatashin@soleen.com> 8 * Copyright (C) 2026 Google LLC, Jason Miu <jasonmiu@google.com> 9 */ 10 11 #define pr_fmt(fmt) "KHO: " fmt 12 13 #include <linux/cleanup.h> 14 #include <linux/cma.h> 15 #include <linux/kmemleak.h> 16 #include <linux/count_zeros.h> 17 #include <linux/kasan.h> 18 #include <linux/kexec.h> 19 #include <linux/kexec_handover.h> 20 #include <linux/kho_radix_tree.h> 21 #include <linux/utsname.h> 22 #include <linux/kho/abi/kexec_handover.h> 23 #include <linux/kho/abi/kexec_metadata.h> 24 #include <linux/libfdt.h> 25 #include <linux/list.h> 26 #include <linux/memblock.h> 27 #include <linux/page-isolation.h> 28 #include <linux/unaligned.h> 29 #include <linux/vmalloc.h> 30 31 #include <asm/early_ioremap.h> 32 33 /* 34 * KHO is tightly coupled with mm init and needs access to some of mm 35 * internal APIs. 36 */ 37 #include "../../mm/mm_init.h" 38 #include "../../mm/vmalloc.h" 39 #include "../kexec_internal.h" 40 #include "kexec_handover_internal.h" 41 42 /* 43 * This is the minimal alignment required by deferred struct page init. 44 * deferred_init_memmap_chunk frees memory to the buddy allocator, which looks 45 * at the neighboring pages (up to MAX_PAGE_ORDER) to merge them. 46 * If KHO scratch is not aligned to that value, buddy can access uninitialized 47 * struct pages, which can cause a crash. 48 */ 49 #define SCRATCH_ALIGNMENT_BYTES (PAGE_SIZE * MAX_ORDER_NR_PAGES) 50 static_assert(SCRATCH_ALIGNMENT_BYTES >= CMA_MIN_ALIGNMENT_BYTES); 51 52 /* The magic token for preserved pages */ 53 #define KHO_PAGE_MAGIC 0x4b484f50U /* ASCII for 'KHOP' */ 54 55 /* 56 * KHO uses page->private, which is an unsigned long, to store page metadata. 57 * Use it to store both the magic and the order. 58 */ 59 union kho_page_info { 60 unsigned long page_private; 61 struct { 62 unsigned int order; 63 unsigned int magic; 64 }; 65 }; 66 67 static_assert(sizeof(union kho_page_info) == sizeof(((struct page *)0)->private)); 68 69 static bool kho_enable __ro_after_init = IS_ENABLED(CONFIG_KEXEC_HANDOVER_ENABLE_DEFAULT); 70 71 bool kho_is_enabled(void) 72 { 73 return kho_enable; 74 } 75 EXPORT_SYMBOL_GPL(kho_is_enabled); 76 77 static int __init kho_parse_enable(char *p) 78 { 79 return kstrtobool(p, &kho_enable); 80 } 81 early_param("kho", kho_parse_enable); 82 83 struct kho_out { 84 void *fdt; 85 struct mutex lock; /* protects KHO FDT */ 86 87 struct kho_radix_tree radix_tree; 88 struct kho_debugfs dbg; 89 }; 90 91 static struct kho_out kho_out = { 92 .lock = __MUTEX_INITIALIZER(kho_out.lock), 93 .radix_tree = { 94 .lock = __MUTEX_INITIALIZER(kho_out.radix_tree.lock), 95 }, 96 }; 97 98 /** 99 * kho_radix_encode_key - Encodes a physical address and order into a radix key. 100 * @phys: The physical address of the page. 101 * @order: The order of the page. 102 * 103 * This function combines a page's physical address and its order into a 104 * single unsigned long, which is used as a key for all radix tree 105 * operations. 106 * 107 * Return: The encoded unsigned long radix key. 108 */ 109 static unsigned long kho_radix_encode_key(phys_addr_t phys, unsigned int order) 110 { 111 /* Order bits part */ 112 unsigned long h = 1UL << (KHO_ORDER_0_LOG2 - order); 113 /* Shifted physical address part */ 114 unsigned long l = phys >> (PAGE_SHIFT + order); 115 116 return h | l; 117 } 118 119 /** 120 * kho_radix_decode_key - Decodes a radix key back into a physical address and order. 121 * @key: The unsigned long key to decode. 122 * @order: An output parameter, a pointer to an unsigned int where the decoded 123 * page order will be stored. 124 * 125 * This function reverses the encoding performed by kho_radix_encode_key(), 126 * extracting the original physical address and page order from a given key. 127 * 128 * Return: The decoded physical address. 129 */ 130 static phys_addr_t kho_radix_decode_key(unsigned long key, unsigned int *order) 131 { 132 unsigned int order_bit = fls64(key); 133 phys_addr_t phys; 134 135 /* order_bit is numbered starting at 1 from fls64 */ 136 *order = KHO_ORDER_0_LOG2 - order_bit + 1; 137 /* The order is discarded by the shift */ 138 phys = key << (PAGE_SHIFT + *order); 139 140 return phys; 141 } 142 143 static unsigned long kho_radix_get_bitmap_index(unsigned long key) 144 { 145 return key % (1 << KHO_BITMAP_SIZE_LOG2); 146 } 147 148 static unsigned long kho_radix_get_table_index(unsigned long key, 149 unsigned int level) 150 { 151 int s; 152 153 s = ((level - 1) * KHO_TABLE_SIZE_LOG2) + KHO_BITMAP_SIZE_LOG2; 154 return (key >> s) % (1 << KHO_TABLE_SIZE_LOG2); 155 } 156 157 /** 158 * kho_radix_add_page - Marks a page as preserved in the radix tree. 159 * @tree: The KHO radix tree. 160 * @pfn: The page frame number of the page to preserve. 161 * @order: The order of the page. 162 * 163 * This function traverses the radix tree based on the key derived from @pfn 164 * and @order. It sets the corresponding bit in the leaf bitmap to mark the 165 * page for preservation. If intermediate nodes do not exist along the path, 166 * they are allocated and added to the tree. 167 * 168 * Return: 0 on success, or a negative error code on failure. 169 */ 170 int kho_radix_add_page(struct kho_radix_tree *tree, 171 unsigned long pfn, unsigned int order) 172 { 173 /* Newly allocated nodes for error cleanup */ 174 struct kho_radix_node *intermediate_nodes[KHO_TREE_MAX_DEPTH] = { 0 }; 175 unsigned long key = kho_radix_encode_key(PFN_PHYS(pfn), order); 176 struct kho_radix_node *anchor_node = NULL; 177 struct kho_radix_node *node = tree->root; 178 struct kho_radix_node *new_node; 179 unsigned int i, idx, anchor_idx; 180 struct kho_radix_leaf *leaf; 181 int err = 0; 182 183 if (WARN_ON_ONCE(!tree->root)) 184 return -EINVAL; 185 186 might_sleep(); 187 188 guard(mutex)(&tree->lock); 189 190 /* Go from high levels to low levels */ 191 for (i = KHO_TREE_MAX_DEPTH - 1; i > 0; i--) { 192 idx = kho_radix_get_table_index(key, i); 193 194 if (node->table[idx]) { 195 node = phys_to_virt(node->table[idx]); 196 continue; 197 } 198 199 /* Next node is empty, create a new node for it */ 200 new_node = (struct kho_radix_node *)get_zeroed_page(GFP_KERNEL); 201 if (!new_node) { 202 err = -ENOMEM; 203 goto err_free_nodes; 204 } 205 206 node->table[idx] = virt_to_phys(new_node); 207 208 /* 209 * Capture the node where the new branch starts for cleanup 210 * if allocation fails. 211 */ 212 if (!anchor_node) { 213 anchor_node = node; 214 anchor_idx = idx; 215 } 216 intermediate_nodes[i] = new_node; 217 218 node = new_node; 219 } 220 221 /* Handle the leaf level bitmap (level 0) */ 222 idx = kho_radix_get_bitmap_index(key); 223 leaf = (struct kho_radix_leaf *)node; 224 __set_bit(idx, leaf->bitmap); 225 226 return 0; 227 228 err_free_nodes: 229 for (i = KHO_TREE_MAX_DEPTH - 1; i > 0; i--) { 230 if (intermediate_nodes[i]) 231 free_page((unsigned long)intermediate_nodes[i]); 232 } 233 if (anchor_node) 234 anchor_node->table[anchor_idx] = 0; 235 236 return err; 237 } 238 EXPORT_SYMBOL_GPL(kho_radix_add_page); 239 240 /** 241 * kho_radix_del_page - Removes a page's preservation status from the radix tree. 242 * @tree: The KHO radix tree. 243 * @pfn: The page frame number of the page to unpreserve. 244 * @order: The order of the page. 245 * 246 * This function traverses the radix tree and clears the bit corresponding to 247 * the page, effectively removing its "preserved" status. It does not free 248 * the tree's intermediate nodes, even if they become empty. 249 */ 250 void kho_radix_del_page(struct kho_radix_tree *tree, unsigned long pfn, 251 unsigned int order) 252 { 253 unsigned long key = kho_radix_encode_key(PFN_PHYS(pfn), order); 254 struct kho_radix_node *node = tree->root; 255 struct kho_radix_leaf *leaf; 256 unsigned int i, idx; 257 258 if (WARN_ON_ONCE(!tree->root)) 259 return; 260 261 might_sleep(); 262 263 guard(mutex)(&tree->lock); 264 265 /* Go from high levels to low levels */ 266 for (i = KHO_TREE_MAX_DEPTH - 1; i > 0; i--) { 267 idx = kho_radix_get_table_index(key, i); 268 269 /* 270 * Attempting to delete a page that has not been preserved, 271 * return with a warning. 272 */ 273 if (WARN_ON(!node->table[idx])) 274 return; 275 276 node = phys_to_virt(node->table[idx]); 277 } 278 279 /* Handle the leaf level bitmap (level 0) */ 280 leaf = (struct kho_radix_leaf *)node; 281 idx = kho_radix_get_bitmap_index(key); 282 __clear_bit(idx, leaf->bitmap); 283 } 284 EXPORT_SYMBOL_GPL(kho_radix_del_page); 285 286 static int kho_radix_walk_leaf(struct kho_radix_leaf *leaf, 287 unsigned long key, 288 kho_radix_tree_walk_callback_t cb) 289 { 290 unsigned long *bitmap = (unsigned long *)leaf; 291 unsigned int order; 292 phys_addr_t phys; 293 unsigned int i; 294 int err; 295 296 for_each_set_bit(i, bitmap, PAGE_SIZE * BITS_PER_BYTE) { 297 phys = kho_radix_decode_key(key | i, &order); 298 err = cb(phys, order); 299 if (err) 300 return err; 301 } 302 303 return 0; 304 } 305 306 static int __kho_radix_walk_tree(struct kho_radix_node *root, 307 unsigned int level, unsigned long start, 308 kho_radix_tree_walk_callback_t cb) 309 { 310 struct kho_radix_node *node; 311 struct kho_radix_leaf *leaf; 312 unsigned long key, i; 313 unsigned int shift; 314 int err; 315 316 for (i = 0; i < PAGE_SIZE / sizeof(phys_addr_t); i++) { 317 if (!root->table[i]) 318 continue; 319 320 shift = ((level - 1) * KHO_TABLE_SIZE_LOG2) + 321 KHO_BITMAP_SIZE_LOG2; 322 key = start | (i << shift); 323 324 node = phys_to_virt(root->table[i]); 325 326 if (level == 1) { 327 /* 328 * we are at level 1, 329 * node is pointing to the level 0 bitmap. 330 */ 331 leaf = (struct kho_radix_leaf *)node; 332 err = kho_radix_walk_leaf(leaf, key, cb); 333 } else { 334 err = __kho_radix_walk_tree(node, level - 1, 335 key, cb); 336 } 337 338 if (err) 339 return err; 340 } 341 342 return 0; 343 } 344 345 /** 346 * kho_radix_walk_tree - Traverses the radix tree and calls a callback for each preserved page. 347 * @tree: A pointer to the KHO radix tree to walk. 348 * @cb: A callback function of type kho_radix_tree_walk_callback_t that will be 349 * invoked for each preserved page found in the tree. The callback receives 350 * the physical address and order of the preserved page. 351 * 352 * This function walks the radix tree, searching from the specified top level 353 * down to the lowest level (level 0). For each preserved page found, it invokes 354 * the provided callback, passing the page's physical address and order. 355 * 356 * Return: 0 if the walk completed the specified tree, or the non-zero return 357 * value from the callback that stopped the walk. 358 */ 359 int kho_radix_walk_tree(struct kho_radix_tree *tree, 360 kho_radix_tree_walk_callback_t cb) 361 { 362 if (WARN_ON_ONCE(!tree->root)) 363 return -EINVAL; 364 365 guard(mutex)(&tree->lock); 366 367 return __kho_radix_walk_tree(tree->root, KHO_TREE_MAX_DEPTH - 1, 0, cb); 368 } 369 EXPORT_SYMBOL_GPL(kho_radix_walk_tree); 370 371 /* For physically contiguous 0-order pages. */ 372 static void kho_init_pages(struct page *page, unsigned long nr_pages) 373 { 374 for (unsigned long i = 0; i < nr_pages; i++) { 375 set_page_count(page + i, 1); 376 /* Clear each page's codetag to avoid accounting mismatch. */ 377 clear_page_tag_ref(page + i); 378 } 379 } 380 381 static void kho_init_folio(struct page *page, unsigned int order) 382 { 383 unsigned long nr_pages = (1 << order); 384 385 /* Head page gets refcount of 1. */ 386 set_page_count(page, 1); 387 /* Clear head page's codetag to avoid accounting mismatch. */ 388 clear_page_tag_ref(page); 389 390 /* For higher order folios, tail pages get a page count of zero. */ 391 for (unsigned long i = 1; i < nr_pages; i++) 392 set_page_count(page + i, 0); 393 394 if (order > 0) 395 prep_compound_page(page, order); 396 } 397 398 static struct page *kho_restore_page(phys_addr_t phys, bool is_folio) 399 { 400 struct page *page = pfn_to_online_page(PHYS_PFN(phys)); 401 unsigned long nr_pages; 402 union kho_page_info info; 403 404 if (!page) 405 return NULL; 406 407 info.page_private = page->private; 408 /* 409 * deserialize_bitmap() only sets the magic on the head page. This magic 410 * check also implicitly makes sure phys is order-aligned since for 411 * non-order-aligned phys addresses, magic will never be set. 412 */ 413 if (WARN_ON_ONCE(info.magic != KHO_PAGE_MAGIC)) 414 return NULL; 415 nr_pages = (1 << info.order); 416 417 /* Clear private to make sure later restores on this page error out. */ 418 page->private = 0; 419 420 if (is_folio) 421 kho_init_folio(page, info.order); 422 else 423 kho_init_pages(page, nr_pages); 424 425 adjust_managed_page_count(page, nr_pages); 426 return page; 427 } 428 429 /** 430 * kho_restore_folio - recreates the folio from the preserved memory. 431 * @phys: physical address of the folio. 432 * 433 * Return: pointer to the struct folio on success, NULL on failure. 434 */ 435 struct folio *kho_restore_folio(phys_addr_t phys) 436 { 437 struct page *page = kho_restore_page(phys, true); 438 439 return page ? page_folio(page) : NULL; 440 } 441 EXPORT_SYMBOL_GPL(kho_restore_folio); 442 443 /** 444 * kho_restore_pages - restore list of contiguous order 0 pages. 445 * @phys: physical address of the first page. 446 * @nr_pages: number of pages. 447 * 448 * Restore a contiguous list of order 0 pages that was preserved with 449 * kho_preserve_pages(). 450 * 451 * Return: the first page on success, NULL on failure. 452 */ 453 struct page *kho_restore_pages(phys_addr_t phys, unsigned long nr_pages) 454 { 455 const unsigned long start_pfn = PHYS_PFN(phys); 456 const unsigned long end_pfn = start_pfn + nr_pages; 457 unsigned long pfn = start_pfn; 458 459 while (pfn < end_pfn) { 460 const unsigned int order = 461 min(count_trailing_zeros(pfn), ilog2(end_pfn - pfn)); 462 struct page *page = kho_restore_page(PFN_PHYS(pfn), false); 463 464 if (!page) 465 return NULL; 466 pfn += 1 << order; 467 } 468 469 return pfn_to_page(start_pfn); 470 } 471 EXPORT_SYMBOL_GPL(kho_restore_pages); 472 473 /* 474 * With CONFIG_DEFERRED_STRUCT_PAGE_INIT, struct pages in higher memory regions 475 * may not be initialized yet at the time KHO deserializes preserved memory. 476 * KHO uses the struct page to store metadata and a later initialization would 477 * overwrite it. 478 * Ensure all the struct pages in the preservation are 479 * initialized. kho_preserved_memory_reserve() marks the reservation as noinit 480 * to make sure they don't get re-initialized later. 481 */ 482 static struct page *__init kho_get_preserved_page(phys_addr_t phys, 483 unsigned int order) 484 { 485 unsigned long pfn = PHYS_PFN(phys); 486 int nid; 487 488 if (!IS_ENABLED(CONFIG_DEFERRED_STRUCT_PAGE_INIT)) 489 return pfn_to_page(pfn); 490 491 nid = early_pfn_to_nid(pfn); 492 for (unsigned long i = 0; i < (1UL << order); i++) 493 init_deferred_page(pfn + i, nid); 494 495 return pfn_to_page(pfn); 496 } 497 498 static int __init kho_preserved_memory_reserve(phys_addr_t phys, 499 unsigned int order) 500 { 501 union kho_page_info info; 502 struct page *page; 503 u64 sz; 504 505 sz = 1UL << (order + PAGE_SHIFT); 506 page = kho_get_preserved_page(phys, order); 507 508 /* Reserve the memory preserved in KHO in memblock */ 509 memblock_reserve(phys, sz); 510 memblock_reserved_mark_noinit(phys, sz); 511 info.magic = KHO_PAGE_MAGIC; 512 info.order = order; 513 page->private = info.page_private; 514 515 return 0; 516 } 517 518 /* Returns physical address of the preserved memory map from FDT */ 519 static phys_addr_t __init kho_get_mem_map_phys(const void *fdt) 520 { 521 const void *mem_ptr; 522 int len; 523 524 mem_ptr = fdt_getprop(fdt, 0, KHO_FDT_MEMORY_MAP_PROP_NAME, &len); 525 if (!mem_ptr || len != sizeof(u64)) { 526 pr_err("failed to get preserved memory map\n"); 527 return 0; 528 } 529 530 return get_unaligned((const u64 *)mem_ptr); 531 } 532 533 /* 534 * With KHO enabled, memory can become fragmented because KHO regions may 535 * be anywhere in physical address space. The scratch regions give us a 536 * safe zones that we will never see KHO allocations from. This is where we 537 * can later safely load our new kexec images into and then use the scratch 538 * area for early allocations that happen before page allocator is 539 * initialized. 540 */ 541 struct kho_scratch *kho_scratch; 542 unsigned int kho_scratch_cnt; 543 544 /* 545 * The scratch areas are scaled by default as percent of memory allocated from 546 * memblock. A user can override the scale with command line parameter: 547 * 548 * kho_scratch=N% 549 * 550 * It is also possible to explicitly define size for a lowmem, a global and 551 * per-node scratch areas: 552 * 553 * kho_scratch=l[KMG],n[KMG],m[KMG] 554 * 555 * The explicit size definition takes precedence over scale definition. 556 */ 557 static unsigned int scratch_scale __initdata = 200; 558 static phys_addr_t scratch_size_global __initdata; 559 static phys_addr_t scratch_size_pernode __initdata; 560 static phys_addr_t scratch_size_lowmem __initdata; 561 562 static int __init kho_parse_scratch_size(char *p) 563 { 564 size_t len; 565 unsigned long sizes[3]; 566 size_t total_size = 0; 567 int i; 568 569 if (!p) 570 return -EINVAL; 571 572 len = strlen(p); 573 if (!len) 574 return -EINVAL; 575 576 /* parse nn% */ 577 if (p[len - 1] == '%') { 578 /* unsigned int max is 4,294,967,295, 10 chars */ 579 char s_scale[11] = {}; 580 int ret = 0; 581 582 if (len > ARRAY_SIZE(s_scale)) 583 return -EINVAL; 584 585 memcpy(s_scale, p, len - 1); 586 ret = kstrtouint(s_scale, 10, &scratch_scale); 587 if (!ret) 588 pr_notice("scratch scale is %d%%\n", scratch_scale); 589 return ret; 590 } 591 592 /* parse ll[KMG],mm[KMG],nn[KMG] */ 593 for (i = 0; i < ARRAY_SIZE(sizes); i++) { 594 char *endp = p; 595 596 if (i > 0) { 597 if (*p != ',') 598 return -EINVAL; 599 p += 1; 600 } 601 602 sizes[i] = memparse(p, &endp); 603 if (endp == p) 604 return -EINVAL; 605 p = endp; 606 total_size += sizes[i]; 607 } 608 609 if (!total_size) 610 return -EINVAL; 611 612 /* The string should be fully consumed by now. */ 613 if (*p) 614 return -EINVAL; 615 616 scratch_size_lowmem = sizes[0]; 617 scratch_size_global = sizes[1]; 618 scratch_size_pernode = sizes[2]; 619 scratch_scale = 0; 620 621 pr_notice("scratch areas: lowmem: %lluMiB global: %lluMiB pernode: %lldMiB\n", 622 (u64)(scratch_size_lowmem >> 20), 623 (u64)(scratch_size_global >> 20), 624 (u64)(scratch_size_pernode >> 20)); 625 626 return 0; 627 } 628 early_param("kho_scratch", kho_parse_scratch_size); 629 630 static void __init scratch_size_update(void) 631 { 632 /* 633 * If fixed sizes are not provided via command line, calculate them 634 * now. 635 */ 636 if (scratch_scale) { 637 phys_addr_t size; 638 639 size = memblock_reserved_kern_size(ARCH_LOW_ADDRESS_LIMIT, 640 NUMA_NO_NODE); 641 size = size * scratch_scale / 100; 642 scratch_size_lowmem = size; 643 644 size = memblock_reserved_kern_size(MEMBLOCK_ALLOC_ANYWHERE, 645 NUMA_NO_NODE); 646 size = size * scratch_scale / 100 - scratch_size_lowmem; 647 scratch_size_global = size; 648 } 649 650 /* 651 * Scratch areas are released as MIGRATE_CMA. Round them up to the right 652 * size. 653 */ 654 scratch_size_lowmem = round_up(scratch_size_lowmem, SCRATCH_ALIGNMENT_BYTES); 655 scratch_size_global = round_up(scratch_size_global, SCRATCH_ALIGNMENT_BYTES); 656 } 657 658 static phys_addr_t __init scratch_size_node(int nid) 659 { 660 phys_addr_t size; 661 662 if (scratch_scale) { 663 size = memblock_reserved_kern_size(MEMBLOCK_ALLOC_ANYWHERE, 664 nid); 665 size = size * scratch_scale / 100; 666 } else { 667 size = scratch_size_pernode; 668 } 669 670 return round_up(size, SCRATCH_ALIGNMENT_BYTES); 671 } 672 673 /** 674 * kho_reserve_scratch - Reserve a contiguous chunk of memory for kexec 675 * 676 * With KHO we can preserve arbitrary pages in the system. To ensure we still 677 * have a large contiguous region of memory when we search the physical address 678 * space for target memory, let's make sure we always have a large CMA region 679 * active. This CMA region will only be used for movable pages which are not a 680 * problem for us during KHO because we can just move them somewhere else. 681 */ 682 static void __init kho_reserve_scratch(void) 683 { 684 phys_addr_t addr, size; 685 int nid, i = 0; 686 687 if (!kho_enable) 688 return; 689 690 scratch_size_update(); 691 692 /* FIXME: deal with node hot-plug/remove */ 693 kho_scratch_cnt = nodes_weight(node_states[N_MEMORY]) + 2; 694 size = kho_scratch_cnt * sizeof(*kho_scratch); 695 kho_scratch = memblock_alloc(size, PAGE_SIZE); 696 if (!kho_scratch) { 697 pr_err("Failed to reserve scratch array\n"); 698 goto err_disable_kho; 699 } 700 701 /* 702 * reserve scratch area in low memory for lowmem allocations in the 703 * next kernel 704 */ 705 size = scratch_size_lowmem; 706 addr = memblock_phys_alloc_range(size, SCRATCH_ALIGNMENT_BYTES, 0, 707 ARCH_LOW_ADDRESS_LIMIT); 708 if (!addr) { 709 pr_err("Failed to reserve lowmem scratch buffer\n"); 710 goto err_free_scratch_desc; 711 } 712 713 kho_scratch[i].addr = addr; 714 kho_scratch[i].size = size; 715 i++; 716 717 /* reserve large contiguous area for allocations without nid */ 718 size = scratch_size_global; 719 addr = memblock_phys_alloc(size, SCRATCH_ALIGNMENT_BYTES); 720 if (!addr) { 721 pr_err("Failed to reserve global scratch buffer\n"); 722 goto err_free_scratch_areas; 723 } 724 725 kho_scratch[i].addr = addr; 726 kho_scratch[i].size = size; 727 i++; 728 729 /* 730 * Loop over nodes that have both memory and are online. Skip 731 * memoryless nodes, as we can not allocate scratch areas there. 732 */ 733 for_each_node_state(nid, N_MEMORY) { 734 size = scratch_size_node(nid); 735 addr = memblock_alloc_range_nid(size, SCRATCH_ALIGNMENT_BYTES, 736 0, MEMBLOCK_ALLOC_ACCESSIBLE, 737 nid, true); 738 if (!addr) { 739 pr_err("Failed to reserve nid %d scratch buffer\n", nid); 740 goto err_free_scratch_areas; 741 } 742 743 kho_scratch[i].addr = addr; 744 kho_scratch[i].size = size; 745 i++; 746 } 747 748 return; 749 750 err_free_scratch_areas: 751 for (i--; i >= 0; i--) 752 memblock_phys_free(kho_scratch[i].addr, kho_scratch[i].size); 753 err_free_scratch_desc: 754 memblock_free(kho_scratch, kho_scratch_cnt * sizeof(*kho_scratch)); 755 err_disable_kho: 756 pr_warn("Failed to reserve scratch area, disabling kexec handover\n"); 757 kho_enable = false; 758 } 759 760 /** 761 * kho_add_subtree - record the physical address of a sub blob in KHO root tree. 762 * @name: name of the sub tree. 763 * @blob: the sub tree blob. 764 * @size: size of the blob in bytes. 765 * 766 * Creates a new child node named @name in KHO root FDT and records 767 * the physical address of @blob. The pages of @blob must also be preserved 768 * by KHO for the new kernel to retrieve it after kexec. 769 * 770 * A debugfs blob entry is also created at 771 * ``/sys/kernel/debug/kho/out/sub_fdts/@name`` when kernel is configured with 772 * CONFIG_KEXEC_HANDOVER_DEBUGFS 773 * 774 * Return: 0 on success, error code on failure 775 */ 776 int kho_add_subtree(const char *name, void *blob, size_t size) 777 { 778 phys_addr_t phys = virt_to_phys(blob); 779 void *root_fdt = kho_out.fdt; 780 u64 size_u64 = size; 781 int err = -ENOMEM; 782 int off, fdt_err; 783 784 guard(mutex)(&kho_out.lock); 785 786 fdt_err = fdt_open_into(root_fdt, root_fdt, PAGE_SIZE); 787 if (fdt_err < 0) 788 return err; 789 790 off = fdt_add_subnode(root_fdt, 0, name); 791 if (off < 0) { 792 if (off == -FDT_ERR_EXISTS) 793 err = -EEXIST; 794 goto out_pack; 795 } 796 797 fdt_err = fdt_setprop(root_fdt, off, KHO_SUB_TREE_PROP_NAME, 798 &phys, sizeof(phys)); 799 if (fdt_err < 0) 800 goto out_del_node; 801 802 fdt_err = fdt_setprop(root_fdt, off, KHO_SUB_TREE_SIZE_PROP_NAME, 803 &size_u64, sizeof(size_u64)); 804 if (fdt_err < 0) 805 goto out_del_node; 806 807 WARN_ON_ONCE(kho_debugfs_blob_add(&kho_out.dbg, name, blob, 808 size, false)); 809 810 err = 0; 811 goto out_pack; 812 813 out_del_node: 814 fdt_del_node(root_fdt, off); 815 out_pack: 816 fdt_pack(root_fdt); 817 818 return err; 819 } 820 EXPORT_SYMBOL_GPL(kho_add_subtree); 821 822 void kho_remove_subtree(void *blob) 823 { 824 phys_addr_t target_phys = virt_to_phys(blob); 825 void *root_fdt = kho_out.fdt; 826 int off; 827 int err; 828 829 guard(mutex)(&kho_out.lock); 830 831 err = fdt_open_into(root_fdt, root_fdt, PAGE_SIZE); 832 if (err < 0) 833 return; 834 835 for (off = fdt_first_subnode(root_fdt, 0); off >= 0; 836 off = fdt_next_subnode(root_fdt, off)) { 837 const u64 *val; 838 int len; 839 840 val = fdt_getprop(root_fdt, off, KHO_SUB_TREE_PROP_NAME, &len); 841 if (!val || len != sizeof(phys_addr_t)) 842 continue; 843 844 if ((phys_addr_t)*val == target_phys) { 845 fdt_del_node(root_fdt, off); 846 kho_debugfs_blob_remove(&kho_out.dbg, blob); 847 break; 848 } 849 } 850 851 fdt_pack(root_fdt); 852 } 853 EXPORT_SYMBOL_GPL(kho_remove_subtree); 854 855 /** 856 * kho_preserve_folio - preserve a folio across kexec. 857 * @folio: folio to preserve. 858 * 859 * Instructs KHO to preserve the whole folio across kexec. The order 860 * will be preserved as well. 861 * 862 * Return: 0 on success, error code on failure 863 */ 864 int kho_preserve_folio(struct folio *folio) 865 { 866 struct kho_radix_tree *tree = &kho_out.radix_tree; 867 const unsigned long pfn = folio_pfn(folio); 868 const unsigned int order = folio_order(folio); 869 870 if (WARN_ON(kho_scratch_overlap(pfn << PAGE_SHIFT, PAGE_SIZE << order))) 871 return -EINVAL; 872 873 return kho_radix_add_page(tree, pfn, order); 874 } 875 EXPORT_SYMBOL_GPL(kho_preserve_folio); 876 877 /** 878 * kho_unpreserve_folio - unpreserve a folio. 879 * @folio: folio to unpreserve. 880 * 881 * Instructs KHO to unpreserve a folio that was preserved by 882 * kho_preserve_folio() before. The provided @folio (pfn and order) 883 * must exactly match a previously preserved folio. 884 */ 885 void kho_unpreserve_folio(struct folio *folio) 886 { 887 struct kho_radix_tree *tree = &kho_out.radix_tree; 888 const unsigned long pfn = folio_pfn(folio); 889 const unsigned int order = folio_order(folio); 890 891 kho_radix_del_page(tree, pfn, order); 892 } 893 EXPORT_SYMBOL_GPL(kho_unpreserve_folio); 894 895 static unsigned int __kho_preserve_pages_order(unsigned long start_pfn, 896 unsigned long end_pfn) 897 { 898 unsigned int order = min(count_trailing_zeros(start_pfn), 899 ilog2(end_pfn - start_pfn)); 900 901 /* 902 * Make sure all the pages in a single preservation are in the same NUMA 903 * node. The restore machinery can not cope with a preservation spanning 904 * multiple NUMA nodes. 905 */ 906 while (pfn_to_nid(start_pfn) != pfn_to_nid(start_pfn + (1UL << order) - 1)) 907 order--; 908 909 return order; 910 } 911 912 static void __kho_unpreserve(struct kho_radix_tree *tree, 913 unsigned long pfn, unsigned long end_pfn) 914 { 915 unsigned int order; 916 917 while (pfn < end_pfn) { 918 order = __kho_preserve_pages_order(pfn, end_pfn); 919 920 kho_radix_del_page(tree, pfn, order); 921 922 pfn += 1 << order; 923 } 924 } 925 926 /** 927 * kho_preserve_pages - preserve contiguous pages across kexec 928 * @page: first page in the list. 929 * @nr_pages: number of pages. 930 * 931 * Preserve a contiguous list of order 0 pages. Must be restored using 932 * kho_restore_pages() to ensure the pages are restored properly as order 0. 933 * 934 * Return: 0 on success, error code on failure 935 */ 936 int kho_preserve_pages(struct page *page, unsigned long nr_pages) 937 { 938 struct kho_radix_tree *tree = &kho_out.radix_tree; 939 const unsigned long start_pfn = page_to_pfn(page); 940 const unsigned long end_pfn = start_pfn + nr_pages; 941 unsigned long pfn = start_pfn; 942 unsigned long failed_pfn = 0; 943 int err = 0; 944 945 if (WARN_ON(kho_scratch_overlap(start_pfn << PAGE_SHIFT, 946 nr_pages << PAGE_SHIFT))) { 947 return -EINVAL; 948 } 949 950 while (pfn < end_pfn) { 951 unsigned int order = __kho_preserve_pages_order(pfn, end_pfn); 952 953 err = kho_radix_add_page(tree, pfn, order); 954 if (err) { 955 failed_pfn = pfn; 956 break; 957 } 958 959 pfn += 1 << order; 960 } 961 962 if (err) 963 __kho_unpreserve(tree, start_pfn, failed_pfn); 964 965 return err; 966 } 967 EXPORT_SYMBOL_GPL(kho_preserve_pages); 968 969 /** 970 * kho_unpreserve_pages - unpreserve contiguous pages. 971 * @page: first page in the list. 972 * @nr_pages: number of pages. 973 * 974 * Instructs KHO to unpreserve @nr_pages contiguous pages starting from @page. 975 * This must be called with the same @page and @nr_pages as the corresponding 976 * kho_preserve_pages() call. Unpreserving arbitrary sub-ranges of larger 977 * preserved blocks is not supported. 978 */ 979 void kho_unpreserve_pages(struct page *page, unsigned long nr_pages) 980 { 981 struct kho_radix_tree *tree = &kho_out.radix_tree; 982 const unsigned long start_pfn = page_to_pfn(page); 983 const unsigned long end_pfn = start_pfn + nr_pages; 984 985 __kho_unpreserve(tree, start_pfn, end_pfn); 986 } 987 EXPORT_SYMBOL_GPL(kho_unpreserve_pages); 988 989 /* vmalloc flags KHO supports */ 990 #define KHO_VMALLOC_SUPPORTED_FLAGS (VM_ALLOC | VM_ALLOW_HUGE_VMAP) 991 992 /* KHO internal flags for vmalloc preservations */ 993 #define KHO_VMALLOC_ALLOC 0x0001 994 #define KHO_VMALLOC_HUGE_VMAP 0x0002 995 996 static unsigned short vmalloc_flags_to_kho(unsigned int vm_flags) 997 { 998 unsigned short kho_flags = 0; 999 1000 if (vm_flags & VM_ALLOC) 1001 kho_flags |= KHO_VMALLOC_ALLOC; 1002 if (vm_flags & VM_ALLOW_HUGE_VMAP) 1003 kho_flags |= KHO_VMALLOC_HUGE_VMAP; 1004 1005 return kho_flags; 1006 } 1007 1008 static unsigned int kho_flags_to_vmalloc(unsigned short kho_flags) 1009 { 1010 unsigned int vm_flags = 0; 1011 1012 if (kho_flags & KHO_VMALLOC_ALLOC) 1013 vm_flags |= VM_ALLOC; 1014 if (kho_flags & KHO_VMALLOC_HUGE_VMAP) 1015 vm_flags |= VM_ALLOW_HUGE_VMAP; 1016 1017 return vm_flags; 1018 } 1019 1020 static struct kho_vmalloc_chunk *new_vmalloc_chunk(struct kho_vmalloc_chunk *cur) 1021 { 1022 struct kho_vmalloc_chunk *chunk; 1023 int err; 1024 1025 chunk = (struct kho_vmalloc_chunk *)get_zeroed_page(GFP_KERNEL); 1026 if (!chunk) 1027 return NULL; 1028 1029 err = kho_preserve_pages(virt_to_page(chunk), 1); 1030 if (err) 1031 goto err_free; 1032 if (cur) 1033 KHOSER_STORE_PTR(cur->hdr.next, chunk); 1034 return chunk; 1035 1036 err_free: 1037 free_page((unsigned long)chunk); 1038 return NULL; 1039 } 1040 1041 static void kho_vmalloc_unpreserve_chunk(struct kho_vmalloc_chunk *chunk, 1042 unsigned short order) 1043 { 1044 struct kho_radix_tree *tree = &kho_out.radix_tree; 1045 unsigned long pfn = PHYS_PFN(virt_to_phys(chunk)); 1046 1047 __kho_unpreserve(tree, pfn, pfn + 1); 1048 1049 for (int i = 0; i < ARRAY_SIZE(chunk->phys) && chunk->phys[i]; i++) { 1050 pfn = PHYS_PFN(chunk->phys[i]); 1051 __kho_unpreserve(tree, pfn, pfn + (1 << order)); 1052 } 1053 } 1054 1055 /** 1056 * kho_preserve_vmalloc - preserve memory allocated with vmalloc() across kexec 1057 * @ptr: pointer to the area in vmalloc address space 1058 * @preservation: placeholder for preservation metadata 1059 * 1060 * Instructs KHO to preserve the area in vmalloc address space at @ptr. The 1061 * physical pages mapped at @ptr will be preserved and on successful return 1062 * @preservation will hold the physical address of a structure that describes 1063 * the preservation. 1064 * 1065 * NOTE: The memory allocated with vmalloc_node() variants cannot be reliably 1066 * restored on the same node 1067 * 1068 * Return: 0 on success, error code on failure 1069 */ 1070 int kho_preserve_vmalloc(void *ptr, struct kho_vmalloc *preservation) 1071 { 1072 struct kho_vmalloc_chunk *chunk; 1073 struct vm_struct *vm = find_vm_area(ptr); 1074 unsigned int order, flags, nr_contig_pages; 1075 unsigned int idx = 0; 1076 int err; 1077 1078 if (!vm) 1079 return -EINVAL; 1080 1081 if (vm->flags & ~KHO_VMALLOC_SUPPORTED_FLAGS) 1082 return -EOPNOTSUPP; 1083 1084 flags = vmalloc_flags_to_kho(vm->flags); 1085 order = get_vm_area_page_order(vm); 1086 1087 chunk = new_vmalloc_chunk(NULL); 1088 if (!chunk) 1089 return -ENOMEM; 1090 KHOSER_STORE_PTR(preservation->first, chunk); 1091 1092 nr_contig_pages = (1 << order); 1093 for (int i = 0; i < vm->nr_pages; i += nr_contig_pages) { 1094 phys_addr_t phys = page_to_phys(vm->pages[i]); 1095 1096 err = kho_preserve_pages(vm->pages[i], nr_contig_pages); 1097 if (err) 1098 goto err_free; 1099 1100 chunk->phys[idx++] = phys; 1101 if (idx == ARRAY_SIZE(chunk->phys)) { 1102 chunk = new_vmalloc_chunk(chunk); 1103 if (!chunk) { 1104 err = -ENOMEM; 1105 goto err_free; 1106 } 1107 idx = 0; 1108 } 1109 } 1110 1111 preservation->total_pages = vm->nr_pages; 1112 preservation->flags = flags; 1113 preservation->order = order; 1114 1115 return 0; 1116 1117 err_free: 1118 kho_unpreserve_vmalloc(preservation); 1119 return err; 1120 } 1121 EXPORT_SYMBOL_GPL(kho_preserve_vmalloc); 1122 1123 /** 1124 * kho_unpreserve_vmalloc - unpreserve memory allocated with vmalloc() 1125 * @preservation: preservation metadata returned by kho_preserve_vmalloc() 1126 * 1127 * Instructs KHO to unpreserve the area in vmalloc address space that was 1128 * previously preserved with kho_preserve_vmalloc(). 1129 */ 1130 void kho_unpreserve_vmalloc(struct kho_vmalloc *preservation) 1131 { 1132 struct kho_vmalloc_chunk *chunk = KHOSER_LOAD_PTR(preservation->first); 1133 1134 while (chunk) { 1135 struct kho_vmalloc_chunk *tmp = chunk; 1136 1137 kho_vmalloc_unpreserve_chunk(chunk, preservation->order); 1138 1139 chunk = KHOSER_LOAD_PTR(chunk->hdr.next); 1140 free_page((unsigned long)tmp); 1141 } 1142 } 1143 EXPORT_SYMBOL_GPL(kho_unpreserve_vmalloc); 1144 1145 /** 1146 * kho_restore_vmalloc - recreates and populates an area in vmalloc address 1147 * space from the preserved memory. 1148 * @preservation: preservation metadata. 1149 * 1150 * Recreates an area in vmalloc address space and populates it with memory that 1151 * was preserved using kho_preserve_vmalloc(). 1152 * 1153 * Return: pointer to the area in the vmalloc address space, NULL on failure. 1154 */ 1155 void *kho_restore_vmalloc(const struct kho_vmalloc *preservation) 1156 { 1157 struct kho_vmalloc_chunk *chunk = KHOSER_LOAD_PTR(preservation->first); 1158 kasan_vmalloc_flags_t kasan_flags = KASAN_VMALLOC_PROT_NORMAL; 1159 unsigned int align, order, shift, vm_flags; 1160 unsigned long total_pages, contig_pages; 1161 unsigned long addr, size; 1162 struct vm_struct *area; 1163 struct page **pages; 1164 unsigned int idx = 0; 1165 int err; 1166 1167 vm_flags = kho_flags_to_vmalloc(preservation->flags); 1168 if (vm_flags & ~KHO_VMALLOC_SUPPORTED_FLAGS) 1169 return NULL; 1170 1171 total_pages = preservation->total_pages; 1172 pages = kvmalloc_objs(*pages, total_pages); 1173 if (!pages) 1174 return NULL; 1175 order = preservation->order; 1176 contig_pages = (1 << order); 1177 shift = PAGE_SHIFT + order; 1178 align = 1 << shift; 1179 1180 while (chunk) { 1181 struct page *page; 1182 1183 for (int i = 0; i < ARRAY_SIZE(chunk->phys) && chunk->phys[i]; i++) { 1184 phys_addr_t phys = chunk->phys[i]; 1185 1186 if (idx + contig_pages > total_pages) 1187 goto err_free_pages_array; 1188 1189 page = kho_restore_pages(phys, contig_pages); 1190 if (!page) 1191 goto err_free_pages_array; 1192 1193 for (int j = 0; j < contig_pages; j++) 1194 pages[idx++] = page + j; 1195 1196 phys += contig_pages * PAGE_SIZE; 1197 } 1198 1199 page = kho_restore_pages(virt_to_phys(chunk), 1); 1200 if (!page) 1201 goto err_free_pages_array; 1202 chunk = KHOSER_LOAD_PTR(chunk->hdr.next); 1203 __free_page(page); 1204 } 1205 1206 if (idx != total_pages) 1207 goto err_free_pages_array; 1208 1209 area = __get_vm_area_node(total_pages * PAGE_SIZE, align, shift, 1210 vm_flags | VM_UNINITIALIZED, 1211 VMALLOC_START, VMALLOC_END, 1212 NUMA_NO_NODE, GFP_KERNEL, 1213 __builtin_return_address(0)); 1214 if (!area) 1215 goto err_free_pages_array; 1216 1217 addr = (unsigned long)area->addr; 1218 size = get_vm_area_size(area); 1219 err = vmap_pages_range(addr, addr + size, PAGE_KERNEL, pages, shift); 1220 if (err) 1221 goto err_free_vm_area; 1222 1223 area->nr_pages = total_pages; 1224 area->pages = pages; 1225 1226 if (vm_flags & VM_ALLOC) 1227 kasan_flags |= KASAN_VMALLOC_VM_ALLOC; 1228 1229 area->addr = kasan_unpoison_vmalloc(area->addr, total_pages * PAGE_SIZE, 1230 kasan_flags); 1231 clear_vm_uninitialized_flag(area); 1232 1233 return area->addr; 1234 1235 err_free_vm_area: 1236 free_vm_area(area); 1237 err_free_pages_array: 1238 kvfree(pages); 1239 return NULL; 1240 } 1241 EXPORT_SYMBOL_GPL(kho_restore_vmalloc); 1242 1243 /** 1244 * kho_alloc_preserve - Allocate, zero, and preserve memory. 1245 * @size: The number of bytes to allocate. 1246 * 1247 * Allocates a physically contiguous block of zeroed pages that is large 1248 * enough to hold @size bytes. The allocated memory is then registered with 1249 * KHO for preservation across a kexec. 1250 * 1251 * Note: The actual allocated size will be rounded up to the nearest 1252 * power-of-two page boundary. 1253 * 1254 * @return A virtual pointer to the allocated and preserved memory on success, 1255 * or an ERR_PTR() encoded error on failure. 1256 */ 1257 void *kho_alloc_preserve(size_t size) 1258 { 1259 struct folio *folio; 1260 int order, ret; 1261 1262 if (!size) 1263 return ERR_PTR(-EINVAL); 1264 1265 order = get_order(size); 1266 if (order > MAX_PAGE_ORDER) 1267 return ERR_PTR(-E2BIG); 1268 1269 folio = folio_alloc(GFP_KERNEL | __GFP_ZERO, order); 1270 if (!folio) 1271 return ERR_PTR(-ENOMEM); 1272 1273 ret = kho_preserve_folio(folio); 1274 if (ret) { 1275 folio_put(folio); 1276 return ERR_PTR(ret); 1277 } 1278 1279 return folio_address(folio); 1280 } 1281 EXPORT_SYMBOL_GPL(kho_alloc_preserve); 1282 1283 /** 1284 * kho_unpreserve_free - Unpreserve and free memory. 1285 * @mem: Pointer to the memory allocated by kho_alloc_preserve(). 1286 * 1287 * Unregisters the memory from KHO preservation and frees the underlying 1288 * pages back to the system. This function should be called to clean up 1289 * memory allocated with kho_alloc_preserve(). 1290 */ 1291 void kho_unpreserve_free(void *mem) 1292 { 1293 struct folio *folio; 1294 1295 if (!mem) 1296 return; 1297 1298 folio = virt_to_folio(mem); 1299 kho_unpreserve_folio(folio); 1300 folio_put(folio); 1301 } 1302 EXPORT_SYMBOL_GPL(kho_unpreserve_free); 1303 1304 /** 1305 * kho_restore_free - Restore and free memory after kexec. 1306 * @mem: Pointer to the memory (in the new kernel's address space) 1307 * that was allocated by the old kernel. 1308 * 1309 * This function is intended to be called in the new kernel (post-kexec) 1310 * to take ownership of and free a memory region that was preserved by the 1311 * old kernel using kho_alloc_preserve(). 1312 * 1313 * It first restores the pages from KHO (using their physical address) 1314 * and then frees the pages back to the new kernel's page allocator. 1315 */ 1316 void kho_restore_free(void *mem) 1317 { 1318 struct folio *folio; 1319 1320 if (!mem) 1321 return; 1322 1323 folio = kho_restore_folio(__pa(mem)); 1324 if (!WARN_ON(!folio)) 1325 folio_put(folio); 1326 } 1327 EXPORT_SYMBOL_GPL(kho_restore_free); 1328 1329 struct kho_in { 1330 phys_addr_t fdt_phys; 1331 phys_addr_t scratch_phys; 1332 char previous_release[__NEW_UTS_LEN + 1]; 1333 u32 kexec_count; 1334 struct kho_debugfs dbg; 1335 }; 1336 1337 static struct kho_in kho_in = { 1338 }; 1339 1340 static const void *kho_get_fdt(void) 1341 { 1342 return kho_in.fdt_phys ? phys_to_virt(kho_in.fdt_phys) : NULL; 1343 } 1344 1345 /** 1346 * is_kho_boot - check if current kernel was booted via KHO-enabled 1347 * kexec 1348 * 1349 * This function checks if the current kernel was loaded through a kexec 1350 * operation with KHO enabled, by verifying that a valid KHO FDT 1351 * was passed. 1352 * 1353 * Note: This function returns reliable results only after 1354 * kho_populate() has been called during early boot. Before that, 1355 * it may return false even if KHO data is present. 1356 * 1357 * Return: true if booted via KHO-enabled kexec, false otherwise 1358 */ 1359 bool is_kho_boot(void) 1360 { 1361 return !!kho_get_fdt(); 1362 } 1363 EXPORT_SYMBOL_GPL(is_kho_boot); 1364 1365 /** 1366 * kho_retrieve_subtree - retrieve a preserved sub blob by its name. 1367 * @name: the name of the sub blob passed to kho_add_subtree(). 1368 * @phys: if found, the physical address of the sub blob is stored in @phys. 1369 * @size: if not NULL and found, the size of the sub blob is stored in @size. 1370 * 1371 * Retrieve a preserved sub blob named @name and store its physical 1372 * address in @phys and optionally its size in @size. 1373 * 1374 * Return: 0 on success, error code on failure 1375 */ 1376 int kho_retrieve_subtree(const char *name, phys_addr_t *phys, size_t *size) 1377 { 1378 const void *fdt = kho_get_fdt(); 1379 const u64 *val; 1380 int offset, len; 1381 1382 if (!fdt) 1383 return -ENOENT; 1384 1385 if (!phys) 1386 return -EINVAL; 1387 1388 offset = fdt_subnode_offset(fdt, 0, name); 1389 if (offset < 0) 1390 return -ENOENT; 1391 1392 val = fdt_getprop(fdt, offset, KHO_SUB_TREE_PROP_NAME, &len); 1393 if (!val || len != sizeof(*val)) 1394 return -EINVAL; 1395 1396 *phys = (phys_addr_t)*val; 1397 1398 val = fdt_getprop(fdt, offset, KHO_SUB_TREE_SIZE_PROP_NAME, &len); 1399 if (!val || len != sizeof(*val)) { 1400 pr_warn("broken KHO subnode '%s': missing or invalid blob-size property\n", 1401 name); 1402 return -EINVAL; 1403 } 1404 1405 if (size) 1406 *size = (size_t)*val; 1407 1408 return 0; 1409 } 1410 EXPORT_SYMBOL_GPL(kho_retrieve_subtree); 1411 1412 static int __init kho_mem_retrieve(const void *fdt) 1413 { 1414 struct kho_radix_tree tree; 1415 const phys_addr_t *mem; 1416 int len; 1417 1418 /* Retrieve the KHO radix tree from passed-in FDT. */ 1419 mem = fdt_getprop(fdt, 0, KHO_FDT_MEMORY_MAP_PROP_NAME, &len); 1420 1421 if (!mem || len != sizeof(*mem)) { 1422 pr_err("failed to get preserved KHO memory tree\n"); 1423 return -ENOENT; 1424 } 1425 1426 if (!*mem) 1427 return -EINVAL; 1428 1429 tree.root = phys_to_virt(*mem); 1430 mutex_init(&tree.lock); 1431 return kho_radix_walk_tree(&tree, kho_preserved_memory_reserve); 1432 } 1433 1434 static __init int kho_out_fdt_setup(void) 1435 { 1436 struct kho_radix_tree *tree = &kho_out.radix_tree; 1437 void *root = kho_out.fdt; 1438 u64 preserved_mem_tree_pa; 1439 int err; 1440 1441 err = fdt_create(root, PAGE_SIZE); 1442 err |= fdt_finish_reservemap(root); 1443 err |= fdt_begin_node(root, ""); 1444 err |= fdt_property_string(root, "compatible", KHO_FDT_COMPATIBLE); 1445 1446 preserved_mem_tree_pa = virt_to_phys(tree->root); 1447 1448 err |= fdt_property(root, KHO_FDT_MEMORY_MAP_PROP_NAME, 1449 &preserved_mem_tree_pa, 1450 sizeof(preserved_mem_tree_pa)); 1451 1452 err |= fdt_end_node(root); 1453 err |= fdt_finish(root); 1454 1455 return err; 1456 } 1457 1458 static void __init kho_in_kexec_metadata(void) 1459 { 1460 struct kho_kexec_metadata *metadata; 1461 phys_addr_t metadata_phys; 1462 size_t blob_size; 1463 int err; 1464 1465 err = kho_retrieve_subtree(KHO_METADATA_NODE_NAME, &metadata_phys, 1466 &blob_size); 1467 if (err) 1468 /* This is fine, previous kernel didn't export metadata */ 1469 return; 1470 1471 /* Check that, at least, "version" is present */ 1472 if (blob_size < sizeof(u32)) { 1473 pr_warn("kexec-metadata blob too small (%zu bytes)\n", 1474 blob_size); 1475 return; 1476 } 1477 1478 metadata = phys_to_virt(metadata_phys); 1479 1480 if (metadata->version != KHO_KEXEC_METADATA_VERSION) { 1481 pr_warn("kexec-metadata version %u not supported (expected %u)\n", 1482 metadata->version, KHO_KEXEC_METADATA_VERSION); 1483 return; 1484 } 1485 1486 if (blob_size < sizeof(*metadata)) { 1487 pr_warn("kexec-metadata blob too small for v%u (%zu < %zu)\n", 1488 metadata->version, blob_size, sizeof(*metadata)); 1489 return; 1490 } 1491 1492 /* 1493 * Copy data to the kernel structure that will persist during 1494 * kernel lifetime. 1495 */ 1496 kho_in.kexec_count = metadata->kexec_count; 1497 strscpy(kho_in.previous_release, metadata->previous_release, 1498 sizeof(kho_in.previous_release)); 1499 1500 pr_info("exec from: %s (count %u)\n", 1501 kho_in.previous_release, kho_in.kexec_count); 1502 } 1503 1504 /* 1505 * Create kexec metadata to pass kernel version and boot count to the 1506 * next kernel. This keeps the core KHO ABI minimal and allows the 1507 * metadata format to evolve independently. 1508 */ 1509 static __init int kho_out_kexec_metadata(void) 1510 { 1511 struct kho_kexec_metadata *metadata; 1512 int err; 1513 1514 metadata = kho_alloc_preserve(sizeof(*metadata)); 1515 if (IS_ERR(metadata)) 1516 return PTR_ERR(metadata); 1517 1518 metadata->version = KHO_KEXEC_METADATA_VERSION; 1519 strscpy(metadata->previous_release, init_uts_ns.name.release, 1520 sizeof(metadata->previous_release)); 1521 /* kho_in.kexec_count is set to 0 on cold boot */ 1522 metadata->kexec_count = kho_in.kexec_count + 1; 1523 1524 err = kho_add_subtree(KHO_METADATA_NODE_NAME, metadata, 1525 sizeof(*metadata)); 1526 if (err) 1527 kho_unpreserve_free(metadata); 1528 1529 return err; 1530 } 1531 1532 static int __init kho_kexec_metadata_init(const void *fdt) 1533 { 1534 int err; 1535 1536 if (fdt) 1537 kho_in_kexec_metadata(); 1538 1539 /* Populate kexec metadata for the possible next kexec */ 1540 err = kho_out_kexec_metadata(); 1541 if (err) 1542 pr_warn("failed to initialize kexec-metadata subtree: %d\n", 1543 err); 1544 1545 return err; 1546 } 1547 1548 static __init int kho_init(void) 1549 { 1550 struct kho_radix_tree *tree = &kho_out.radix_tree; 1551 const void *fdt = kho_get_fdt(); 1552 int err = 0; 1553 1554 if (!kho_enable) 1555 return 0; 1556 1557 tree->root = kzalloc(PAGE_SIZE, GFP_KERNEL); 1558 if (!tree->root) { 1559 err = -ENOMEM; 1560 goto err_free_scratch; 1561 } 1562 1563 kho_out.fdt = kho_alloc_preserve(PAGE_SIZE); 1564 if (IS_ERR(kho_out.fdt)) { 1565 err = PTR_ERR(kho_out.fdt); 1566 goto err_free_kho_radix_tree_root; 1567 } 1568 1569 err = kho_debugfs_init(); 1570 if (err) 1571 goto err_free_fdt; 1572 1573 err = kho_out_debugfs_init(&kho_out.dbg); 1574 if (err) 1575 goto err_free_fdt; 1576 1577 err = kho_out_fdt_setup(); 1578 if (err) 1579 goto err_free_fdt; 1580 1581 err = kho_kexec_metadata_init(fdt); 1582 if (err) 1583 goto err_free_fdt; 1584 1585 if (fdt) { 1586 kho_in_debugfs_init(&kho_in.dbg, fdt); 1587 return 0; 1588 } 1589 1590 for (int i = 0; i < kho_scratch_cnt; i++) { 1591 unsigned long base_pfn = PHYS_PFN(kho_scratch[i].addr); 1592 unsigned long count = kho_scratch[i].size >> PAGE_SHIFT; 1593 unsigned long pfn; 1594 1595 /* 1596 * When debug_pagealloc is enabled, __free_pages() clears the 1597 * corresponding PRESENT bit in the kernel page table. 1598 * Subsequent kmemleak scans of these pages cause the 1599 * non-PRESENT page faults. 1600 * Mark scratch areas with kmemleak_ignore_phys() to exclude 1601 * them from kmemleak scanning. 1602 */ 1603 kmemleak_ignore_phys(kho_scratch[i].addr); 1604 for (pfn = base_pfn; pfn < base_pfn + count; 1605 pfn += pageblock_nr_pages) 1606 init_cma_reserved_pageblock(pfn_to_page(pfn)); 1607 } 1608 1609 WARN_ON_ONCE(kho_debugfs_blob_add(&kho_out.dbg, "fdt", 1610 kho_out.fdt, 1611 fdt_totalsize(kho_out.fdt), true)); 1612 1613 return 0; 1614 1615 err_free_fdt: 1616 kho_unpreserve_free(kho_out.fdt); 1617 err_free_kho_radix_tree_root: 1618 kfree(tree->root); 1619 tree->root = NULL; 1620 err_free_scratch: 1621 kho_out.fdt = NULL; 1622 for (int i = 0; i < kho_scratch_cnt; i++) { 1623 void *start = __va(kho_scratch[i].addr); 1624 void *end = start + kho_scratch[i].size; 1625 1626 free_reserved_area(start, end, -1, ""); 1627 } 1628 kho_enable = false; 1629 return err; 1630 } 1631 fs_initcall(kho_init); 1632 1633 void __init kho_memory_init(void) 1634 { 1635 if (kho_in.scratch_phys) { 1636 kho_scratch = phys_to_virt(kho_in.scratch_phys); 1637 1638 if (kho_mem_retrieve(kho_get_fdt())) 1639 kho_in.fdt_phys = 0; 1640 } else { 1641 kho_reserve_scratch(); 1642 } 1643 } 1644 1645 void __init kho_populate(phys_addr_t fdt_phys, u64 fdt_len, 1646 phys_addr_t scratch_phys, u64 scratch_len) 1647 { 1648 unsigned int scratch_cnt = scratch_len / sizeof(*kho_scratch); 1649 struct kho_scratch *scratch = NULL; 1650 phys_addr_t mem_map_phys; 1651 void *fdt = NULL; 1652 bool populated = false; 1653 int err; 1654 1655 /* Validate the input FDT */ 1656 fdt = early_memremap(fdt_phys, fdt_len); 1657 if (!fdt) { 1658 pr_warn("setup: failed to memremap FDT (0x%llx)\n", fdt_phys); 1659 goto report; 1660 } 1661 err = fdt_check_header(fdt); 1662 if (err) { 1663 pr_warn("setup: handover FDT (0x%llx) is invalid: %d\n", 1664 fdt_phys, err); 1665 goto unmap_fdt; 1666 } 1667 err = fdt_node_check_compatible(fdt, 0, KHO_FDT_COMPATIBLE); 1668 if (err) { 1669 pr_warn("setup: handover FDT (0x%llx) is incompatible with '%s': %d\n", 1670 fdt_phys, KHO_FDT_COMPATIBLE, err); 1671 goto unmap_fdt; 1672 } 1673 1674 mem_map_phys = kho_get_mem_map_phys(fdt); 1675 if (!mem_map_phys) 1676 goto unmap_fdt; 1677 1678 scratch = early_memremap(scratch_phys, scratch_len); 1679 if (!scratch) { 1680 pr_warn("setup: failed to memremap scratch (phys=0x%llx, len=%lld)\n", 1681 scratch_phys, scratch_len); 1682 goto unmap_fdt; 1683 } 1684 1685 /* 1686 * We pass a safe contiguous blocks of memory to use for early boot 1687 * purporses from the previous kernel so that we can resize the 1688 * memblock array as needed. 1689 */ 1690 for (int i = 0; i < scratch_cnt; i++) { 1691 struct kho_scratch *area = &scratch[i]; 1692 u64 size = area->size; 1693 1694 memblock_add(area->addr, size); 1695 err = memblock_mark_kho_scratch(area->addr, size); 1696 if (err) { 1697 pr_warn("failed to mark the scratch region 0x%pa+0x%pa: %pe", 1698 &area->addr, &size, ERR_PTR(err)); 1699 goto unmap_scratch; 1700 } 1701 pr_debug("Marked 0x%pa+0x%pa as scratch", &area->addr, &size); 1702 } 1703 1704 memblock_reserve(scratch_phys, scratch_len); 1705 1706 /* 1707 * Now that we have a viable region of scratch memory, let's tell 1708 * the memblocks allocator to only use that for any allocations. 1709 * That way we ensure that nothing scribbles over in use data while 1710 * we initialize the page tables which we will need to ingest all 1711 * memory reservations from the previous kernel. 1712 */ 1713 memblock_set_kho_scratch_only(); 1714 1715 kho_in.fdt_phys = fdt_phys; 1716 kho_in.scratch_phys = scratch_phys; 1717 kho_scratch_cnt = scratch_cnt; 1718 1719 populated = true; 1720 pr_info("found kexec handover data.\n"); 1721 1722 unmap_scratch: 1723 early_memunmap(scratch, scratch_len); 1724 unmap_fdt: 1725 early_memunmap(fdt, fdt_len); 1726 report: 1727 if (!populated) 1728 pr_warn("disabling KHO revival\n"); 1729 } 1730 1731 /* Helper functions for kexec_file_load */ 1732 1733 int kho_fill_kimage(struct kimage *image) 1734 { 1735 ssize_t scratch_size; 1736 int err = 0; 1737 struct kexec_buf scratch; 1738 1739 if (!kho_enable || image->type == KEXEC_TYPE_CRASH) 1740 return 0; 1741 1742 image->kho.fdt = virt_to_phys(kho_out.fdt); 1743 1744 scratch_size = sizeof(*kho_scratch) * kho_scratch_cnt; 1745 scratch = (struct kexec_buf){ 1746 .image = image, 1747 .buffer = kho_scratch, 1748 .bufsz = scratch_size, 1749 .mem = KEXEC_BUF_MEM_UNKNOWN, 1750 .memsz = scratch_size, 1751 .buf_align = SZ_64K, /* Makes it easier to map */ 1752 .buf_max = ULONG_MAX, 1753 .top_down = true, 1754 }; 1755 err = kexec_add_buffer(&scratch); 1756 if (err) 1757 return err; 1758 image->kho.scratch = &image->segment[image->nr_segments - 1]; 1759 1760 return 0; 1761 } 1762 1763 static int kho_walk_scratch(struct kexec_buf *kbuf, 1764 int (*func)(struct resource *, void *)) 1765 { 1766 int ret = 0; 1767 int i; 1768 1769 for (i = 0; i < kho_scratch_cnt; i++) { 1770 struct resource res = { 1771 .start = kho_scratch[i].addr, 1772 .end = kho_scratch[i].addr + kho_scratch[i].size - 1, 1773 }; 1774 1775 /* Try to fit the kimage into our KHO scratch region */ 1776 ret = func(&res, kbuf); 1777 if (ret) 1778 break; 1779 } 1780 1781 return ret; 1782 } 1783 1784 int kho_locate_mem_hole(struct kexec_buf *kbuf, 1785 int (*func)(struct resource *, void *)) 1786 { 1787 int ret; 1788 1789 if (!kho_enable || kbuf->image->type == KEXEC_TYPE_CRASH) 1790 return 1; 1791 1792 ret = kho_walk_scratch(kbuf, func); 1793 1794 return ret == 1 ? 0 : -EADDRNOTAVAIL; 1795 } 1796