1 // SPDX-License-Identifier: GPL-2.0-only 2 /* 3 * AMD SVM-SEV Host Support. 4 * 5 * Copyright (C) 2023 Advanced Micro Devices, Inc. 6 * 7 * Author: Ashish Kalra <ashish.kalra@amd.com> 8 * 9 */ 10 11 #include <linux/cc_platform.h> 12 #include <linux/printk.h> 13 #include <linux/mm_types.h> 14 #include <linux/set_memory.h> 15 #include <linux/memblock.h> 16 #include <linux/kernel.h> 17 #include <linux/mm.h> 18 #include <linux/cpumask.h> 19 #include <linux/iommu.h> 20 #include <linux/amd-iommu.h> 21 #include <linux/nospec.h> 22 23 #include <asm/sev.h> 24 #include <asm/processor.h> 25 #include <asm/setup.h> 26 #include <asm/svm.h> 27 #include <asm/smp.h> 28 #include <asm/cpu.h> 29 #include <asm/apic.h> 30 #include <asm/cpuid/api.h> 31 #include <asm/cmdline.h> 32 #include <asm/iommu.h> 33 #include <asm/msr.h> 34 35 /* 36 * The RMP entry information as returned by the RMPREAD instruction. 37 */ 38 struct rmpentry { 39 u64 gpa; 40 u8 assigned :1, 41 rsvd1 :7; 42 u8 pagesize :1, 43 hpage_region_status :1, 44 rsvd2 :6; 45 u8 immutable :1, 46 rsvd3 :7; 47 u8 rsvd4; 48 u32 asid; 49 } __packed; 50 51 /* 52 * The raw RMP entry format is not architectural. The format is defined in PPR 53 * Family 19h Model 01h, Rev B1 processor. This format represents the actual 54 * entry in the RMP table memory. The bitfield definitions are used for machines 55 * without the RMPREAD instruction (Zen3 and Zen4), otherwise the "hi" and "lo" 56 * fields are only used for dumping the raw data. 57 */ 58 struct rmpentry_raw { 59 union { 60 struct { 61 u64 assigned : 1, 62 pagesize : 1, 63 immutable : 1, 64 rsvd1 : 9, 65 gpa : 39, 66 asid : 10, 67 vmsa : 1, 68 validated : 1, 69 rsvd2 : 1; 70 }; 71 u64 lo; 72 }; 73 u64 hi; 74 } __packed; 75 76 /* 77 * The first 16KB from the RMP_BASE is used by the processor for the 78 * bookkeeping, the range needs to be added during the RMP entry lookup. 79 */ 80 #define RMPTABLE_CPU_BOOKKEEPING_SZ 0x4000 81 82 /* 83 * For a non-segmented RMP table, use the maximum physical addressing as the 84 * segment size in order to always arrive at index 0 in the table. 85 */ 86 #define RMPTABLE_NON_SEGMENTED_SHIFT 52 87 88 struct rmp_segment_desc { 89 struct rmpentry_raw *rmp_entry; 90 u64 max_index; 91 u64 size; 92 }; 93 94 /* 95 * Segmented RMP Table support. 96 * - The segment size is used for two purposes: 97 * - Identify the amount of memory covered by an RMP segment 98 * - Quickly locate an RMP segment table entry for a physical address 99 * 100 * - The RMP segment table contains pointers to an RMP table that covers 101 * a specific portion of memory. There can be up to 512 8-byte entries, 102 * one pages worth. 103 */ 104 #define RST_ENTRY_MAPPED_SIZE(x) ((x) & GENMASK_ULL(19, 0)) 105 #define RST_ENTRY_SEGMENT_BASE(x) ((x) & GENMASK_ULL(51, 20)) 106 107 #define RST_SIZE SZ_4K 108 static struct rmp_segment_desc **rmp_segment_table __ro_after_init; 109 static unsigned int rst_max_index __ro_after_init = 512; 110 111 static unsigned int rmp_segment_shift; 112 static u64 rmp_segment_size; 113 static u64 rmp_segment_mask; 114 115 #define RST_ENTRY_INDEX(x) ((x) >> rmp_segment_shift) 116 #define RMP_ENTRY_INDEX(x) ((u64)(PHYS_PFN((x) & rmp_segment_mask))) 117 118 static u64 rmp_cfg; 119 120 static void *rmp_bookkeeping __ro_after_init; 121 122 /* Mask to apply to a PFN to get the first PFN of a 2MB page */ 123 #define PFN_PMD_MASK GENMASK_ULL(63, PMD_SHIFT - PAGE_SHIFT) 124 125 static u64 probed_rmp_base, probed_rmp_size; 126 127 static LIST_HEAD(snp_leaked_pages_list); 128 static DEFINE_SPINLOCK(snp_leaked_pages_list_lock); 129 130 static unsigned long snp_nr_leaked_pages; 131 132 #undef pr_fmt 133 #define pr_fmt(fmt) "SEV-SNP: " fmt 134 135 static void mfd_reconfigure(void *arg) 136 { 137 if (!cc_platform_has(CC_ATTR_HOST_SEV_SNP)) 138 return; 139 140 if (arg) 141 msr_set_bit(MSR_AMD64_SYSCFG, MSR_AMD64_SYSCFG_MFDM_BIT); 142 else 143 msr_clear_bit(MSR_AMD64_SYSCFG, MSR_AMD64_SYSCFG_MFDM_BIT); 144 } 145 146 static void snp_enable(void *arg) 147 { 148 u64 val; 149 150 if (!cc_platform_has(CC_ATTR_HOST_SEV_SNP)) 151 return; 152 153 rdmsrq(MSR_AMD64_SYSCFG, val); 154 155 val |= MSR_AMD64_SYSCFG_SNP_EN; 156 val |= MSR_AMD64_SYSCFG_SNP_VMPL_EN; 157 158 wrmsrq(MSR_AMD64_SYSCFG, val); 159 } 160 161 static void __init __snp_fixup_e820_tables(u64 pa) 162 { 163 if (IS_ALIGNED(pa, PMD_SIZE)) 164 return; 165 166 /* 167 * Handle cases where the RMP table placement by the BIOS is not 168 * 2M aligned and the kexec kernel could try to allocate 169 * from within that chunk which then causes a fatal RMP fault. 170 * 171 * The e820_table needs to be updated as it is converted to 172 * kernel memory resources and used by KEXEC_FILE_LOAD syscall 173 * to load kexec segments. 174 * 175 * The e820_table_firmware needs to be updated as it is exposed 176 * to sysfs and used by the KEXEC_LOAD syscall to load kexec 177 * segments. 178 * 179 * The e820_table_kexec needs to be updated as it passed to 180 * the kexec-ed kernel. 181 */ 182 pa = ALIGN_DOWN(pa, PMD_SIZE); 183 if (e820__mapped_any(pa, pa + PMD_SIZE, E820_TYPE_RAM)) { 184 pr_info("Reserving start/end of RMP table on a 2MB boundary [0x%016llx]\n", pa); 185 e820__range_update(pa, PMD_SIZE, E820_TYPE_RAM, E820_TYPE_RESERVED); 186 e820__range_update_table(e820_table_kexec, pa, PMD_SIZE, E820_TYPE_RAM, E820_TYPE_RESERVED); 187 if (!memblock_is_region_reserved(pa, PMD_SIZE)) 188 memblock_reserve(pa, PMD_SIZE); 189 } 190 } 191 192 static void __init fixup_e820_tables_for_segmented_rmp(void) 193 { 194 u64 pa, *rst, size, mapped_size; 195 unsigned int i; 196 197 __snp_fixup_e820_tables(probed_rmp_base); 198 199 pa = probed_rmp_base + RMPTABLE_CPU_BOOKKEEPING_SZ; 200 201 __snp_fixup_e820_tables(pa + RST_SIZE); 202 203 rst = early_memremap(pa, RST_SIZE); 204 if (!rst) 205 return; 206 207 for (i = 0; i < rst_max_index; i++) { 208 pa = RST_ENTRY_SEGMENT_BASE(rst[i]); 209 mapped_size = RST_ENTRY_MAPPED_SIZE(rst[i]); 210 if (!mapped_size) 211 continue; 212 213 __snp_fixup_e820_tables(pa); 214 215 /* 216 * Mapped size in GB. Mapped size is allowed to exceed 217 * the segment coverage size, but gets reduced to the 218 * segment coverage size. 219 */ 220 mapped_size <<= 30; 221 if (mapped_size > rmp_segment_size) 222 mapped_size = rmp_segment_size; 223 224 /* Calculate the RMP segment size (16 bytes/page mapped) */ 225 size = PHYS_PFN(mapped_size) << 4; 226 227 __snp_fixup_e820_tables(pa + size); 228 } 229 230 early_memunmap(rst, RST_SIZE); 231 } 232 233 static void __init fixup_e820_tables_for_contiguous_rmp(void) 234 { 235 __snp_fixup_e820_tables(probed_rmp_base); 236 __snp_fixup_e820_tables(probed_rmp_base + probed_rmp_size); 237 } 238 239 void __init snp_fixup_e820_tables(void) 240 { 241 if (rmp_cfg & MSR_AMD64_SEG_RMP_ENABLED) { 242 fixup_e820_tables_for_segmented_rmp(); 243 } else { 244 fixup_e820_tables_for_contiguous_rmp(); 245 } 246 } 247 248 static void clear_rmp(void) 249 { 250 unsigned int i; 251 u64 val; 252 253 if (!cc_platform_has(CC_ATTR_HOST_SEV_SNP)) 254 return; 255 256 /* Clearing the RMP while SNP is enabled will cause an exception */ 257 rdmsrq(MSR_AMD64_SYSCFG, val); 258 if (WARN_ON_ONCE(val & MSR_AMD64_SYSCFG_SNP_EN)) 259 return; 260 261 memset(rmp_bookkeeping, 0, RMPTABLE_CPU_BOOKKEEPING_SZ); 262 263 for (i = 0; i < rst_max_index; i++) { 264 struct rmp_segment_desc *desc; 265 266 desc = rmp_segment_table[i]; 267 if (!desc) 268 continue; 269 270 memset(desc->rmp_entry, 0, desc->size); 271 } 272 } 273 274 static bool __init alloc_rmp_segment_desc(u64 segment_pa, u64 segment_size, u64 pa) 275 { 276 u64 rst_index, rmp_segment_size_max; 277 struct rmp_segment_desc *desc; 278 void *rmp_segment; 279 280 /* Calculate the maximum size an RMP can be (16 bytes/page mapped) */ 281 rmp_segment_size_max = PHYS_PFN(rmp_segment_size) << 4; 282 283 /* Validate the RMP segment size */ 284 if (segment_size > rmp_segment_size_max) { 285 pr_err("Invalid RMP size 0x%llx for configured segment size 0x%llx\n", 286 segment_size, rmp_segment_size_max); 287 return false; 288 } 289 290 /* Validate the RMP segment table index */ 291 rst_index = RST_ENTRY_INDEX(pa); 292 if (rst_index >= rst_max_index) { 293 pr_err("Invalid RMP segment base address 0x%llx for configured segment size 0x%llx\n", 294 pa, rmp_segment_size); 295 return false; 296 } 297 298 if (rmp_segment_table[rst_index]) { 299 pr_err("RMP segment descriptor already exists at index %llu\n", rst_index); 300 return false; 301 } 302 303 rmp_segment = memremap(segment_pa, segment_size, MEMREMAP_WB); 304 if (!rmp_segment) { 305 pr_err("Failed to map RMP segment addr 0x%llx size 0x%llx\n", 306 segment_pa, segment_size); 307 return false; 308 } 309 310 desc = kzalloc_obj(*desc); 311 if (!desc) { 312 memunmap(rmp_segment); 313 return false; 314 } 315 316 desc->rmp_entry = rmp_segment; 317 desc->max_index = segment_size / sizeof(*desc->rmp_entry); 318 desc->size = segment_size; 319 320 rmp_segment_table[rst_index] = desc; 321 322 return true; 323 } 324 325 static void __init free_rmp_segment_table(void) 326 { 327 unsigned int i; 328 329 for (i = 0; i < rst_max_index; i++) { 330 struct rmp_segment_desc *desc; 331 332 desc = rmp_segment_table[i]; 333 if (!desc) 334 continue; 335 336 memunmap(desc->rmp_entry); 337 338 kfree(desc); 339 } 340 341 free_page((unsigned long)rmp_segment_table); 342 343 rmp_segment_table = NULL; 344 } 345 346 /* Allocate the table used to index into the RMP segments */ 347 static bool __init alloc_rmp_segment_table(void) 348 { 349 struct page *page; 350 351 page = alloc_page(__GFP_ZERO); 352 if (!page) 353 return false; 354 355 rmp_segment_table = page_address(page); 356 357 return true; 358 } 359 360 static bool __init setup_contiguous_rmptable(void) 361 { 362 u64 max_rmp_pfn, calc_rmp_sz, rmptable_segment, rmptable_size, rmp_end; 363 364 if (!probed_rmp_size) 365 return false; 366 367 rmp_end = probed_rmp_base + probed_rmp_size - 1; 368 369 /* 370 * Calculate the amount of memory that must be reserved by the BIOS to 371 * address the whole RAM, including the bookkeeping area. The RMP itself 372 * must also be covered. 373 */ 374 max_rmp_pfn = max_pfn; 375 if (PFN_UP(rmp_end) > max_pfn) 376 max_rmp_pfn = PFN_UP(rmp_end); 377 378 calc_rmp_sz = (max_rmp_pfn << 4) + RMPTABLE_CPU_BOOKKEEPING_SZ; 379 if (calc_rmp_sz > probed_rmp_size) { 380 pr_err("Memory reserved for the RMP table does not cover full system RAM (expected 0x%llx got 0x%llx)\n", 381 calc_rmp_sz, probed_rmp_size); 382 return false; 383 } 384 385 if (!alloc_rmp_segment_table()) 386 return false; 387 388 /* Map only the RMP entries */ 389 rmptable_segment = probed_rmp_base + RMPTABLE_CPU_BOOKKEEPING_SZ; 390 rmptable_size = probed_rmp_size - RMPTABLE_CPU_BOOKKEEPING_SZ; 391 392 if (!alloc_rmp_segment_desc(rmptable_segment, rmptable_size, 0)) { 393 free_rmp_segment_table(); 394 return false; 395 } 396 397 return true; 398 } 399 400 static bool __init setup_segmented_rmptable(void) 401 { 402 u64 rst_pa, *rst, pa, ram_pa_end, ram_pa_max; 403 unsigned int i, max_index; 404 405 if (!probed_rmp_base) 406 return false; 407 408 if (!alloc_rmp_segment_table()) 409 return false; 410 411 rst_pa = probed_rmp_base + RMPTABLE_CPU_BOOKKEEPING_SZ; 412 rst = memremap(rst_pa, RST_SIZE, MEMREMAP_WB); 413 if (!rst) { 414 pr_err("Failed to map RMP segment table addr 0x%llx\n", rst_pa); 415 goto e_free; 416 } 417 418 pr_info("Segmented RMP using %lluGB segments\n", rmp_segment_size >> 30); 419 420 ram_pa_max = max_pfn << PAGE_SHIFT; 421 422 max_index = 0; 423 ram_pa_end = 0; 424 for (i = 0; i < rst_max_index; i++) { 425 u64 rmp_segment, rmp_size, mapped_size; 426 427 mapped_size = RST_ENTRY_MAPPED_SIZE(rst[i]); 428 if (!mapped_size) 429 continue; 430 431 max_index = i; 432 433 /* 434 * Mapped size in GB. Mapped size is allowed to exceed the 435 * segment coverage size, but gets reduced to the segment 436 * coverage size. 437 */ 438 mapped_size <<= 30; 439 if (mapped_size > rmp_segment_size) { 440 pr_info("RMP segment %u mapped size (0x%llx) reduced to 0x%llx\n", 441 i, mapped_size, rmp_segment_size); 442 mapped_size = rmp_segment_size; 443 } 444 445 rmp_segment = RST_ENTRY_SEGMENT_BASE(rst[i]); 446 447 /* Calculate the RMP segment size (16 bytes/page mapped) */ 448 rmp_size = PHYS_PFN(mapped_size) << 4; 449 450 pa = (u64)i << rmp_segment_shift; 451 452 /* 453 * Some segments may be for MMIO mapped above system RAM. These 454 * segments are used for Trusted I/O. 455 */ 456 if (pa < ram_pa_max) 457 ram_pa_end = pa + mapped_size; 458 459 if (!alloc_rmp_segment_desc(rmp_segment, rmp_size, pa)) 460 goto e_unmap; 461 462 pr_info("RMP segment %u physical address [0x%llx - 0x%llx] covering [0x%llx - 0x%llx]\n", 463 i, rmp_segment, rmp_segment + rmp_size - 1, pa, pa + mapped_size - 1); 464 } 465 466 if (ram_pa_max > ram_pa_end) { 467 pr_err("Segmented RMP does not cover full system RAM (expected 0x%llx got 0x%llx)\n", 468 ram_pa_max, ram_pa_end); 469 goto e_unmap; 470 } 471 472 /* Adjust the maximum index based on the found segments */ 473 rst_max_index = max_index + 1; 474 475 memunmap(rst); 476 477 return true; 478 479 e_unmap: 480 memunmap(rst); 481 482 e_free: 483 free_rmp_segment_table(); 484 485 return false; 486 } 487 488 static bool __init setup_rmptable(void) 489 { 490 if (rmp_cfg & MSR_AMD64_SEG_RMP_ENABLED) { 491 if (!setup_segmented_rmptable()) 492 return false; 493 } else { 494 if (!setup_contiguous_rmptable()) 495 return false; 496 } 497 498 rmp_bookkeeping = memremap(probed_rmp_base, RMPTABLE_CPU_BOOKKEEPING_SZ, MEMREMAP_WB); 499 if (!rmp_bookkeeping) { 500 pr_err("Failed to map RMP bookkeeping area\n"); 501 free_rmp_segment_table(); 502 503 return false; 504 } 505 506 return true; 507 } 508 509 static void clear_hsave_pa(void *arg) 510 { 511 wrmsrq(MSR_VM_HSAVE_PA, 0); 512 } 513 514 int snp_prepare(void) 515 { 516 int ret; 517 u64 val; 518 519 /* 520 * Check if SEV-SNP is already enabled, this can happen in case of 521 * kexec boot. 522 */ 523 rdmsrq(MSR_AMD64_SYSCFG, val); 524 if (val & MSR_AMD64_SYSCFG_SNP_EN) 525 return 0; 526 527 clear_rmp(); 528 529 cpus_read_lock(); 530 531 if (!cpumask_equal(cpu_online_mask, cpu_present_mask)) { 532 ret = -EOPNOTSUPP; 533 pr_warn("SNP init failed: not all CPUs online. (%*pbl online <-> %*pbl present masks).\n", 534 cpumask_pr_args(cpu_online_mask), 535 cpumask_pr_args(cpu_present_mask)); 536 goto unlock; 537 } 538 539 wbinvd_on_all_cpus(); 540 541 /* 542 * MtrrFixDramModEn is not shared between threads on a core, 543 * therefore it must be set on all CPUs prior to enabling SNP. 544 */ 545 on_each_cpu(mfd_reconfigure, (void *)1, 1); 546 on_each_cpu(snp_enable, NULL, 1); 547 548 /* SNP_INIT requires MSR_VM_HSAVE_PA to be cleared on all CPUs. */ 549 on_each_cpu(clear_hsave_pa, NULL, 1); 550 551 ret = 0; 552 553 unlock: 554 cpus_read_unlock(); 555 556 return ret; 557 } 558 EXPORT_SYMBOL_FOR_MODULES(snp_prepare, "ccp"); 559 560 void snp_shutdown(void) 561 { 562 u64 syscfg; 563 564 rdmsrq(MSR_AMD64_SYSCFG, syscfg); 565 if (syscfg & MSR_AMD64_SYSCFG_SNP_EN) 566 return; 567 568 clear_rmp(); 569 on_each_cpu(mfd_reconfigure, NULL, 1); 570 } 571 EXPORT_SYMBOL_FOR_MODULES(snp_shutdown, "ccp"); 572 573 /* 574 * Do the necessary preparations which are verified by the firmware as 575 * described in the SNP_INIT_EX firmware command description in the SNP 576 * firmware ABI spec. 577 */ 578 int __init snp_rmptable_init(void) 579 { 580 if (WARN_ON_ONCE(!cc_platform_has(CC_ATTR_HOST_SEV_SNP))) 581 return -ENOSYS; 582 583 if (WARN_ON_ONCE(!amd_iommu_snp_en)) 584 return -ENOSYS; 585 586 if (!setup_rmptable()) 587 return -ENOSYS; 588 589 /* 590 * Setting crash_kexec_post_notifiers to 'true' to ensure that SNP panic 591 * notifier is invoked to do SNP IOMMU shutdown before kdump. 592 */ 593 crash_kexec_post_notifiers = true; 594 595 return 0; 596 } 597 598 static void set_rmp_segment_info(unsigned int segment_shift) 599 { 600 rmp_segment_shift = segment_shift; 601 rmp_segment_size = 1ULL << rmp_segment_shift; 602 rmp_segment_mask = rmp_segment_size - 1; 603 } 604 605 #define RMP_ADDR_MASK GENMASK_ULL(51, 13) 606 607 static bool probe_contiguous_rmptable_info(void) 608 { 609 u64 rmp_sz, rmp_base, rmp_end; 610 611 rdmsrq(MSR_AMD64_RMP_BASE, rmp_base); 612 rdmsrq(MSR_AMD64_RMP_END, rmp_end); 613 614 if (!(rmp_base & RMP_ADDR_MASK) || !(rmp_end & RMP_ADDR_MASK)) { 615 pr_err("Memory for the RMP table has not been reserved by BIOS\n"); 616 return false; 617 } 618 619 if (rmp_base > rmp_end) { 620 pr_err("RMP configuration not valid: base=%#llx, end=%#llx\n", rmp_base, rmp_end); 621 return false; 622 } 623 624 rmp_sz = rmp_end - rmp_base + 1; 625 626 /* Treat the contiguous RMP table as a single segment */ 627 rst_max_index = 1; 628 629 set_rmp_segment_info(RMPTABLE_NON_SEGMENTED_SHIFT); 630 631 probed_rmp_base = rmp_base; 632 probed_rmp_size = rmp_sz; 633 634 pr_info("RMP table physical range [0x%016llx - 0x%016llx]\n", 635 rmp_base, rmp_end); 636 637 return true; 638 } 639 640 static bool probe_segmented_rmptable_info(void) 641 { 642 unsigned int eax, ebx, segment_shift, segment_shift_min, segment_shift_max; 643 u64 rmp_base, rmp_end; 644 645 rdmsrq(MSR_AMD64_RMP_BASE, rmp_base); 646 if (!(rmp_base & RMP_ADDR_MASK)) { 647 pr_err("Memory for the RMP table has not been reserved by BIOS\n"); 648 return false; 649 } 650 651 rdmsrq(MSR_AMD64_RMP_END, rmp_end); 652 WARN_ONCE(rmp_end & RMP_ADDR_MASK, 653 "Segmented RMP enabled but RMP_END MSR is non-zero\n"); 654 655 /* Obtain the min and max supported RMP segment size */ 656 eax = cpuid_eax(0x80000025); 657 segment_shift_min = eax & GENMASK(5, 0); 658 segment_shift_max = (eax & GENMASK(11, 6)) >> 6; 659 660 /* Verify the segment size is within the supported limits */ 661 segment_shift = MSR_AMD64_RMP_SEGMENT_SHIFT(rmp_cfg); 662 if (segment_shift > segment_shift_max || segment_shift < segment_shift_min) { 663 pr_err("RMP segment size (%u) is not within advertised bounds (min=%u, max=%u)\n", 664 segment_shift, segment_shift_min, segment_shift_max); 665 return false; 666 } 667 668 /* Override the max supported RST index if a hardware limit exists */ 669 ebx = cpuid_ebx(0x80000025); 670 if (ebx & BIT(10)) 671 rst_max_index = ebx & GENMASK(9, 0); 672 673 set_rmp_segment_info(segment_shift); 674 675 probed_rmp_base = rmp_base; 676 probed_rmp_size = 0; 677 678 pr_info("Segmented RMP base table physical range [0x%016llx - 0x%016llx]\n", 679 rmp_base, rmp_base + RMPTABLE_CPU_BOOKKEEPING_SZ + RST_SIZE); 680 681 return true; 682 } 683 684 bool snp_probe_rmptable_info(void) 685 { 686 if (cpu_feature_enabled(X86_FEATURE_SEGMENTED_RMP)) 687 rdmsrq(MSR_AMD64_RMP_CFG, rmp_cfg); 688 689 if (rmp_cfg & MSR_AMD64_SEG_RMP_ENABLED) 690 return probe_segmented_rmptable_info(); 691 else 692 return probe_contiguous_rmptable_info(); 693 } 694 695 /* 696 * About the array_index_nospec() usage below: 697 * 698 * This function can get called by exported functions like 699 * snp_lookup_rmpentry(), which is used by the KVM #PF handler, among 700 * others, and since the @pfn passed in cannot always be trusted, 701 * speculation should be stopped as a protective measure. 702 */ 703 static struct rmpentry_raw *get_raw_rmpentry(u64 pfn) 704 { 705 u64 paddr, rst_index, segment_index; 706 struct rmp_segment_desc *desc; 707 708 if (!rmp_segment_table) 709 return ERR_PTR(-ENODEV); 710 711 paddr = pfn << PAGE_SHIFT; 712 713 rst_index = RST_ENTRY_INDEX(paddr); 714 if (unlikely(rst_index >= rst_max_index)) 715 return ERR_PTR(-EFAULT); 716 717 rst_index = array_index_nospec(rst_index, rst_max_index); 718 719 desc = rmp_segment_table[rst_index]; 720 if (unlikely(!desc)) 721 return ERR_PTR(-EFAULT); 722 723 segment_index = RMP_ENTRY_INDEX(paddr); 724 if (unlikely(segment_index >= desc->max_index)) 725 return ERR_PTR(-EFAULT); 726 727 segment_index = array_index_nospec(segment_index, desc->max_index); 728 729 return desc->rmp_entry + segment_index; 730 } 731 732 static int get_rmpentry(u64 pfn, struct rmpentry *e) 733 { 734 struct rmpentry_raw *e_raw; 735 736 if (cpu_feature_enabled(X86_FEATURE_RMPREAD)) { 737 int ret; 738 739 /* Binutils version 2.44 supports the RMPREAD mnemonic. */ 740 asm volatile(".byte 0xf2, 0x0f, 0x01, 0xfd" 741 : "=a" (ret) 742 : "a" (pfn << PAGE_SHIFT), "c" (e) 743 : "memory", "cc"); 744 745 return ret; 746 } 747 748 e_raw = get_raw_rmpentry(pfn); 749 if (IS_ERR(e_raw)) 750 return PTR_ERR(e_raw); 751 752 /* 753 * Map the raw RMP table entry onto the RMPREAD output format. 754 * The 2MB region status indicator (hpage_region_status field) is not 755 * calculated, since the overhead could be significant and the field 756 * is not used. 757 */ 758 memset(e, 0, sizeof(*e)); 759 e->gpa = e_raw->gpa << PAGE_SHIFT; 760 e->asid = e_raw->asid; 761 e->assigned = e_raw->assigned; 762 e->pagesize = e_raw->pagesize; 763 e->immutable = e_raw->immutable; 764 765 return 0; 766 } 767 768 static int __snp_lookup_rmpentry(u64 pfn, struct rmpentry *e, int *level) 769 { 770 struct rmpentry e_large; 771 int ret; 772 773 if (!cc_platform_has(CC_ATTR_HOST_SEV_SNP)) 774 return -ENODEV; 775 776 ret = get_rmpentry(pfn, e); 777 if (ret) 778 return ret; 779 780 /* 781 * Find the authoritative RMP entry for a PFN. This can be either a 4K 782 * RMP entry or a special large RMP entry that is authoritative for a 783 * whole 2M area. 784 */ 785 ret = get_rmpentry(pfn & PFN_PMD_MASK, &e_large); 786 if (ret) 787 return ret; 788 789 *level = RMP_TO_PG_LEVEL(e_large.pagesize); 790 791 return 0; 792 } 793 794 int snp_lookup_rmpentry(u64 pfn, bool *assigned, int *level) 795 { 796 struct rmpentry e; 797 int ret; 798 799 ret = __snp_lookup_rmpentry(pfn, &e, level); 800 if (ret) 801 return ret; 802 803 *assigned = !!e.assigned; 804 return 0; 805 } 806 EXPORT_SYMBOL_GPL(snp_lookup_rmpentry); 807 808 /* 809 * Dump the raw RMP entry for a particular PFN. These bits are documented in the 810 * PPR for a particular CPU model and provide useful information about how a 811 * particular PFN is being utilized by the kernel/firmware at the time certain 812 * unexpected events occur, such as RMP faults. 813 */ 814 static void dump_rmpentry(u64 pfn) 815 { 816 struct rmpentry_raw *e_raw; 817 u64 pfn_i, pfn_end; 818 struct rmpentry e; 819 int level, ret; 820 821 ret = __snp_lookup_rmpentry(pfn, &e, &level); 822 if (ret) { 823 pr_err("Failed to read RMP entry for PFN 0x%llx, error %d\n", 824 pfn, ret); 825 return; 826 } 827 828 if (e.assigned) { 829 e_raw = get_raw_rmpentry(pfn); 830 if (IS_ERR(e_raw)) { 831 pr_err("Failed to read RMP contents for PFN 0x%llx, error %ld\n", 832 pfn, PTR_ERR(e_raw)); 833 return; 834 } 835 836 pr_info("PFN 0x%llx, RMP entry: [0x%016llx - 0x%016llx]\n", 837 pfn, e_raw->lo, e_raw->hi); 838 return; 839 } 840 841 /* 842 * If the RMP entry for a particular PFN is not in an assigned state, 843 * then it is sometimes useful to get an idea of whether or not any RMP 844 * entries for other PFNs within the same 2MB region are assigned, since 845 * those too can affect the ability to access a particular PFN in 846 * certain situations, such as when the PFN is being accessed via a 2MB 847 * mapping in the host page table. 848 */ 849 pfn_i = ALIGN_DOWN(pfn, PTRS_PER_PMD); 850 pfn_end = pfn_i + PTRS_PER_PMD; 851 852 pr_info("PFN 0x%llx unassigned, dumping non-zero entries in 2M PFN region: [0x%llx - 0x%llx]\n", 853 pfn, pfn_i, pfn_end); 854 855 while (pfn_i < pfn_end) { 856 e_raw = get_raw_rmpentry(pfn_i); 857 if (IS_ERR(e_raw)) { 858 pr_err("Error %ld reading RMP contents for PFN 0x%llx\n", 859 PTR_ERR(e_raw), pfn_i); 860 pfn_i++; 861 continue; 862 } 863 864 if (e_raw->lo || e_raw->hi) 865 pr_info("PFN: 0x%llx, [0x%016llx - 0x%016llx]\n", pfn_i, e_raw->lo, e_raw->hi); 866 pfn_i++; 867 } 868 } 869 870 void snp_dump_hva_rmpentry(unsigned long hva) 871 { 872 unsigned long paddr; 873 unsigned int level; 874 pgd_t *pgd; 875 pte_t *pte; 876 877 pgd = __va(read_cr3_pa()); 878 pgd += pgd_index(hva); 879 pte = lookup_address_in_pgd(pgd, hva, &level); 880 881 if (!pte) { 882 pr_err("Can't dump RMP entry for HVA %lx: no PTE/PFN found\n", hva); 883 return; 884 } 885 886 paddr = PFN_PHYS(pte_pfn(*pte)) | (hva & ~page_level_mask(level)); 887 dump_rmpentry(PHYS_PFN(paddr)); 888 } 889 890 /* 891 * PSMASH a 2MB aligned page into 4K pages in the RMP table while preserving the 892 * Validated bit. 893 */ 894 int psmash(u64 pfn) 895 { 896 unsigned long paddr = pfn << PAGE_SHIFT; 897 int ret; 898 899 if (!cc_platform_has(CC_ATTR_HOST_SEV_SNP)) 900 return -ENODEV; 901 902 if (!pfn_valid(pfn)) 903 return -EINVAL; 904 905 /* Binutils version 2.36 supports the PSMASH mnemonic. */ 906 asm volatile(".byte 0xF3, 0x0F, 0x01, 0xFF" 907 : "=a" (ret) 908 : "a" (paddr) 909 : "memory", "cc"); 910 911 return ret; 912 } 913 EXPORT_SYMBOL_GPL(psmash); 914 915 /* 916 * If the kernel uses a 2MB or larger directmap mapping to write to an address, 917 * and that mapping contains any 4KB pages that are set to private in the RMP 918 * table, an RMP #PF will trigger and cause a host crash. Hypervisor code that 919 * owns the PFNs being transitioned will never attempt such a write, but other 920 * kernel tasks writing to other PFNs in the range may trigger these checks 921 * inadvertently due a large directmap mapping that happens to overlap such a 922 * PFN. 923 * 924 * Prevent this by splitting any 2MB+ mappings that might end up containing a 925 * mix of private/shared PFNs as a result of a subsequent RMPUPDATE for the 926 * PFN/rmp_level passed in. 927 * 928 * Note that there is no attempt here to scan all the RMP entries for the 2MB 929 * physical range, since it would only be worthwhile in determining if a 930 * subsequent RMPUPDATE for a 4KB PFN would result in all the entries being of 931 * the same shared/private state, thus avoiding the need to split the mapping. 932 * But that would mean the entries are currently in a mixed state, and so the 933 * mapping would have already been split as a result of prior transitions. 934 * And since the 4K split is only done if the mapping is 2MB+, and there isn't 935 * currently a mechanism in place to restore 2MB+ mappings, such a check would 936 * not provide any usable benefit. 937 * 938 * More specifics on how these checks are carried out can be found in APM 939 * Volume 2, "RMP and VMPL Access Checks". 940 */ 941 static int adjust_direct_map(u64 pfn, int rmp_level) 942 { 943 unsigned long vaddr; 944 unsigned int level; 945 int npages, ret; 946 pte_t *pte; 947 948 /* 949 * pfn_to_kaddr() will return a vaddr only within the direct 950 * map range. 951 */ 952 vaddr = (unsigned long)pfn_to_kaddr(pfn); 953 954 /* Only 4KB/2MB RMP entries are supported by current hardware. */ 955 if (WARN_ON_ONCE(rmp_level > PG_LEVEL_2M)) 956 return -EINVAL; 957 958 if (!pfn_valid(pfn)) 959 return -EINVAL; 960 961 if (rmp_level == PG_LEVEL_2M && 962 (!IS_ALIGNED(pfn, PTRS_PER_PMD) || !pfn_valid(pfn + PTRS_PER_PMD - 1))) 963 return -EINVAL; 964 965 /* 966 * If an entire 2MB physical range is being transitioned, then there is 967 * no risk of RMP #PFs due to write accesses from overlapping mappings, 968 * since even accesses from 1GB mappings will be treated as 2MB accesses 969 * as far as RMP table checks are concerned. 970 */ 971 if (rmp_level == PG_LEVEL_2M) 972 return 0; 973 974 pte = lookup_address(vaddr, &level); 975 if (!pte || pte_none(*pte)) 976 return 0; 977 978 if (level == PG_LEVEL_4K) 979 return 0; 980 981 npages = page_level_size(rmp_level) / PAGE_SIZE; 982 ret = set_memory_4k(vaddr, npages); 983 if (ret) 984 pr_warn("Failed to split direct map for PFN 0x%llx, ret: %d\n", 985 pfn, ret); 986 987 return ret; 988 } 989 990 /* 991 * It is expected that those operations are seldom enough so that no mutual 992 * exclusion of updaters is needed and thus the overlap error condition below 993 * should happen very rarely and would get resolved relatively quickly by 994 * the firmware. 995 * 996 * If not, one could consider introducing a mutex or so here to sync concurrent 997 * RMP updates and thus diminish the amount of cases where firmware needs to 998 * lock 2M ranges to protect against concurrent updates. 999 * 1000 * The optimal solution would be range locking to avoid locking disjoint 1001 * regions unnecessarily but there's no support for that yet. 1002 */ 1003 static int rmpupdate(u64 pfn, struct rmp_state *state) 1004 { 1005 unsigned long paddr = pfn << PAGE_SHIFT; 1006 int ret, level; 1007 1008 if (!cc_platform_has(CC_ATTR_HOST_SEV_SNP)) 1009 return -ENODEV; 1010 1011 level = RMP_TO_PG_LEVEL(state->pagesize); 1012 1013 if (adjust_direct_map(pfn, level)) 1014 return -EFAULT; 1015 1016 do { 1017 /* Binutils version 2.36 supports the RMPUPDATE mnemonic. */ 1018 asm volatile(".byte 0xF2, 0x0F, 0x01, 0xFE" 1019 : "=a" (ret) 1020 : "a" (paddr), "c" ((unsigned long)state) 1021 : "memory", "cc"); 1022 } while (ret == RMPUPDATE_FAIL_OVERLAP); 1023 1024 if (ret) { 1025 pr_err("RMPUPDATE failed for PFN %llx, pg_level: %d, ret: %d\n", 1026 pfn, level, ret); 1027 dump_rmpentry(pfn); 1028 dump_stack(); 1029 return -EFAULT; 1030 } 1031 1032 return 0; 1033 } 1034 1035 /* Transition a page to guest-owned/private state in the RMP table. */ 1036 int rmp_make_private(u64 pfn, u64 gpa, enum pg_level level, u32 asid, bool immutable) 1037 { 1038 struct rmp_state state; 1039 1040 memset(&state, 0, sizeof(state)); 1041 state.assigned = 1; 1042 state.asid = asid; 1043 state.immutable = immutable; 1044 state.gpa = gpa; 1045 state.pagesize = PG_LEVEL_TO_RMP(level); 1046 1047 return rmpupdate(pfn, &state); 1048 } 1049 EXPORT_SYMBOL_GPL(rmp_make_private); 1050 1051 /* Transition a page to hypervisor-owned/shared state in the RMP table. */ 1052 int rmp_make_shared(u64 pfn, enum pg_level level) 1053 { 1054 struct rmp_state state; 1055 1056 memset(&state, 0, sizeof(state)); 1057 state.pagesize = PG_LEVEL_TO_RMP(level); 1058 1059 return rmpupdate(pfn, &state); 1060 } 1061 EXPORT_SYMBOL_GPL(rmp_make_shared); 1062 1063 void __snp_leak_pages(u64 pfn, unsigned int npages, bool dump_rmp) 1064 { 1065 struct page *page = pfn_to_page(pfn); 1066 1067 pr_warn("Leaking PFN range 0x%llx-0x%llx\n", pfn, pfn + npages); 1068 1069 spin_lock(&snp_leaked_pages_list_lock); 1070 while (npages--) { 1071 1072 /* 1073 * Reuse the page's buddy list for chaining into the leaked 1074 * pages list. This page should not be on a free list currently 1075 * and is also unsafe to be added to a free list. 1076 */ 1077 if (likely(!PageCompound(page)) || 1078 1079 /* 1080 * Skip inserting tail pages of compound page as 1081 * page->buddy_list of tail pages is not usable. 1082 */ 1083 (PageHead(page) && compound_nr(page) <= npages)) 1084 list_add_tail(&page->buddy_list, &snp_leaked_pages_list); 1085 1086 if (dump_rmp) 1087 dump_rmpentry(pfn); 1088 snp_nr_leaked_pages++; 1089 pfn++; 1090 page++; 1091 } 1092 spin_unlock(&snp_leaked_pages_list_lock); 1093 } 1094 EXPORT_SYMBOL_GPL(__snp_leak_pages); 1095 1096 void kdump_sev_callback(void) 1097 { 1098 /* 1099 * Do wbinvd() on remote CPUs when SNP is enabled in order to 1100 * safely do SNP_SHUTDOWN on the local CPU. 1101 */ 1102 if (cc_platform_has(CC_ATTR_HOST_SEV_SNP)) 1103 wbinvd(); 1104 } 1105