1 // SPDX-License-Identifier: GPL-2.0-only 2 /* 3 * Copyright (C) 2017 - Columbia University and Linaro Ltd. 4 * Author: Jintack Lim <jintack.lim@linaro.org> 5 */ 6 7 #include <linux/bitfield.h> 8 #include <linux/kvm.h> 9 #include <linux/kvm_host.h> 10 11 #include <asm/fixmap.h> 12 #include <asm/kvm_arm.h> 13 #include <asm/kvm_emulate.h> 14 #include <asm/kvm_mmu.h> 15 #include <asm/kvm_nested.h> 16 #include <asm/sysreg.h> 17 18 #include "sys_regs.h" 19 #include "vgic/vgic.h" 20 21 struct vncr_tlb { 22 /* The guest's VNCR_EL2 */ 23 u64 gva; 24 struct s1_walk_info wi; 25 struct s1_walk_result wr; 26 27 u64 hpa; 28 bool hpa_writable; 29 30 /* -1 when not mapped on a CPU */ 31 atomic_t cpu; 32 33 /* 34 * true if the TLB is valid. Can only be changed with the 35 * mmu_lock held. 36 */ 37 bool valid; 38 }; 39 40 /* 41 * Ratio of live shadow S2 MMU per vcpu. This is a trade-off between 42 * memory usage and potential number of different sets of S2 PTs in 43 * the guests. Running out of S2 MMUs only affects performance (we 44 * will invalidate them more often). 45 */ 46 #define S2_MMU_PER_VCPU 2 47 48 int kvm_init_nested(struct kvm *kvm) 49 { 50 kvm->arch.nested_mmus = kvmalloc_objs(struct kvm_s2_mmu *, 51 KVM_MAX_VCPUS * S2_MMU_PER_VCPU, 52 GFP_KERNEL_ACCOUNT); 53 kvm->arch.nested_mmus_size = 0; 54 atomic_set(&kvm->arch.vncr_tlb_count, 0); 55 56 return kvm->arch.nested_mmus ? 0 : -ENOMEM; 57 } 58 59 void kvm_destroy_nested(struct kvm *kvm) 60 { 61 for (int i = 0; i < kvm->arch.nested_mmus_size; i+= S2_MMU_PER_VCPU) 62 kvfree(kvm->arch.nested_mmus[i]); 63 64 kvm->arch.nested_mmus_size = 0; 65 kvfree(kvm->arch.nested_mmus); 66 } 67 68 static int init_nested_s2_mmu(struct kvm *kvm, struct kvm_s2_mmu *mmu) 69 { 70 /* 71 * We only initialise the IPA range on the canonical MMU, which 72 * defines the contract between KVM and userspace on where the 73 * "hardware" is in the IPA space. This affects the validity of MMIO 74 * exits forwarded to userspace, for example. 75 * 76 * For nested S2s, we use the PARange as exposed to the guest, as it 77 * is allowed to use it at will to expose whatever memory map it 78 * wants to its own guests as it would be on real HW. 79 */ 80 return kvm_init_stage2_mmu(kvm, mmu, kvm_get_pa_bits(kvm)); 81 } 82 83 int kvm_vcpu_init_nested(struct kvm_vcpu *vcpu) 84 { 85 struct kvm *kvm = vcpu->kvm; 86 int num_mmus; 87 88 lockdep_assert_held(&kvm->arch.config_lock); 89 90 if (test_bit(KVM_ARM_VCPU_HAS_EL2_E2H0, kvm->arch.vcpu_features) && 91 !cpus_have_final_cap(ARM64_HAS_HCR_NV1)) 92 return -EINVAL; 93 94 if (!vcpu->arch.ctxt.vncr_array) 95 vcpu->arch.ctxt.vncr_array = (u64 *)__get_free_page(GFP_KERNEL_ACCOUNT | 96 __GFP_ZERO); 97 98 if (!vcpu->arch.ctxt.vncr_array) 99 return -ENOMEM; 100 101 num_mmus = atomic_read(&kvm->online_vcpus) * S2_MMU_PER_VCPU; 102 103 if (num_mmus > kvm->arch.nested_mmus_size) { 104 struct kvm_s2_mmu *tmp; 105 int i, ret = 0; 106 107 tmp = kvzalloc_objs(*tmp, S2_MMU_PER_VCPU, GFP_KERNEL_ACCOUNT); 108 if (!tmp) 109 ret = -ENOMEM; 110 111 for (i = 0; !ret && i < S2_MMU_PER_VCPU; i++) { 112 ret = init_nested_s2_mmu(kvm, &tmp[i]); 113 if (ret) 114 break; 115 } 116 117 if (ret) { 118 while (--i >= 0) 119 kvm_free_stage2_pgd(&tmp[i]); 120 121 kvfree(tmp); 122 free_page((unsigned long)vcpu->arch.ctxt.vncr_array); 123 vcpu->arch.ctxt.vncr_array = NULL; 124 return ret; 125 } 126 127 guard(write_lock)(&kvm->mmu_lock); 128 129 for (i = 0; i < S2_MMU_PER_VCPU; i++) 130 kvm->arch.nested_mmus[i + kvm->arch.nested_mmus_size] = &tmp[i]; 131 132 kvm->arch.nested_mmus_size += S2_MMU_PER_VCPU; 133 } 134 135 return 0; 136 } 137 138 struct s2_walk_info { 139 u64 baddr; 140 unsigned int max_oa_bits; 141 unsigned int pgshift; 142 unsigned int sl; 143 unsigned int t0sz; 144 bool be; 145 bool ha; 146 }; 147 148 static u32 compute_fsc(int level, u32 fsc) 149 { 150 return fsc | (level & 0x3); 151 } 152 153 static int esr_s2_fault(struct kvm_vcpu *vcpu, int level, u32 fsc) 154 { 155 u32 esr; 156 157 esr = kvm_vcpu_get_esr(vcpu) & ~ESR_ELx_FSC; 158 esr |= compute_fsc(level, fsc); 159 return esr; 160 } 161 162 static int get_ia_size(struct s2_walk_info *wi) 163 { 164 return 64 - wi->t0sz; 165 } 166 167 static int check_base_s2_limits(struct kvm_vcpu *vcpu, struct s2_walk_info *wi, 168 int level, int input_size, int stride) 169 { 170 int start_size, pa_max; 171 172 pa_max = kvm_get_pa_bits(vcpu->kvm); 173 174 /* Check translation limits */ 175 switch (BIT(wi->pgshift)) { 176 case SZ_64K: 177 if (level == 0 || (level == 1 && pa_max <= 42)) 178 return -EFAULT; 179 break; 180 case SZ_16K: 181 if (level == 0 || (level == 1 && pa_max <= 40)) 182 return -EFAULT; 183 break; 184 case SZ_4K: 185 if (level < 0 || (level == 0 && pa_max <= 42)) 186 return -EFAULT; 187 break; 188 } 189 190 /* Check input size limits */ 191 if (input_size > pa_max) 192 return -EFAULT; 193 194 /* Check number of entries in starting level table */ 195 start_size = input_size - ((3 - level) * stride + wi->pgshift); 196 if (start_size < 1 || start_size > stride + 4) 197 return -EFAULT; 198 199 return 0; 200 } 201 202 /* Check if output is within boundaries */ 203 static int check_output_size(struct s2_walk_info *wi, phys_addr_t output) 204 { 205 unsigned int output_size = wi->max_oa_bits; 206 207 if (output_size != 48 && (output & GENMASK_ULL(47, output_size))) 208 return -1; 209 210 return 0; 211 } 212 213 static int read_guest_s2_desc(struct kvm_vcpu *vcpu, phys_addr_t pa, u64 *desc, 214 struct s2_walk_info *wi) 215 { 216 u64 val; 217 int r; 218 219 r = kvm_read_guest(vcpu->kvm, pa, &val, sizeof(val)); 220 if (r) 221 return r; 222 223 /* 224 * Handle reversedescriptors if endianness differs between the 225 * host and the guest hypervisor. 226 */ 227 if (wi->be) 228 *desc = be64_to_cpu((__force __be64)val); 229 else 230 *desc = le64_to_cpu((__force __le64)val); 231 232 return 0; 233 } 234 235 static int swap_guest_s2_desc(struct kvm_vcpu *vcpu, phys_addr_t pa, u64 old, u64 new, 236 struct s2_walk_info *wi) 237 { 238 if (wi->be) { 239 old = (__force u64)cpu_to_be64(old); 240 new = (__force u64)cpu_to_be64(new); 241 } else { 242 old = (__force u64)cpu_to_le64(old); 243 new = (__force u64)cpu_to_le64(new); 244 } 245 246 return __kvm_at_swap_desc(vcpu->kvm, pa, old, new); 247 } 248 249 /* 250 * This is essentially a C-version of the pseudo code from the ARM ARM 251 * AArch64.TranslationTableWalk function. I strongly recommend looking at 252 * that pseudocode in trying to understand this. 253 * 254 * Must be called with the kvm->srcu read lock held 255 */ 256 static int walk_nested_s2_pgd(struct kvm_vcpu *vcpu, phys_addr_t ipa, 257 struct s2_walk_info *wi, struct kvm_s2_trans *out) 258 { 259 int first_block_level, level, stride, input_size, base_lower_bound; 260 phys_addr_t base_addr; 261 unsigned int addr_top, addr_bottom; 262 u64 desc, new_desc; /* page table entry */ 263 int ret; 264 phys_addr_t paddr; 265 266 switch (BIT(wi->pgshift)) { 267 default: 268 case SZ_64K: 269 case SZ_16K: 270 level = 3 - wi->sl; 271 first_block_level = 2; 272 break; 273 case SZ_4K: 274 level = 2 - wi->sl; 275 first_block_level = 1; 276 break; 277 } 278 279 stride = wi->pgshift - 3; 280 input_size = get_ia_size(wi); 281 if (input_size > 48 || input_size < 25) 282 return -EFAULT; 283 284 ret = check_base_s2_limits(vcpu, wi, level, input_size, stride); 285 if (WARN_ON(ret)) { 286 out->esr = compute_fsc(0, ESR_ELx_FSC_FAULT); 287 return ret; 288 } 289 290 base_lower_bound = 3 + input_size - ((3 - level) * stride + 291 wi->pgshift); 292 base_addr = wi->baddr & GENMASK_ULL(47, base_lower_bound); 293 294 if (check_output_size(wi, base_addr)) { 295 /* R_BFHQH */ 296 out->esr = compute_fsc(0, ESR_ELx_FSC_ADDRSZ); 297 return 1; 298 } 299 300 addr_top = input_size - 1; 301 302 while (1) { 303 phys_addr_t index; 304 305 addr_bottom = (3 - level) * stride + wi->pgshift; 306 index = (ipa & GENMASK_ULL(addr_top, addr_bottom)) 307 >> (addr_bottom - 3); 308 309 paddr = base_addr | index; 310 ret = read_guest_s2_desc(vcpu, paddr, &desc, wi); 311 if (ret < 0) { 312 out->esr = ESR_ELx_FSC_SEA_TTW(level); 313 return ret; 314 } 315 316 new_desc = desc; 317 318 /* Check for valid descriptor at this point */ 319 if (!(desc & KVM_PTE_VALID)) { 320 out->esr = compute_fsc(level, ESR_ELx_FSC_FAULT); 321 out->desc = desc; 322 return 1; 323 } 324 325 if (FIELD_GET(KVM_PTE_TYPE, desc) == KVM_PTE_TYPE_BLOCK) { 326 if (level < 3) 327 break; 328 329 out->esr = compute_fsc(level, ESR_ELx_FSC_FAULT); 330 out->desc = desc; 331 return 1; 332 } 333 334 /* We're at the final level */ 335 if (level == 3) 336 break; 337 338 if (check_output_size(wi, desc)) { 339 out->esr = compute_fsc(level, ESR_ELx_FSC_ADDRSZ); 340 out->desc = desc; 341 return 1; 342 } 343 344 base_addr = desc & GENMASK_ULL(47, wi->pgshift); 345 346 level += 1; 347 addr_top = addr_bottom - 1; 348 } 349 350 if (level < first_block_level) { 351 out->esr = compute_fsc(level, ESR_ELx_FSC_FAULT); 352 out->desc = desc; 353 return 1; 354 } 355 356 if (check_output_size(wi, desc)) { 357 out->esr = compute_fsc(level, ESR_ELx_FSC_ADDRSZ); 358 out->desc = desc; 359 return 1; 360 } 361 362 if (wi->ha) 363 new_desc |= KVM_PTE_LEAF_ATTR_LO_S2_AF; 364 365 if (new_desc != desc) { 366 ret = swap_guest_s2_desc(vcpu, paddr, desc, new_desc, wi); 367 if (ret == -EAGAIN) 368 return ret; 369 if (ret) { 370 out->esr = ESR_ELx_FSC_SEA_TTW(level); 371 out->desc = desc; 372 return 1; 373 } 374 375 desc = new_desc; 376 } 377 378 if (!(desc & KVM_PTE_LEAF_ATTR_LO_S2_AF)) { 379 out->esr = compute_fsc(level, ESR_ELx_FSC_ACCESS); 380 out->desc = desc; 381 return 1; 382 } 383 384 addr_bottom += contiguous_bit_shift(desc, wi, level); 385 386 /* Calculate and return the result */ 387 paddr = (desc & GENMASK_ULL(47, addr_bottom)) | 388 (ipa & GENMASK_ULL(addr_bottom - 1, 0)); 389 out->output = paddr; 390 out->block_size = 1UL << ((3 - level) * stride + wi->pgshift); 391 out->readable = desc & KVM_PTE_LEAF_ATTR_LO_S2_S2AP_R; 392 out->writable = desc & KVM_PTE_LEAF_ATTR_LO_S2_S2AP_W; 393 out->level = level; 394 out->desc = desc; 395 return 0; 396 } 397 398 #define _has_tgran_2(__r, __sz) \ 399 ({ \ 400 u64 _s1, _s2, _mmfr0 = __r; \ 401 \ 402 _s2 = SYS_FIELD_GET(ID_AA64MMFR0_EL1, \ 403 TGRAN##__sz##_2, _mmfr0); \ 404 \ 405 _s1 = SYS_FIELD_GET(ID_AA64MMFR0_EL1, \ 406 TGRAN##__sz, _mmfr0); \ 407 \ 408 ((_s2 != ID_AA64MMFR0_EL1_TGRAN##__sz##_2_NI && \ 409 _s2 != ID_AA64MMFR0_EL1_TGRAN##__sz##_2_TGRAN##__sz) || \ 410 (_s2 == ID_AA64MMFR0_EL1_TGRAN##__sz##_2_TGRAN##__sz && \ 411 _s1 != ID_AA64MMFR0_EL1_TGRAN##__sz##_NI)); \ 412 }) 413 414 static bool has_tgran_2(u64 mmfr0, unsigned int shift) 415 { 416 switch (shift) { 417 case 12: 418 return _has_tgran_2(mmfr0, 4); 419 case 14: 420 return _has_tgran_2(mmfr0, 16); 421 case 16: 422 return _has_tgran_2(mmfr0, 64); 423 default: 424 BUG(); 425 } 426 } 427 428 static unsigned int fallback_tgran2_shift(u64 mmfr0) 429 { 430 if (has_tgran_2(mmfr0, PAGE_SHIFT)) 431 return PAGE_SHIFT; 432 else if (has_tgran_2(mmfr0, 12)) 433 return 12; 434 else if (has_tgran_2(mmfr0, 14)) 435 return 14; 436 else if (has_tgran_2(mmfr0, 16)) 437 return 16; 438 else 439 return PAGE_SHIFT; 440 } 441 442 static unsigned int vtcr_to_tg0_pgshift(struct kvm *kvm, u64 vtcr) 443 { 444 u64 tg0 = FIELD_GET(VTCR_EL2_TG0_MASK, vtcr); 445 u64 mmfr0 = kvm_read_vm_id_reg(kvm, SYS_ID_AA64MMFR0_EL1); 446 unsigned int shift; 447 448 switch (tg0) { 449 case VTCR_EL2_TG0_4K: 450 shift = 12; 451 break; 452 case VTCR_EL2_TG0_16K: 453 shift = 14; 454 break; 455 case VTCR_EL2_TG0_64K: 456 /* IMPDEF: treat any other value as 64k, subject to fallback */ 457 default: 458 shift = 16; 459 } 460 461 /* 462 * If TGx is programmed to an unimplemented value (not advertised in 463 * ID_AA64MMFR0_EL1), we should treat it as if an implemented value is 464 * written, as per the architecture. Choose an available one while 465 * prioritizing PAGE_SIZE. 466 */ 467 if (!has_tgran_2(mmfr0, shift)) 468 return fallback_tgran2_shift(mmfr0); 469 470 return shift; 471 } 472 473 static size_t vtcr_to_tg0_pgsize(struct kvm *kvm, u64 vtcr) 474 { 475 return BIT(vtcr_to_tg0_pgshift(kvm, vtcr)); 476 } 477 478 static void setup_s2_walk(struct kvm_vcpu *vcpu, struct s2_walk_info *wi) 479 { 480 u64 vtcr = vcpu_read_sys_reg(vcpu, VTCR_EL2); 481 482 wi->baddr = vcpu_read_sys_reg(vcpu, VTTBR_EL2); 483 wi->t0sz = vtcr & VTCR_EL2_T0SZ_MASK; 484 wi->pgshift = vtcr_to_tg0_pgshift(vcpu->kvm, vtcr); 485 wi->sl = FIELD_GET(VTCR_EL2_SL0_MASK, vtcr); 486 /* Global limit for now, should eventually be per-VM */ 487 wi->max_oa_bits = min(get_kvm_ipa_limit(), 488 ps_to_output_size(FIELD_GET(VTCR_EL2_PS_MASK, vtcr), false)); 489 wi->ha = vtcr & VTCR_EL2_HA; 490 wi->be = vcpu_read_sys_reg(vcpu, SCTLR_EL2) & SCTLR_ELx_EE; 491 } 492 493 int kvm_walk_nested_s2(struct kvm_vcpu *vcpu, phys_addr_t gipa, 494 struct kvm_s2_trans *result) 495 { 496 struct s2_walk_info wi; 497 int ret; 498 499 result->esr = 0; 500 501 if (!vcpu_has_nv(vcpu)) 502 return 0; 503 504 setup_s2_walk(vcpu, &wi); 505 506 ret = walk_nested_s2_pgd(vcpu, gipa, &wi, result); 507 if (ret) 508 result->esr |= (kvm_vcpu_get_esr(vcpu) & ~ESR_ELx_FSC); 509 510 return ret; 511 } 512 513 static unsigned int __ttl_to_size(u8 ttl) 514 { 515 int level = ttl & 3; 516 int gran = (ttl >> 2) & 3; 517 unsigned int max_size = 0; 518 519 switch (gran) { 520 case TLBI_TTL_TG_4K: 521 switch (level) { 522 case 0: 523 break; 524 case 1: 525 max_size = SZ_1G; 526 break; 527 case 2: 528 max_size = SZ_2M; 529 break; 530 case 3: 531 max_size = SZ_4K; 532 break; 533 } 534 break; 535 case TLBI_TTL_TG_16K: 536 switch (level) { 537 case 0: 538 case 1: 539 break; 540 case 2: 541 max_size = SZ_32M; 542 break; 543 case 3: 544 max_size = SZ_16K; 545 break; 546 } 547 break; 548 case TLBI_TTL_TG_64K: 549 switch (level) { 550 case 0: 551 case 1: 552 /* No 52bit IPA support */ 553 break; 554 case 2: 555 max_size = SZ_512M; 556 break; 557 case 3: 558 max_size = SZ_64K; 559 break; 560 } 561 break; 562 default: /* No size information */ 563 break; 564 } 565 566 return max_size; 567 } 568 569 static unsigned int ttl_to_size(u8 ttl) 570 { 571 return __ttl_to_size(ttl) ?: SZ_1G; 572 } 573 574 static u8 pgshift_level_to_ttl(u16 shift, s8 level) 575 { 576 u8 ttl; 577 578 /* 579 * If we don't have a proper level, fallback to the maximum 580 * size. 581 */ 582 if (level < 0) 583 return 0; 584 585 switch(shift) { 586 case 12: 587 ttl = TLBI_TTL_TG_4K; 588 break; 589 case 14: 590 ttl = TLBI_TTL_TG_16K; 591 break; 592 case 16: 593 ttl = TLBI_TTL_TG_64K; 594 break; 595 default: 596 BUG(); 597 } 598 599 ttl <<= 2; 600 ttl |= level & 3; 601 602 return ttl; 603 } 604 605 /* 606 * Compute the equivalent of the TTL field by parsing the shadow PT. The 607 * granule size is extracted from the cached VTCR_EL2.TG0 while the level is 608 * retrieved from first entry carrying the level as a tag. 609 */ 610 static u8 get_guest_mapping_ttl(struct kvm_s2_mmu *mmu, u64 addr) 611 { 612 size_t tg0_size = vtcr_to_tg0_pgsize(kvm_s2_mmu_to_kvm(mmu), mmu->tlb_vtcr); 613 u64 tmp, sz = 0; 614 kvm_pte_t pte; 615 u8 ttl, level; 616 617 lockdep_assert_held_write(&kvm_s2_mmu_to_kvm(mmu)->mmu_lock); 618 619 switch (tg0_size) { 620 case SZ_4K: 621 ttl = (TLBI_TTL_TG_4K << 2); 622 break; 623 case SZ_16K: 624 ttl = (TLBI_TTL_TG_16K << 2); 625 break; 626 case SZ_64K: 627 default: /* IMPDEF: treat any other value as 64k */ 628 ttl = (TLBI_TTL_TG_64K << 2); 629 break; 630 } 631 632 tmp = addr; 633 634 again: 635 /* Iteratively compute the block sizes for a particular granule size */ 636 switch (tg0_size) { 637 case SZ_4K: 638 if (sz < SZ_4K) sz = SZ_4K; 639 else if (sz < SZ_2M) sz = SZ_2M; 640 else if (sz < SZ_1G) sz = SZ_1G; 641 else sz = 0; 642 break; 643 case SZ_16K: 644 if (sz < SZ_16K) sz = SZ_16K; 645 else if (sz < SZ_32M) sz = SZ_32M; 646 else sz = 0; 647 break; 648 case SZ_64K: 649 default: /* IMPDEF: treat any other value as 64k */ 650 if (sz < SZ_64K) sz = SZ_64K; 651 else if (sz < SZ_512M) sz = SZ_512M; 652 else sz = 0; 653 break; 654 } 655 656 if (sz == 0) 657 return 0; 658 659 tmp &= ~(sz - 1); 660 if (kvm_pgtable_get_leaf(mmu->pgt, tmp, &pte, NULL)) 661 goto again; 662 if (!(pte & PTE_VALID)) 663 goto again; 664 level = FIELD_GET(KVM_NV_GUEST_MAP_SZ, pte); 665 if (!level) 666 goto again; 667 668 ttl |= level; 669 670 /* 671 * We now have found some level information in the shadow S2. Check 672 * that the resulting range is actually including the original IPA. 673 */ 674 sz = ttl_to_size(ttl); 675 if (addr < (tmp + sz)) 676 return ttl; 677 678 return 0; 679 } 680 681 unsigned long compute_tlb_inval_range(struct kvm_s2_mmu *mmu, u64 val) 682 { 683 struct kvm *kvm = kvm_s2_mmu_to_kvm(mmu); 684 unsigned long max_size; 685 u8 ttl; 686 687 ttl = FIELD_GET(TLBI_TTL_MASK, val); 688 689 if (!ttl || !kvm_has_feat(kvm, ID_AA64MMFR2_EL1, TTL, IMP)) { 690 /* No TTL, check the shadow S2 for a hint */ 691 u64 addr = (val & GENMASK_ULL(35, 0)) << 12; 692 ttl = get_guest_mapping_ttl(mmu, addr); 693 } 694 695 /* 696 * Don't use the default 1GB fallback, as we can adapt to the 697 * max mapping size we allow at S2. 698 */ 699 max_size = __ttl_to_size(ttl); 700 701 if (!max_size) { 702 /* Compute the maximum extent of the invalidation */ 703 switch (vtcr_to_tg0_pgsize(kvm, mmu->tlb_vtcr)) { 704 case SZ_4K: 705 max_size = SZ_1G; 706 break; 707 case SZ_16K: 708 max_size = SZ_32M; 709 break; 710 case SZ_64K: 711 default: /* IMPDEF: treat any other value as 64k */ 712 /* 713 * No, we do not support 52bit IPA in nested yet. Once 714 * we do, this should be 4TB. 715 */ 716 max_size = SZ_512M; 717 break; 718 } 719 } 720 721 WARN_ON(!max_size); 722 return max_size; 723 } 724 725 /* 726 * We can have multiple *different* MMU contexts with the same VMID: 727 * 728 * - S2 being enabled or not, hence differing by the HCR_EL2.VM bit 729 * 730 * - Multiple vcpus using private S2s (huh huh...), hence differing by the 731 * VBBTR_EL2.BADDR address 732 * 733 * - A combination of the above... 734 * 735 * We can always identify which MMU context to pick at run-time. However, 736 * TLB invalidation involving a VMID must take action on all the TLBs using 737 * this particular VMID. This translates into applying the same invalidation 738 * operation to all the contexts that are using this VMID. Moar phun! 739 */ 740 void kvm_s2_mmu_iterate_by_vmid(struct kvm *kvm, u16 vmid, 741 const union tlbi_info *info, 742 void (*tlbi_callback)(struct kvm_s2_mmu *, 743 const union tlbi_info *)) 744 { 745 write_lock(&kvm->mmu_lock); 746 747 for (int i = 0; i < kvm->arch.nested_mmus_size; i++) { 748 struct kvm_s2_mmu *mmu = kvm->arch.nested_mmus[i]; 749 750 if (!kvm_s2_mmu_valid(mmu)) 751 continue; 752 753 if (vmid == get_vmid(mmu->tlb_vttbr)) 754 tlbi_callback(mmu, info); 755 } 756 757 write_unlock(&kvm->mmu_lock); 758 } 759 760 struct kvm_s2_mmu *lookup_s2_mmu(struct kvm_vcpu *vcpu) 761 { 762 struct kvm *kvm = vcpu->kvm; 763 bool nested_stage2_enabled; 764 u64 vttbr, vtcr, hcr; 765 766 lockdep_assert_held_write(&kvm->mmu_lock); 767 768 vttbr = vcpu_read_sys_reg(vcpu, VTTBR_EL2); 769 vtcr = vcpu_read_sys_reg(vcpu, VTCR_EL2); 770 hcr = vcpu_read_sys_reg(vcpu, HCR_EL2); 771 772 nested_stage2_enabled = hcr & HCR_VM; 773 774 /* Don't consider the CnP bit for the vttbr match */ 775 vttbr &= ~VTTBR_CNP_BIT; 776 777 /* 778 * Two possibilities when looking up a S2 MMU context: 779 * 780 * - either S2 is enabled in the guest, and we need a context that is 781 * S2-enabled and matches the full VTTBR (VMID+BADDR) and VTCR, 782 * which makes it safe from a TLB conflict perspective (a broken 783 * guest won't be able to generate them), 784 * 785 * - or S2 is disabled, and we need a context that is S2-disabled 786 * and matches the VMID only, as all TLBs are tagged by VMID even 787 * if S2 translation is disabled. 788 */ 789 for (int i = 0; i < kvm->arch.nested_mmus_size; i++) { 790 struct kvm_s2_mmu *mmu = kvm->arch.nested_mmus[i]; 791 792 if (!kvm_s2_mmu_valid(mmu)) 793 continue; 794 795 if (nested_stage2_enabled && 796 mmu->nested_stage2_enabled && 797 vttbr == mmu->tlb_vttbr && 798 vtcr == mmu->tlb_vtcr) 799 return mmu; 800 801 if (!nested_stage2_enabled && 802 !mmu->nested_stage2_enabled && 803 get_vmid(vttbr) == get_vmid(mmu->tlb_vttbr)) 804 return mmu; 805 } 806 return NULL; 807 } 808 809 static struct kvm_s2_mmu *get_s2_mmu_nested(struct kvm_vcpu *vcpu) 810 { 811 struct kvm *kvm = vcpu->kvm; 812 struct kvm_s2_mmu *s2_mmu; 813 int i; 814 815 lockdep_assert_held_write(&vcpu->kvm->mmu_lock); 816 817 s2_mmu = lookup_s2_mmu(vcpu); 818 if (s2_mmu) 819 goto out; 820 821 /* 822 * Make sure we don't always search from the same point, or we 823 * will always reuse a potentially active context, leaving 824 * free contexts unused. 825 */ 826 for (i = kvm->arch.nested_mmus_next; 827 i < (kvm->arch.nested_mmus_size + kvm->arch.nested_mmus_next); 828 i++) { 829 s2_mmu = kvm->arch.nested_mmus[i % kvm->arch.nested_mmus_size]; 830 831 if (atomic_read(&s2_mmu->refcnt) == 0) 832 break; 833 } 834 BUG_ON(atomic_read(&s2_mmu->refcnt)); /* We have struct MMUs to spare */ 835 836 /* Set the scene for the next search */ 837 kvm->arch.nested_mmus_next = (i + 1) % kvm->arch.nested_mmus_size; 838 839 /* Make sure we don't forget to do the laundry */ 840 if (kvm_s2_mmu_valid(s2_mmu)) { 841 kvm_nested_s2_ptdump_remove_debugfs(s2_mmu); 842 s2_mmu->pending_unmap = true; 843 } 844 845 /* 846 * The virtual VMID (modulo CnP) will be used as a key when matching 847 * an existing kvm_s2_mmu. 848 * 849 * We cache VTCR at allocation time, once and for all. It'd be great 850 * if the guest didn't screw that one up, as this is not very 851 * forgiving... 852 */ 853 s2_mmu->tlb_vttbr = vcpu_read_sys_reg(vcpu, VTTBR_EL2) & ~VTTBR_CNP_BIT; 854 s2_mmu->tlb_vtcr = vcpu_read_sys_reg(vcpu, VTCR_EL2); 855 s2_mmu->nested_stage2_enabled = vcpu_read_sys_reg(vcpu, HCR_EL2) & HCR_VM; 856 857 kvm_nested_s2_ptdump_create_debugfs(s2_mmu); 858 859 out: 860 atomic_inc(&s2_mmu->refcnt); 861 862 /* 863 * Set the vCPU request to perform an unmap, even if the pending unmap 864 * originates from another vCPU. This guarantees that the MMU has been 865 * completely unmapped before any vCPU actually uses it, and allows 866 * multiple vCPUs to lend a hand with completing the unmap. 867 */ 868 if (s2_mmu->pending_unmap) 869 kvm_make_request(KVM_REQ_NESTED_S2_UNMAP, vcpu); 870 871 return s2_mmu; 872 } 873 874 void kvm_init_nested_s2_mmu(struct kvm_s2_mmu *mmu) 875 { 876 /* CnP being set denotes an invalid entry */ 877 mmu->tlb_vttbr = VTTBR_CNP_BIT; 878 mmu->nested_stage2_enabled = false; 879 atomic_set(&mmu->refcnt, 0); 880 } 881 882 void kvm_vcpu_load_hw_mmu(struct kvm_vcpu *vcpu) 883 { 884 /* 885 * If the vCPU kept its reference on the MMU after the last put, 886 * keep rolling with it. 887 */ 888 if (is_hyp_ctxt(vcpu)) { 889 if (!vcpu->arch.hw_mmu) 890 vcpu->arch.hw_mmu = &vcpu->kvm->arch.mmu; 891 } else { 892 if (!vcpu->arch.hw_mmu) { 893 scoped_guard(write_lock, &vcpu->kvm->mmu_lock) 894 vcpu->arch.hw_mmu = get_s2_mmu_nested(vcpu); 895 } 896 897 if (__vcpu_sys_reg(vcpu, HCR_EL2) & HCR_NV) 898 kvm_make_request(KVM_REQ_MAP_L1_VNCR_EL2, vcpu); 899 } 900 } 901 902 /* 903 * Unmapping an L1 VNCR can happen concurrently without the mmu lock being 904 * effective (vcpu_put() vs TLBI handling). The atomic_xchg below ensures 905 * that only one CPU sets it to -1 while getting a valid CPU number back. 906 */ 907 static int unmap_l1_vncr(struct vncr_tlb *vt) 908 { 909 int cpu = atomic_xchg_relaxed(&vt->cpu, -1); 910 911 if (cpu != -1) 912 clear_fixmap(vncr_fixmap(cpu)); 913 914 return cpu; 915 } 916 917 static void this_cpu_reset_vncr_fixmap(struct kvm_vcpu *vcpu) 918 { 919 if (!host_data_test_flag(L1_VNCR_MAPPED)) 920 return; 921 922 BUG_ON(is_hyp_ctxt(vcpu)); 923 924 /* 925 * Unconditionally unmap the local VNCR if we have lost the race 926 * against a concurrent TLBI. Otherwise we could end-up running 927 * another vcpu with VNCR still mapped if the TLBI thread is 928 * preempted between the exchange and the clear_fixmap(). 929 * 930 * Note that we do not care about the TLBI nuking the fixmap behind 931 * the back of an running vcpu. This will only generate a fault and 932 * possibly a retranslation. 933 */ 934 if (unmap_l1_vncr(vcpu->arch.vncr_tlb) == -1) 935 clear_fixmap(vncr_fixmap(smp_processor_id())); 936 host_data_clear_flag(L1_VNCR_MAPPED); 937 } 938 939 void kvm_vcpu_put_hw_mmu(struct kvm_vcpu *vcpu) 940 { 941 /* Unconditionally drop the VNCR mapping if we have one */ 942 this_cpu_reset_vncr_fixmap(vcpu); 943 944 /* 945 * Keep a reference on the associated stage-2 MMU if the vCPU is 946 * scheduling out and not in WFI emulation, suggesting it is likely to 947 * reuse the MMU sometime soon. 948 */ 949 if (vcpu->scheduled_out && !vcpu_get_flag(vcpu, IN_WFI)) 950 return; 951 952 if (kvm_is_nested_s2_mmu(vcpu->kvm, vcpu->arch.hw_mmu)) 953 atomic_dec(&vcpu->arch.hw_mmu->refcnt); 954 955 vcpu->arch.hw_mmu = NULL; 956 } 957 958 /* 959 * Returns non-zero if permission fault is handled by injecting it to the next 960 * level hypervisor. 961 */ 962 int kvm_s2_handle_perm_fault(struct kvm_vcpu *vcpu, struct kvm_s2_trans *trans) 963 { 964 bool forward_fault = false; 965 966 trans->esr = 0; 967 968 if (!kvm_vcpu_trap_is_permission_fault(vcpu)) 969 return 0; 970 971 if (kvm_vcpu_trap_is_iabt(vcpu)) { 972 if (vcpu_mode_priv(vcpu)) 973 forward_fault = !kvm_s2_trans_exec_el1(vcpu->kvm, trans); 974 else 975 forward_fault = !kvm_s2_trans_exec_el0(vcpu->kvm, trans); 976 } else { 977 bool write_fault = kvm_is_write_fault(vcpu); 978 979 forward_fault = ((write_fault && !trans->writable) || 980 (!write_fault && !trans->readable)); 981 } 982 983 if (forward_fault) 984 trans->esr = esr_s2_fault(vcpu, trans->level, ESR_ELx_FSC_PERM); 985 986 return forward_fault; 987 } 988 989 int kvm_inject_s2_fault(struct kvm_vcpu *vcpu, u64 esr_el2) 990 { 991 vcpu_write_sys_reg(vcpu, vcpu->arch.fault.far_el2, FAR_EL2); 992 vcpu_write_sys_reg(vcpu, vcpu->arch.fault.hpfar_el2, HPFAR_EL2); 993 994 return kvm_inject_nested_sync(vcpu, esr_el2); 995 } 996 997 u16 get_asid_by_regime(struct kvm_vcpu *vcpu, enum trans_regime regime) 998 { 999 enum vcpu_sysreg ttbr_elx; 1000 u64 tcr; 1001 u16 asid; 1002 1003 switch (regime) { 1004 case TR_EL10: 1005 tcr = vcpu_read_sys_reg(vcpu, TCR_EL1); 1006 ttbr_elx = (tcr & TCR_A1) ? TTBR1_EL1 : TTBR0_EL1; 1007 break; 1008 case TR_EL20: 1009 tcr = vcpu_read_sys_reg(vcpu, TCR_EL2); 1010 ttbr_elx = (tcr & TCR_A1) ? TTBR1_EL2 : TTBR0_EL2; 1011 break; 1012 default: 1013 BUG(); 1014 } 1015 1016 asid = FIELD_GET(TTBRx_EL1_ASID, vcpu_read_sys_reg(vcpu, ttbr_elx)); 1017 if (!kvm_has_feat_enum(vcpu->kvm, ID_AA64MMFR0_EL1, ASIDBITS, 16) || 1018 !(tcr & TCR_ASID16)) 1019 asid &= GENMASK(7, 0); 1020 1021 return asid; 1022 } 1023 1024 static void invalidate_vncr(struct kvm *kvm, struct vncr_tlb *vt) 1025 { 1026 BUG_ON(!vt->valid); 1027 vt->valid = false; 1028 unmap_l1_vncr(vt); 1029 atomic_dec(&kvm->arch.vncr_tlb_count); 1030 } 1031 1032 static bool vncr_tlb_intersects(struct vncr_tlb *vt, u64 addr, 1033 u64 scope_start, u64 scope_size) 1034 { 1035 u64 tlb_size, tlb_start, tlb_end, scope_end; 1036 1037 tlb_size = ttl_to_size(pgshift_level_to_ttl(vt->wi.pgshift, vt->wr.level)); 1038 1039 tlb_start = addr & ~(tlb_size - 1); 1040 tlb_end = tlb_start + tlb_size - 1; 1041 scope_end = scope_start + scope_size - 1; 1042 1043 return !(tlb_end < scope_start || tlb_start > scope_end); 1044 } 1045 1046 /* 1047 * VNCR TLB invalidation occurs from MMU notifiers or TLBI instructions, and 1048 * either can race against a vcpu not being onlined yet (no pseudo-TLB 1049 * allocated). Similarly, the TLB might be invalid. Skip those, as they 1050 * obviously don't participate in the invalidation at this stage. 1051 */ 1052 #define kvm_for_each_vncr_tlb(idx, vcpup, tlbp, kvm) \ 1053 kvm_for_each_vcpu(idx, vcpup, kvm) \ 1054 if (((tlbp) = vcpup->arch.vncr_tlb) && \ 1055 (tlbp)->valid) 1056 1057 static void kvm_invalidate_vncr_ipa(struct kvm *kvm, u64 start, u64 end) 1058 { 1059 struct kvm_vcpu *vcpu; 1060 struct vncr_tlb *vt; 1061 unsigned long i; 1062 1063 lockdep_assert_held_write(&kvm->mmu_lock); 1064 1065 if (!kvm_has_feat(kvm, ID_AA64MMFR4_EL1, NV_frac, NV2_ONLY)) 1066 return; 1067 1068 /* 1069 * Note that invalidating the VNCR on the back of an MMU notifier 1070 * doesn't require messing with the invalidation counter for a 1071 * parallel walk. The notifier itself will have bumped the counter, 1072 * making sure we rewalk. 1073 */ 1074 kvm_for_each_vncr_tlb(i, vcpu, vt, kvm) 1075 if (vncr_tlb_intersects(vt, vt->wr.pa, start, end - start)) 1076 invalidate_vncr(kvm, vt); 1077 } 1078 1079 struct s1e2_tlbi_scope { 1080 enum { 1081 TLBI_ALL, 1082 TLBI_VA, 1083 TLBI_VAA, 1084 TLBI_ASID, 1085 } type; 1086 1087 u16 asid; 1088 u64 va; 1089 u64 size; 1090 }; 1091 1092 static void invalidate_vncr_va(struct kvm *kvm, 1093 struct s1e2_tlbi_scope *scope) 1094 { 1095 struct kvm_vcpu *vcpu; 1096 struct vncr_tlb *vt; 1097 unsigned long i; 1098 1099 lockdep_assert_held_write(&kvm->mmu_lock); 1100 1101 /* 1102 * We might be performing a parallel S1 walk, so bump up the 1103 * invalidation counter even in the absence of an actual VNCR TLB 1104 * invalidation, as this could indicate that the guest has gone 1105 * through a BBM sequence. 1106 */ 1107 kvm->mmu_invalidate_seq++; 1108 smp_wmb(); 1109 1110 kvm_for_each_vncr_tlb(i, vcpu, vt, kvm) { 1111 switch (scope->type) { 1112 case TLBI_ALL: 1113 break; 1114 1115 case TLBI_VA: 1116 if (!vncr_tlb_intersects(vt, vt->gva, scope->va, scope->size)) 1117 continue; 1118 if (vt->wr.nG && vt->wr.asid != scope->asid) 1119 continue; 1120 break; 1121 1122 case TLBI_VAA: 1123 if (!vncr_tlb_intersects(vt, vt->gva, scope->va, scope->size)) 1124 continue; 1125 break; 1126 1127 case TLBI_ASID: 1128 if (!vt->wr.nG || vt->wr.asid != scope->asid) 1129 continue; 1130 break; 1131 } 1132 1133 invalidate_vncr(kvm, vt); 1134 } 1135 } 1136 1137 #define tlbi_va_s1_to_va(v) (u64)sign_extend64((v) << 12, 48) 1138 1139 static void compute_s1_tlbi_range(struct kvm_vcpu *vcpu, u32 inst, u64 val, 1140 struct s1e2_tlbi_scope *scope) 1141 { 1142 switch (inst) { 1143 case OP_TLBI_ALLE2: 1144 case OP_TLBI_ALLE2IS: 1145 case OP_TLBI_ALLE2OS: 1146 case OP_TLBI_VMALLE1: 1147 case OP_TLBI_VMALLE1IS: 1148 case OP_TLBI_VMALLE1OS: 1149 case OP_TLBI_ALLE2NXS: 1150 case OP_TLBI_ALLE2ISNXS: 1151 case OP_TLBI_ALLE2OSNXS: 1152 case OP_TLBI_VMALLE1NXS: 1153 case OP_TLBI_VMALLE1ISNXS: 1154 case OP_TLBI_VMALLE1OSNXS: 1155 scope->type = TLBI_ALL; 1156 break; 1157 case OP_TLBI_VAE2: 1158 case OP_TLBI_VAE2IS: 1159 case OP_TLBI_VAE2OS: 1160 case OP_TLBI_VAE1: 1161 case OP_TLBI_VAE1IS: 1162 case OP_TLBI_VAE1OS: 1163 case OP_TLBI_VAE2NXS: 1164 case OP_TLBI_VAE2ISNXS: 1165 case OP_TLBI_VAE2OSNXS: 1166 case OP_TLBI_VAE1NXS: 1167 case OP_TLBI_VAE1ISNXS: 1168 case OP_TLBI_VAE1OSNXS: 1169 case OP_TLBI_VALE2: 1170 case OP_TLBI_VALE2IS: 1171 case OP_TLBI_VALE2OS: 1172 case OP_TLBI_VALE1: 1173 case OP_TLBI_VALE1IS: 1174 case OP_TLBI_VALE1OS: 1175 case OP_TLBI_VALE2NXS: 1176 case OP_TLBI_VALE2ISNXS: 1177 case OP_TLBI_VALE2OSNXS: 1178 case OP_TLBI_VALE1NXS: 1179 case OP_TLBI_VALE1ISNXS: 1180 case OP_TLBI_VALE1OSNXS: 1181 scope->type = TLBI_VA; 1182 scope->size = ttl_to_size(FIELD_GET(TLBI_TTL_MASK, val)); 1183 scope->va = tlbi_va_s1_to_va(val) & ~(scope->size - 1); 1184 scope->asid = FIELD_GET(TLBIR_ASID_MASK, val); 1185 break; 1186 case OP_TLBI_ASIDE1: 1187 case OP_TLBI_ASIDE1IS: 1188 case OP_TLBI_ASIDE1OS: 1189 case OP_TLBI_ASIDE1NXS: 1190 case OP_TLBI_ASIDE1ISNXS: 1191 case OP_TLBI_ASIDE1OSNXS: 1192 scope->type = TLBI_ASID; 1193 scope->asid = FIELD_GET(TLBIR_ASID_MASK, val); 1194 break; 1195 case OP_TLBI_VAAE1: 1196 case OP_TLBI_VAAE1IS: 1197 case OP_TLBI_VAAE1OS: 1198 case OP_TLBI_VAAE1NXS: 1199 case OP_TLBI_VAAE1ISNXS: 1200 case OP_TLBI_VAAE1OSNXS: 1201 case OP_TLBI_VAALE1: 1202 case OP_TLBI_VAALE1IS: 1203 case OP_TLBI_VAALE1OS: 1204 case OP_TLBI_VAALE1NXS: 1205 case OP_TLBI_VAALE1ISNXS: 1206 case OP_TLBI_VAALE1OSNXS: 1207 scope->type = TLBI_VAA; 1208 scope->size = ttl_to_size(FIELD_GET(TLBI_TTL_MASK, val)); 1209 scope->va = tlbi_va_s1_to_va(val) & ~(scope->size - 1); 1210 break; 1211 case OP_TLBI_RVAE2: 1212 case OP_TLBI_RVAE2IS: 1213 case OP_TLBI_RVAE2OS: 1214 case OP_TLBI_RVAE1: 1215 case OP_TLBI_RVAE1IS: 1216 case OP_TLBI_RVAE1OS: 1217 case OP_TLBI_RVAE2NXS: 1218 case OP_TLBI_RVAE2ISNXS: 1219 case OP_TLBI_RVAE2OSNXS: 1220 case OP_TLBI_RVAE1NXS: 1221 case OP_TLBI_RVAE1ISNXS: 1222 case OP_TLBI_RVAE1OSNXS: 1223 case OP_TLBI_RVALE2: 1224 case OP_TLBI_RVALE2IS: 1225 case OP_TLBI_RVALE2OS: 1226 case OP_TLBI_RVALE1: 1227 case OP_TLBI_RVALE1IS: 1228 case OP_TLBI_RVALE1OS: 1229 case OP_TLBI_RVALE2NXS: 1230 case OP_TLBI_RVALE2ISNXS: 1231 case OP_TLBI_RVALE2OSNXS: 1232 case OP_TLBI_RVALE1NXS: 1233 case OP_TLBI_RVALE1ISNXS: 1234 case OP_TLBI_RVALE1OSNXS: 1235 scope->type = TLBI_VA; 1236 scope->va = decode_range_tlbi(val, &scope->size, &scope->asid); 1237 break; 1238 case OP_TLBI_RVAAE1: 1239 case OP_TLBI_RVAAE1IS: 1240 case OP_TLBI_RVAAE1OS: 1241 case OP_TLBI_RVAAE1NXS: 1242 case OP_TLBI_RVAAE1ISNXS: 1243 case OP_TLBI_RVAAE1OSNXS: 1244 case OP_TLBI_RVAALE1: 1245 case OP_TLBI_RVAALE1IS: 1246 case OP_TLBI_RVAALE1OS: 1247 case OP_TLBI_RVAALE1NXS: 1248 case OP_TLBI_RVAALE1ISNXS: 1249 case OP_TLBI_RVAALE1OSNXS: 1250 scope->type = TLBI_VAA; 1251 scope->va = decode_range_tlbi(val, &scope->size, NULL); 1252 break; 1253 } 1254 } 1255 1256 void kvm_handle_s1e2_tlbi(struct kvm_vcpu *vcpu, u32 inst, u64 val) 1257 { 1258 struct s1e2_tlbi_scope scope = {}; 1259 1260 compute_s1_tlbi_range(vcpu, inst, val, &scope); 1261 1262 guard(write_lock)(&vcpu->kvm->mmu_lock); 1263 invalidate_vncr_va(vcpu->kvm, &scope); 1264 } 1265 1266 static void kvm_invalidate_vncr_ipa_all(struct kvm *kvm) 1267 { 1268 struct kvm_pgtable *pgt = kvm->arch.mmu.pgt; 1269 1270 lockdep_assert_held_write(&kvm->mmu_lock); 1271 1272 /* if the mmu lock was dropped, pgt teardown may have raced. */ 1273 if (pgt) 1274 kvm_invalidate_vncr_ipa(kvm, 0, BIT(pgt->ia_bits)); 1275 } 1276 1277 void kvm_nested_s2_wp(struct kvm *kvm) 1278 { 1279 int i; 1280 1281 lockdep_assert_held_write(&kvm->mmu_lock); 1282 1283 if (!kvm->arch.nested_mmus_size) 1284 return; 1285 1286 for (i = 0; i < kvm->arch.nested_mmus_size; i++) { 1287 struct kvm_s2_mmu *mmu = kvm->arch.nested_mmus[i]; 1288 1289 if (kvm_s2_mmu_valid(mmu)) 1290 kvm_stage2_wp_range(mmu, 0, kvm_phys_size(mmu)); 1291 } 1292 1293 kvm_invalidate_vncr_ipa_all(kvm); 1294 } 1295 1296 void kvm_nested_s2_unmap(struct kvm *kvm, bool may_block) 1297 { 1298 int i; 1299 1300 lockdep_assert_held_write(&kvm->mmu_lock); 1301 1302 if (!kvm->arch.nested_mmus_size) 1303 return; 1304 1305 for (i = 0; i < kvm->arch.nested_mmus_size; i++) { 1306 struct kvm_s2_mmu *mmu = kvm->arch.nested_mmus[i]; 1307 1308 if (kvm_s2_mmu_valid(mmu)) 1309 kvm_stage2_unmap_range(mmu, 0, kvm_phys_size(mmu), may_block); 1310 } 1311 1312 kvm_invalidate_vncr_ipa_all(kvm); 1313 } 1314 1315 void kvm_nested_s2_flush(struct kvm *kvm) 1316 { 1317 int i; 1318 1319 lockdep_assert_held_write(&kvm->mmu_lock); 1320 1321 if (!kvm->arch.nested_mmus_size) 1322 return; 1323 1324 for (i = 0; i < kvm->arch.nested_mmus_size; i++) { 1325 struct kvm_s2_mmu *mmu = kvm->arch.nested_mmus[i]; 1326 1327 if (kvm_s2_mmu_valid(mmu)) 1328 kvm_stage2_flush_range(mmu, 0, kvm_phys_size(mmu)); 1329 } 1330 } 1331 1332 void kvm_arch_flush_shadow_all(struct kvm *kvm) 1333 { 1334 for (int i = 0; i < kvm->arch.nested_mmus_size; i++) { 1335 struct kvm_s2_mmu *mmu = kvm->arch.nested_mmus[i]; 1336 1337 if (!WARN_ON(atomic_read(&mmu->refcnt))) 1338 kvm_free_stage2_pgd(mmu); 1339 } 1340 kvm_uninit_stage2_mmu(kvm); 1341 } 1342 1343 /* 1344 * Dealing with VNCR_EL2 exposed by the *guest* is a complicated matter: 1345 * 1346 * - We introduce an internal representation of a vcpu-private TLB, 1347 * representing the mapping between the guest VA contained in VNCR_EL2, 1348 * the IPA the guest's EL2 PTs point to, and the actual PA this lives at. 1349 * 1350 * - On translation fault from a nested VNCR access, we create such a TLB. 1351 * If there is no mapping to describe, the guest inherits the fault. 1352 * Crucially, no actual mapping is done at this stage. 1353 * 1354 * - On vcpu_load() in a non-HYP context with HCR_EL2.NV==1, if the above 1355 * TLB exists, we map it in the fixmap for this CPU, and run with it. We 1356 * have to respect the permissions dictated by the guest, but not the 1357 * memory type (FWB is a must). 1358 * 1359 * - Note that we usually don't do a vcpu_load() on the back of a fault 1360 * (unless we are preempted), so the resolution of a translation fault 1361 * must go via a request that will map the VNCR page in the fixmap. 1362 * vcpu_load() might as well use the same mechanism. 1363 * 1364 * - On vcpu_put() in a non-HYP context with HCR_EL2.NV==1, if the TLB was 1365 * mapped, we unmap it. Yes it is that simple. The TLB still exists 1366 * though, and may be reused at a later load. 1367 * 1368 * - On permission fault, we simply forward the fault to the guest's EL2. 1369 * Get out of my way. 1370 * 1371 * - On any TLBI for the EL2&0 translation regime, we must find any TLB that 1372 * intersects with the TLBI request, invalidate it, and unmap the page 1373 * from the fixmap. Because we need to look at all the vcpu-private TLBs, 1374 * this requires some wide-ranging locking to ensure that nothing races 1375 * against it. This requires some refcounting to avoid the search when 1376 * no such TLB is present (see below). 1377 * 1378 * - On MMU notifiers, we must invalidate our TLB in a similar way, but 1379 * looking at the IPA instead. The funny part is that there may not be a 1380 * stage-2 mapping for this page if L1 hasn't accessed it using LD/ST 1381 * instructions. 1382 * 1383 * - vncr_tlb_count tracks the number of valid VNCR TLBs VM-wide. This isn't 1384 * the number of *mapped* L1 VNCR pages, which is likely be a subset (and 1385 * by definition, a TLBI handled from L1 runs with the canonical VNCR 1386 * page, not the L1's). The innermost trap handling code checks this to 1387 * find out whether to return to the guest ASAP (no L1 TLBs) or to visit 1388 * this part of the world for some extra invalidation work. 1389 */ 1390 1391 int kvm_vcpu_allocate_vncr_tlb(struct kvm_vcpu *vcpu) 1392 { 1393 if (!kvm_has_feat(vcpu->kvm, ID_AA64MMFR4_EL1, NV_frac, NV2_ONLY)) 1394 return 0; 1395 1396 if (!vcpu->arch.vncr_tlb) { 1397 struct vncr_tlb *vt = kzalloc_obj(*vcpu->arch.vncr_tlb, 1398 GFP_KERNEL_ACCOUNT); 1399 1400 /* 1401 * Taking the lock on assignment ensures that the TLB is 1402 * seen as initialised when following the pointer (release 1403 * semantics of the unlock), and avoids having acquires on 1404 * each user which already take the lock. 1405 */ 1406 scoped_guard(write_lock, &vcpu->kvm->mmu_lock) 1407 vcpu->arch.vncr_tlb = vt; 1408 } 1409 1410 if (!vcpu->arch.vncr_tlb) 1411 return -ENOMEM; 1412 1413 return 0; 1414 } 1415 1416 static u64 read_vncr_el2(struct kvm_vcpu *vcpu) 1417 { 1418 return (u64)sign_extend64(__vcpu_sys_reg(vcpu, VNCR_EL2), 48); 1419 } 1420 1421 static int kvm_translate_vncr(struct kvm_vcpu *vcpu, bool *is_gmem) 1422 { 1423 struct kvm_memory_slot *memslot; 1424 bool write_fault, writable; 1425 unsigned long mmu_seq; 1426 struct vncr_tlb *vt; 1427 struct page *page; 1428 u64 va, pfn, gfn; 1429 int ret; 1430 1431 vt = vcpu->arch.vncr_tlb; 1432 1433 /* 1434 * If we're about to walk the EL2 S1 PTs, we must invalidate the 1435 * current TLB, as it could be sampled from another vcpu doing a 1436 * TLBI *IS. A real CPU wouldn't do that, but we only keep a single 1437 * translation, so not much of a choice. 1438 * 1439 * We also prepare the next walk wilst we're at it. 1440 */ 1441 scoped_guard(write_lock, &vcpu->kvm->mmu_lock) { 1442 this_cpu_reset_vncr_fixmap(vcpu); 1443 if (vt->valid) 1444 invalidate_vncr(vcpu->kvm, vt); 1445 1446 vt->wi = (struct s1_walk_info) { 1447 .regime = TR_EL20, 1448 .as_el0 = false, 1449 .pan = false, 1450 }; 1451 vt->wr = (struct s1_walk_result){}; 1452 } 1453 1454 guard(srcu)(&vcpu->kvm->srcu); 1455 1456 va = read_vncr_el2(vcpu); 1457 1458 mmu_seq = vcpu->kvm->mmu_invalidate_seq; 1459 smp_rmb(); 1460 1461 ret = __kvm_translate_va(vcpu, &vt->wi, &vt->wr, va); 1462 if (ret) 1463 return ret; 1464 1465 write_fault = kvm_is_write_fault(vcpu); 1466 1467 gfn = vt->wr.pa >> PAGE_SHIFT; 1468 memslot = gfn_to_memslot(vcpu->kvm, gfn); 1469 if (!memslot) { 1470 fail_s1_walk(&vt->wr, ESR_ELx_FSC_EXTABT, false); 1471 return -EFAULT; 1472 } 1473 1474 *is_gmem = kvm_slot_has_gmem(memslot); 1475 if (!*is_gmem) { 1476 pfn = __kvm_faultin_pfn(memslot, gfn, write_fault ? FOLL_WRITE : 0, 1477 &writable, &page); 1478 if (is_error_noslot_pfn(pfn)) { 1479 fail_s1_walk(&vt->wr, ESR_ELx_FSC_EXTABT, false); 1480 return -EFAULT; 1481 } 1482 } else { 1483 ret = kvm_gmem_get_pfn(vcpu->kvm, memslot, gfn, &pfn, &page, NULL); 1484 if (ret) { 1485 kvm_prepare_memory_fault_exit(vcpu, vt->wr.pa, PAGE_SIZE, 1486 write_fault, false, false); 1487 return ret; 1488 } 1489 1490 writable = !(memslot->flags & KVM_MEM_READONLY); 1491 } 1492 1493 /* 1494 * FIXME: This check is too restrictive as KVM allows cacheable memory 1495 * attributes for PFNMAP VMAs that have cacheable attributes in host 1496 * stage-1. 1497 */ 1498 if (!pfn_is_map_memory(pfn)) { 1499 kvm_release_faultin_page(vcpu->kvm, page, true, false); 1500 fail_s1_walk(&vt->wr, ESR_ELx_FSC_EXTABT, false); 1501 return -EINVAL; 1502 } 1503 1504 scoped_guard(write_lock, &vcpu->kvm->mmu_lock) { 1505 if (mmu_invalidate_retry(vcpu->kvm, mmu_seq)) { 1506 kvm_release_faultin_page(vcpu->kvm, page, true, false); 1507 return -EAGAIN; 1508 } 1509 1510 vt->gva = va; 1511 vt->hpa = pfn << PAGE_SHIFT; 1512 vt->hpa_writable = writable; 1513 vt->valid = true; 1514 atomic_set(&vt->cpu, -1); 1515 1516 kvm_make_request(KVM_REQ_MAP_L1_VNCR_EL2, vcpu); 1517 kvm_release_faultin_page(vcpu->kvm, page, false, vt->wr.pw && vt->hpa_writable); 1518 } 1519 1520 if (vt->wr.pw && vt->hpa_writable) 1521 mark_page_dirty(vcpu->kvm, gfn); 1522 1523 return 0; 1524 } 1525 1526 static void handle_vncr_perm(struct kvm_vcpu *vcpu) 1527 { 1528 struct vncr_tlb *vt = vcpu->arch.vncr_tlb; 1529 u64 esr = kvm_vcpu_get_esr(vcpu); 1530 u64 fsc; 1531 1532 /* 1533 * Promote to an external abort if the stage-1 permits writes but the 1534 * HPA is read-only (e.g. RO memslot). 1535 */ 1536 if (kvm_is_write_fault(vcpu) && vt->wr.pw && !vt->hpa_writable) 1537 fsc = ESR_ELx_FSC_EXTABT; 1538 /* 1539 * Otherwise, inject a permission fault using the guest's translation 1540 * level rather than the host's. 1541 */ 1542 else 1543 fsc = ESR_ELx_FSC_PERM_L(vt->wr.level); 1544 1545 esr &= ~ESR_ELx_FSC; 1546 esr |= FIELD_PREP(ESR_ELx_FSC, fsc); 1547 1548 kvm_inject_nested_sync(vcpu, esr); 1549 } 1550 1551 int kvm_handle_vncr_abort(struct kvm_vcpu *vcpu) 1552 { 1553 struct vncr_tlb *vt = vcpu->arch.vncr_tlb; 1554 u64 esr = kvm_vcpu_get_esr(vcpu); 1555 bool is_gmem = false; 1556 bool perm; 1557 int ret; 1558 1559 WARN_ON_ONCE(!(esr & ESR_ELx_VNCR)); 1560 1561 if (kvm_vcpu_abt_issea(vcpu)) 1562 return kvm_handle_guest_sea(vcpu); 1563 1564 if (!esr_fsc_is_translation_fault(esr) && !esr_fsc_is_permission_fault(esr)) { 1565 KVM_BUG(1, vcpu->kvm, "Unhandled VNCR abort, ESR=%llx\n", esr); 1566 return -EIO; 1567 } 1568 1569 /* 1570 * Speculatively increment the TLB count to make sure concurrent 1571 * TLBIs will take the slow path, and will interact with the retry 1572 * mechanism. Drop it again on error. 1573 */ 1574 atomic_inc(&vcpu->kvm->arch.vncr_tlb_count); 1575 smp_mb__after_atomic(); 1576 1577 ret = kvm_translate_vncr(vcpu, &is_gmem); 1578 if (ret) { 1579 smp_mb__before_atomic(); 1580 atomic_dec(&vcpu->kvm->arch.vncr_tlb_count); 1581 } 1582 1583 switch (ret) { 1584 case -EAGAIN: 1585 /* Let's try again... */ 1586 return 1; 1587 case -ENOMEM: 1588 /* 1589 * For guest_memfd, this indicates that it failed to 1590 * create a folio to back the memory. Inform userspace. 1591 */ 1592 if (is_gmem) 1593 return 0; 1594 /* Otherwise, let's try again... */ 1595 break; 1596 case -EFAULT: 1597 case -EIO: 1598 case -EHWPOISON: 1599 if (is_gmem) 1600 return 0; 1601 fallthrough; 1602 case -EINVAL: 1603 case -ENOENT: 1604 case -EACCES: 1605 /* 1606 * Translation failed, inject the corresponding 1607 * exception back to EL2. 1608 */ 1609 esr &= ~ESR_ELx_FSC; 1610 esr |= FIELD_PREP(ESR_ELx_FSC, vt->wr.fst); 1611 1612 kvm_inject_nested_sync(vcpu, esr); 1613 break; 1614 case 0: 1615 perm = kvm_is_write_fault(vcpu) ? vt->wr.pw && vt->hpa_writable : vt->wr.pr; 1616 if (!perm) 1617 handle_vncr_perm(vcpu); 1618 break; 1619 } 1620 1621 return 1; 1622 } 1623 1624 static void kvm_map_l1_vncr(struct kvm_vcpu *vcpu) 1625 { 1626 struct vncr_tlb *vt = vcpu->arch.vncr_tlb; 1627 pgprot_t prot; 1628 1629 guard(preempt)(); 1630 guard(read_lock)(&vcpu->kvm->mmu_lock); 1631 1632 /* 1633 * The request to map VNCR may have raced against some other 1634 * event, such as an interrupt, and may not be valid anymore. 1635 */ 1636 if (is_hyp_ctxt(vcpu)) 1637 return; 1638 1639 /* 1640 * Check that the pseudo-TLB is valid and that VNCR_EL2 still 1641 * contains the expected value. If it doesn't, we simply bail out 1642 * without a mapping -- a transformed MSR/MRS will generate the 1643 * fault and allows us to populate the pseudo-TLB. 1644 */ 1645 if (!vt->valid) 1646 return; 1647 1648 /* We cache the MMU state in the TLB. Check that it matches. */ 1649 if (!!(vcpu_read_sys_reg(vcpu, SCTLR_EL2) & SCTLR_ELx_M) != s1_walk_translated(&vt->wr)) 1650 return; 1651 1652 if (read_vncr_el2(vcpu) != vt->gva) 1653 return; 1654 1655 if (vt->wr.nG && get_asid_by_regime(vcpu, TR_EL20) != vt->wr.asid) 1656 return; 1657 1658 if (vt->hpa_writable && vt->wr.pw && vt->wr.pr) 1659 prot = PAGE_KERNEL; 1660 else if (vt->wr.pr) 1661 prot = PAGE_KERNEL_RO; 1662 else 1663 prot = PAGE_NONE; 1664 1665 /* 1666 * We can't map write-only (or no permission at all) in the kernel, 1667 * but the guest can do it if using POE, so we'll have to turn a 1668 * translation fault into a permission fault at runtime. 1669 * FIXME: WO doesn't work at all, need POE support in the kernel. 1670 */ 1671 if (pgprot_val(prot) != pgprot_val(PAGE_NONE)) { 1672 atomic_set(&vt->cpu, smp_processor_id()); 1673 __set_fixmap(vncr_fixmap(atomic_read(&vt->cpu)), vt->hpa, prot); 1674 host_data_set_flag(L1_VNCR_MAPPED); 1675 } 1676 } 1677 1678 /* 1679 * Our emulated CPU doesn't support all the possible features. For the 1680 * sake of simplicity (and probably mental sanity), wipe out a number 1681 * of feature bits we don't intend to support for the time being. 1682 * This list should get updated as new features get added to the NV 1683 * support, and new extension to the architecture. 1684 */ 1685 u64 limit_nv_id_reg(struct kvm *kvm, u32 reg, u64 val) 1686 { 1687 u64 orig_val = val; 1688 1689 switch (reg) { 1690 case SYS_ID_AA64ISAR1_EL1: 1691 /* Support everything but LS64 and Spec Invalidation */ 1692 val &= ~(ID_AA64ISAR1_EL1_LS64 | 1693 ID_AA64ISAR1_EL1_SPECRES); 1694 break; 1695 1696 case SYS_ID_AA64PFR0_EL1: 1697 /* No RME, AMU, MPAM, or S-EL2 */ 1698 val &= ~(ID_AA64PFR0_EL1_RME | 1699 ID_AA64PFR0_EL1_AMU | 1700 ID_AA64PFR0_EL1_MPAM | 1701 ID_AA64PFR0_EL1_SEL2 | 1702 ID_AA64PFR0_EL1_EL3 | 1703 ID_AA64PFR0_EL1_EL2 | 1704 ID_AA64PFR0_EL1_EL1 | 1705 ID_AA64PFR0_EL1_EL0); 1706 /* 64bit only at any EL */ 1707 val |= SYS_FIELD_PREP_ENUM(ID_AA64PFR0_EL1, EL0, IMP); 1708 val |= SYS_FIELD_PREP_ENUM(ID_AA64PFR0_EL1, EL1, IMP); 1709 val |= SYS_FIELD_PREP_ENUM(ID_AA64PFR0_EL1, EL2, IMP); 1710 val |= SYS_FIELD_PREP_ENUM(ID_AA64PFR0_EL1, EL3, IMP); 1711 break; 1712 1713 case SYS_ID_AA64PFR1_EL1: 1714 /* Only support BTI, SSBS, CSV2_frac */ 1715 val &= ~(ID_AA64PFR1_EL1_PFAR | 1716 ID_AA64PFR1_EL1_MTEX | 1717 ID_AA64PFR1_EL1_THE | 1718 ID_AA64PFR1_EL1_GCS | 1719 ID_AA64PFR1_EL1_MTE_frac | 1720 ID_AA64PFR1_EL1_NMI | 1721 ID_AA64PFR1_EL1_SME | 1722 ID_AA64PFR1_EL1_RES0 | 1723 ID_AA64PFR1_EL1_MPAM_frac | 1724 ID_AA64PFR1_EL1_MTE); 1725 break; 1726 1727 case SYS_ID_AA64PFR2_EL1: 1728 /* GICv5 is not yet supported for NV */ 1729 val &= ~ID_AA64PFR2_EL1_GCIE; 1730 break; 1731 1732 case SYS_ID_AA64MMFR0_EL1: 1733 /* Hide ExS, Secure Memory */ 1734 val &= ~(ID_AA64MMFR0_EL1_EXS | 1735 ID_AA64MMFR0_EL1_TGRAN4_2 | 1736 ID_AA64MMFR0_EL1_TGRAN16_2 | 1737 ID_AA64MMFR0_EL1_TGRAN64_2 | 1738 ID_AA64MMFR0_EL1_SNSMEM); 1739 1740 /* Hide CNTPOFF if present */ 1741 val = ID_REG_LIMIT_FIELD_ENUM(val, ID_AA64MMFR0_EL1, ECV, IMP); 1742 1743 /* Disallow unsupported S2 page sizes */ 1744 switch (PAGE_SIZE) { 1745 case SZ_64K: 1746 val |= SYS_FIELD_PREP_ENUM(ID_AA64MMFR0_EL1, TGRAN16_2, NI); 1747 fallthrough; 1748 case SZ_16K: 1749 val |= SYS_FIELD_PREP_ENUM(ID_AA64MMFR0_EL1, TGRAN4_2, NI); 1750 fallthrough; 1751 case SZ_4K: 1752 /* Support everything */ 1753 break; 1754 } 1755 1756 /* 1757 * Since we can't support a guest S2 page size smaller 1758 * than the host's own page size (due to KVM only 1759 * populating its own S2 using the kernel's page 1760 * size), advertise the limitation using FEAT_GTG. 1761 */ 1762 switch (PAGE_SIZE) { 1763 case SZ_4K: 1764 if (_has_tgran_2(orig_val, 4)) 1765 val |= SYS_FIELD_PREP_ENUM(ID_AA64MMFR0_EL1, TGRAN4_2, IMP); 1766 fallthrough; 1767 case SZ_16K: 1768 if (_has_tgran_2(orig_val, 16)) 1769 val |= SYS_FIELD_PREP_ENUM(ID_AA64MMFR0_EL1, TGRAN16_2, IMP); 1770 fallthrough; 1771 case SZ_64K: 1772 if (_has_tgran_2(orig_val, 64)) 1773 val |= SYS_FIELD_PREP_ENUM(ID_AA64MMFR0_EL1, TGRAN64_2, IMP); 1774 break; 1775 } 1776 1777 /* Cap PARange to 48bits */ 1778 val = ID_REG_LIMIT_FIELD_ENUM(val, ID_AA64MMFR0_EL1, PARANGE, 48); 1779 break; 1780 1781 case SYS_ID_AA64MMFR1_EL1: 1782 val &= ~(ID_AA64MMFR1_EL1_CMOW | 1783 ID_AA64MMFR1_EL1_nTLBPA | 1784 ID_AA64MMFR1_EL1_ETS); 1785 1786 /* FEAT_E2H0 implies no VHE */ 1787 if (test_bit(KVM_ARM_VCPU_HAS_EL2_E2H0, kvm->arch.vcpu_features)) 1788 val &= ~ID_AA64MMFR1_EL1_VH; 1789 1790 val = ID_REG_LIMIT_FIELD_ENUM(val, ID_AA64MMFR1_EL1, HAFDBS, AF); 1791 break; 1792 1793 case SYS_ID_AA64MMFR2_EL1: 1794 val &= ~(ID_AA64MMFR2_EL1_BBM | 1795 ID_AA64MMFR2_EL1_TTL | 1796 GENMASK_ULL(47, 44) | 1797 ID_AA64MMFR2_EL1_ST | 1798 ID_AA64MMFR2_EL1_CCIDX | 1799 ID_AA64MMFR2_EL1_VARange); 1800 1801 /* Force TTL support */ 1802 val |= SYS_FIELD_PREP_ENUM(ID_AA64MMFR2_EL1, TTL, IMP); 1803 break; 1804 1805 case SYS_ID_AA64MMFR4_EL1: 1806 /* 1807 * You get EITHER 1808 * 1809 * - FEAT_VHE without FEAT_E2H0 1810 * - FEAT_NV limited to FEAT_NV2(p1)/NV3 1811 * - HCR_EL2.NV1 being RES0 1812 * 1813 * OR 1814 * 1815 * - FEAT_E2H0 without FEAT_VHE nor FEAT_NV 1816 * 1817 * Life is too short for anything else. 1818 */ 1819 if (test_bit(KVM_ARM_VCPU_HAS_EL2_E2H0, kvm->arch.vcpu_features)) { 1820 val = 0; 1821 } else { 1822 val &= ID_AA64MMFR4_EL1_NV_frac; 1823 if (cpus_have_final_cap(ARM64_HAS_NV3)) 1824 val = ID_REG_LIMIT_FIELD_ENUM(val, ID_AA64MMFR4_EL1, NV_frac, NV3); 1825 else if (cpus_have_final_cap(ARM64_HAS_NV2P1)) 1826 val = ID_REG_LIMIT_FIELD_ENUM(val, ID_AA64MMFR4_EL1, NV_frac, NV2P1); 1827 else 1828 val = SYS_FIELD_PREP_ENUM(ID_AA64MMFR4_EL1, NV_frac, NV2_ONLY); 1829 val |= SYS_FIELD_PREP_ENUM(ID_AA64MMFR4_EL1, E2H0, NI_NV1); 1830 } 1831 break; 1832 1833 case SYS_ID_AA64DFR0_EL1: 1834 /* Only limited support for PMU, Debug, BPs, WPs, and HPMN0 */ 1835 val &= ~(ID_AA64DFR0_EL1_ExtTrcBuff | 1836 ID_AA64DFR0_EL1_BRBE | 1837 ID_AA64DFR0_EL1_MTPMU | 1838 ID_AA64DFR0_EL1_TraceBuffer | 1839 ID_AA64DFR0_EL1_TraceFilt | 1840 ID_AA64DFR0_EL1_PMSVer | 1841 ID_AA64DFR0_EL1_CTX_CMPs | 1842 ID_AA64DFR0_EL1_SEBEP | 1843 ID_AA64DFR0_EL1_PMSS | 1844 ID_AA64DFR0_EL1_TraceVer); 1845 1846 /* 1847 * FEAT_Debugv8p9 requires support for extended breakpoints / 1848 * watchpoints. 1849 */ 1850 val = ID_REG_LIMIT_FIELD_ENUM(val, ID_AA64DFR0_EL1, DebugVer, V8P8); 1851 break; 1852 } 1853 1854 return val; 1855 } 1856 1857 u64 kvm_vcpu_apply_reg_masks(const struct kvm_vcpu *vcpu, 1858 enum vcpu_sysreg sr, u64 v) 1859 { 1860 struct resx resx; 1861 1862 resx = kvm_get_sysreg_resx(vcpu->kvm, sr); 1863 v &= ~resx.res0; 1864 v |= resx.res1; 1865 1866 return v; 1867 } 1868 1869 static __always_inline void set_sysreg_masks(struct kvm *kvm, int sr, struct resx resx) 1870 { 1871 BUILD_BUG_ON(!__builtin_constant_p(sr)); 1872 BUILD_BUG_ON(sr < __SANITISED_REG_START__); 1873 BUILD_BUG_ON(sr >= NR_SYS_REGS); 1874 1875 kvm_set_sysreg_resx(kvm, sr, resx); 1876 } 1877 1878 int kvm_init_nv_sysregs(struct kvm_vcpu *vcpu) 1879 { 1880 struct kvm *kvm = vcpu->kvm; 1881 struct resx resx; 1882 1883 lockdep_assert_held(&kvm->arch.config_lock); 1884 1885 if (kvm->arch.sysreg_masks) 1886 goto out; 1887 1888 kvm->arch.sysreg_masks = kzalloc_obj(*(kvm->arch.sysreg_masks), 1889 GFP_KERNEL_ACCOUNT); 1890 if (!kvm->arch.sysreg_masks) 1891 return -ENOMEM; 1892 1893 /* VTTBR_EL2 */ 1894 resx = (typeof(resx)){}; 1895 if (!kvm_has_feat_enum(kvm, ID_AA64MMFR1_EL1, VMIDBits, 16)) 1896 resx.res0 |= GENMASK(63, 56); 1897 if (!kvm_has_feat(kvm, ID_AA64MMFR2_EL1, CnP, IMP)) 1898 resx.res0 |= VTTBR_CNP_BIT; 1899 set_sysreg_masks(kvm, VTTBR_EL2, resx); 1900 1901 /* VTCR_EL2 */ 1902 resx = get_reg_fixed_bits(kvm, VTCR_EL2); 1903 set_sysreg_masks(kvm, VTCR_EL2, resx); 1904 1905 /* VMPIDR_EL2 */ 1906 resx.res0 = GENMASK(63, 40) | GENMASK(30, 24); 1907 resx.res1 = BIT(31); 1908 set_sysreg_masks(kvm, VMPIDR_EL2, resx); 1909 1910 /* HCR_EL2 */ 1911 resx = get_reg_fixed_bits(kvm, HCR_EL2); 1912 set_sysreg_masks(kvm, HCR_EL2, resx); 1913 1914 /* NVHCR_EL2 */ 1915 resx = get_reg_fixed_bits(kvm, NVHCR_EL2); 1916 set_sysreg_masks(kvm, NVHCR_EL2, resx); 1917 1918 /* HCRX_EL2 */ 1919 resx = get_reg_fixed_bits(kvm, HCRX_EL2); 1920 set_sysreg_masks(kvm, HCRX_EL2, resx); 1921 1922 /* HFG[RW]TR_EL2 */ 1923 resx = get_reg_fixed_bits(kvm, HFGRTR_EL2); 1924 set_sysreg_masks(kvm, HFGRTR_EL2, resx); 1925 resx = get_reg_fixed_bits(kvm, HFGWTR_EL2); 1926 set_sysreg_masks(kvm, HFGWTR_EL2, resx); 1927 1928 /* HDFG[RW]TR_EL2 */ 1929 resx = get_reg_fixed_bits(kvm, HDFGRTR_EL2); 1930 set_sysreg_masks(kvm, HDFGRTR_EL2, resx); 1931 resx = get_reg_fixed_bits(kvm, HDFGWTR_EL2); 1932 set_sysreg_masks(kvm, HDFGWTR_EL2, resx); 1933 1934 /* HFGITR_EL2 */ 1935 resx = get_reg_fixed_bits(kvm, HFGITR_EL2); 1936 set_sysreg_masks(kvm, HFGITR_EL2, resx); 1937 1938 /* HAFGRTR_EL2 - not a lot to see here */ 1939 resx = get_reg_fixed_bits(kvm, HAFGRTR_EL2); 1940 set_sysreg_masks(kvm, HAFGRTR_EL2, resx); 1941 1942 /* HFG[RW]TR2_EL2 */ 1943 resx = get_reg_fixed_bits(kvm, HFGRTR2_EL2); 1944 set_sysreg_masks(kvm, HFGRTR2_EL2, resx); 1945 resx = get_reg_fixed_bits(kvm, HFGWTR2_EL2); 1946 set_sysreg_masks(kvm, HFGWTR2_EL2, resx); 1947 1948 /* HDFG[RW]TR2_EL2 */ 1949 resx = get_reg_fixed_bits(kvm, HDFGRTR2_EL2); 1950 set_sysreg_masks(kvm, HDFGRTR2_EL2, resx); 1951 resx = get_reg_fixed_bits(kvm, HDFGWTR2_EL2); 1952 set_sysreg_masks(kvm, HDFGWTR2_EL2, resx); 1953 1954 /* HFGITR2_EL2 */ 1955 resx = get_reg_fixed_bits(kvm, HFGITR2_EL2); 1956 set_sysreg_masks(kvm, HFGITR2_EL2, resx); 1957 1958 /* TCR2_EL2 */ 1959 resx = get_reg_fixed_bits(kvm, TCR2_EL2); 1960 set_sysreg_masks(kvm, TCR2_EL2, resx); 1961 1962 /* SCTLR_EL1 */ 1963 resx = get_reg_fixed_bits(kvm, SCTLR_EL1); 1964 set_sysreg_masks(kvm, SCTLR_EL1, resx); 1965 1966 /* SCTLR_EL2 */ 1967 resx = get_reg_fixed_bits(kvm, SCTLR_EL2); 1968 set_sysreg_masks(kvm, SCTLR_EL2, resx); 1969 1970 /* SCTLR2_ELx */ 1971 resx = get_reg_fixed_bits(kvm, SCTLR2_EL1); 1972 set_sysreg_masks(kvm, SCTLR2_EL1, resx); 1973 resx = get_reg_fixed_bits(kvm, SCTLR2_EL2); 1974 set_sysreg_masks(kvm, SCTLR2_EL2, resx); 1975 1976 /* MDCR_EL2 */ 1977 resx = get_reg_fixed_bits(kvm, MDCR_EL2); 1978 set_sysreg_masks(kvm, MDCR_EL2, resx); 1979 1980 /* CNTHCTL_EL2 */ 1981 resx.res0 = GENMASK(63, 20); 1982 resx.res1 = 0; 1983 if (!kvm_has_feat(kvm, ID_AA64PFR0_EL1, RME, IMP)) 1984 resx.res0 |= CNTHCTL_CNTPMASK | CNTHCTL_CNTVMASK; 1985 if (!kvm_has_feat(kvm, ID_AA64MMFR0_EL1, ECV, CNTPOFF)) { 1986 resx.res0 |= CNTHCTL_ECV; 1987 if (!kvm_has_feat(kvm, ID_AA64MMFR0_EL1, ECV, IMP)) 1988 resx.res0 |= (CNTHCTL_EL1TVT | CNTHCTL_EL1TVCT | 1989 CNTHCTL_EL1NVPCT | CNTHCTL_EL1NVVCT); 1990 } 1991 if (!kvm_has_feat(kvm, ID_AA64MMFR1_EL1, VH, IMP)) 1992 resx.res0 |= GENMASK(11, 8); 1993 set_sysreg_masks(kvm, CNTHCTL_EL2, resx); 1994 1995 /* ICH_HCR_EL2 */ 1996 resx.res0 = ICH_HCR_EL2_RES0; 1997 resx.res1 = ICH_HCR_EL2_RES1; 1998 if (!(vgic_ich_vtr() & ICH_VTR_EL2_TDS)) 1999 resx.res0 |= ICH_HCR_EL2_TDIR; 2000 /* No GICv4 is presented to the guest */ 2001 resx.res0 |= ICH_HCR_EL2_DVIM | ICH_HCR_EL2_vSGIEOICount; 2002 set_sysreg_masks(kvm, ICH_HCR_EL2, resx); 2003 2004 /* VNCR_EL2 */ 2005 resx.res0 = VNCR_EL2_RES0; 2006 resx.res1 = VNCR_EL2_RES1; 2007 set_sysreg_masks(kvm, VNCR_EL2, resx); 2008 2009 /* ZCR_EL2 - bits 8:4 are RAZ/WI so treat them as RES0 */ 2010 resx.res0 = ZCR_ELx_RES0 | GENMASK_ULL(8, 4); 2011 resx.res1 = ZCR_ELx_RES1; 2012 set_sysreg_masks(kvm, ZCR_EL2, resx); 2013 2014 out: 2015 for (enum vcpu_sysreg sr = __SANITISED_REG_START__; sr < NR_SYS_REGS; sr++) 2016 __vcpu_rmw_sys_reg(vcpu, sr, |=, 0); 2017 2018 return 0; 2019 } 2020 2021 void check_nested_vcpu_requests(struct kvm_vcpu *vcpu) 2022 { 2023 if (kvm_check_request(KVM_REQ_NESTED_S2_UNMAP, vcpu)) { 2024 struct kvm_s2_mmu *mmu = vcpu->arch.hw_mmu; 2025 2026 write_lock(&vcpu->kvm->mmu_lock); 2027 if (mmu->pending_unmap) { 2028 kvm_stage2_unmap_range(mmu, 0, kvm_phys_size(mmu), true); 2029 mmu->pending_unmap = false; 2030 } 2031 write_unlock(&vcpu->kvm->mmu_lock); 2032 } 2033 2034 if (kvm_check_request(KVM_REQ_MAP_L1_VNCR_EL2, vcpu)) 2035 kvm_map_l1_vncr(vcpu); 2036 2037 /* Must be last, as may switch context! */ 2038 if (kvm_check_request(KVM_REQ_GUEST_HYP_IRQ_PENDING, vcpu)) 2039 kvm_inject_nested_irq(vcpu); 2040 } 2041 2042 /* 2043 * One of the many architectural bugs in FEAT_NV2 is that the guest hypervisor 2044 * can write to HCR_EL2 behind our back, potentially changing the exception 2045 * routing / masking for even the host context. 2046 * 2047 * What follows is some slop to (1) react to exception routing / masking and (2) 2048 * preserve the pending SError state across translation regimes. 2049 */ 2050 void kvm_nested_flush_hwstate(struct kvm_vcpu *vcpu) 2051 { 2052 if (!vcpu_has_nv(vcpu)) 2053 return; 2054 2055 if (unlikely(vcpu_test_and_clear_flag(vcpu, NESTED_SERROR_PENDING))) 2056 kvm_inject_serror_esr(vcpu, vcpu_get_vsesr(vcpu)); 2057 } 2058 2059 void kvm_nested_sync_hwstate(struct kvm_vcpu *vcpu) 2060 { 2061 unsigned long *hcr = vcpu_hcr(vcpu); 2062 2063 if (!vcpu_has_nv(vcpu)) 2064 return; 2065 2066 /* 2067 * We previously decided that an SError was deliverable to the guest. 2068 * Reap the pending state from HCR_EL2 and... 2069 */ 2070 if (unlikely(__test_and_clear_bit(__ffs(HCR_VSE), hcr))) 2071 vcpu_set_flag(vcpu, NESTED_SERROR_PENDING); 2072 2073 /* 2074 * Re-attempt SError injection in case the deliverability has changed, 2075 * which is necessary to faithfully emulate WFI the case of a pending 2076 * SError being a wakeup condition. 2077 */ 2078 if (unlikely(vcpu_test_and_clear_flag(vcpu, NESTED_SERROR_PENDING))) 2079 kvm_inject_serror_esr(vcpu, vcpu_get_vsesr(vcpu)); 2080 } 2081 2082 /* 2083 * KVM unconditionally sets most of these traps anyway but use an allowlist 2084 * to document the guest hypervisor traps that may take precedence and guard 2085 * against future changes to the non-nested trap configuration. 2086 */ 2087 #define NV_MDCR_GUEST_INCLUDE (MDCR_EL2_TDE | \ 2088 MDCR_EL2_TDA | \ 2089 MDCR_EL2_TDRA | \ 2090 MDCR_EL2_TTRF | \ 2091 MDCR_EL2_TPMS | \ 2092 MDCR_EL2_TPM | \ 2093 MDCR_EL2_TPMCR | \ 2094 MDCR_EL2_TDCC | \ 2095 MDCR_EL2_TDOSA) 2096 2097 void kvm_nested_setup_mdcr_el2(struct kvm_vcpu *vcpu) 2098 { 2099 u64 guest_mdcr = __vcpu_sys_reg(vcpu, MDCR_EL2); 2100 2101 if (is_nested_ctxt(vcpu)) 2102 vcpu->arch.mdcr_el2 |= (guest_mdcr & NV_MDCR_GUEST_INCLUDE); 2103 /* 2104 * In yet another example where FEAT_NV2 is fscking broken, accesses 2105 * to MDSCR_EL1 are redirected to the VNCR despite having an effect 2106 * at EL2. Use a big hammer to apply sanity. 2107 * 2108 * Unless of course we have FEAT_FGT, in which case we can precisely 2109 * trap MDSCR_EL1. 2110 */ 2111 else if (!cpus_have_final_cap(ARM64_HAS_FGT)) 2112 vcpu->arch.mdcr_el2 |= MDCR_EL2_TDA; 2113 } 2114