1 // SPDX-License-Identifier: GPL-2.0-only
2 /*
3 * Copyright (C) 2007-2010 Advanced Micro Devices, Inc.
4 * Author: Joerg Roedel <jroedel@suse.de>
5 * Leo Duran <leo.duran@amd.com>
6 */
7
8 #define pr_fmt(fmt) "AMD-Vi: " fmt
9 #define dev_fmt(fmt) pr_fmt(fmt)
10
11 #include <linux/pci.h>
12 #include <linux/acpi.h>
13 #include <linux/list.h>
14 #include <linux/bitmap.h>
15 #include <linux/syscore_ops.h>
16 #include <linux/interrupt.h>
17 #include <linux/msi.h>
18 #include <linux/irq.h>
19 #include <linux/amd-iommu.h>
20 #include <linux/export.h>
21 #include <linux/kmemleak.h>
22 #include <linux/cc_platform.h>
23 #include <linux/iopoll.h>
24 #include <asm/pci-direct.h>
25 #include <asm/iommu.h>
26 #include <asm/apic.h>
27 #include <asm/gart.h>
28 #include <asm/x86_init.h>
29 #include <asm/io_apic.h>
30 #include <asm/irq_remapping.h>
31 #include <asm/set_memory.h>
32 #include <asm/sev.h>
33
34 #include <linux/crash_dump.h>
35
36 #include "amd_iommu.h"
37 #include "../irq_remapping.h"
38 #include "../iommu-pages.h"
39
40 /*
41 * definitions for the ACPI scanning code
42 */
43 #define IVRS_HEADER_LENGTH 48
44
45 #define ACPI_IVHD_TYPE_MAX_SUPPORTED 0x40
46 #define ACPI_IVMD_TYPE_ALL 0x20
47 #define ACPI_IVMD_TYPE 0x21
48 #define ACPI_IVMD_TYPE_RANGE 0x22
49
50 #define IVHD_DEV_ALL 0x01
51 #define IVHD_DEV_SELECT 0x02
52 #define IVHD_DEV_SELECT_RANGE_START 0x03
53 #define IVHD_DEV_RANGE_END 0x04
54 #define IVHD_DEV_ALIAS 0x42
55 #define IVHD_DEV_ALIAS_RANGE 0x43
56 #define IVHD_DEV_EXT_SELECT 0x46
57 #define IVHD_DEV_EXT_SELECT_RANGE 0x47
58 #define IVHD_DEV_SPECIAL 0x48
59 #define IVHD_DEV_ACPI_HID 0xf0
60
61 #define UID_NOT_PRESENT 0
62 #define UID_IS_INTEGER 1
63 #define UID_IS_CHARACTER 2
64
65 #define IVHD_SPECIAL_IOAPIC 1
66 #define IVHD_SPECIAL_HPET 2
67
68 #define IVHD_FLAG_HT_TUN_EN_MASK 0x01
69 #define IVHD_FLAG_PASSPW_EN_MASK 0x02
70 #define IVHD_FLAG_RESPASSPW_EN_MASK 0x04
71 #define IVHD_FLAG_ISOC_EN_MASK 0x08
72
73 #define IVMD_FLAG_EXCL_RANGE 0x08
74 #define IVMD_FLAG_IW 0x04
75 #define IVMD_FLAG_IR 0x02
76 #define IVMD_FLAG_UNITY_MAP 0x01
77
78 #define ACPI_DEVFLAG_INITPASS 0x01
79 #define ACPI_DEVFLAG_EXTINT 0x02
80 #define ACPI_DEVFLAG_NMI 0x04
81 #define ACPI_DEVFLAG_SYSMGT1 0x10
82 #define ACPI_DEVFLAG_SYSMGT2 0x20
83 #define ACPI_DEVFLAG_LINT0 0x40
84 #define ACPI_DEVFLAG_LINT1 0x80
85 #define ACPI_DEVFLAG_ATSDIS 0x10000000
86
87 #define IVRS_GET_SBDF_ID(seg, bus, dev, fn) (((seg & 0xffff) << 16) | ((bus & 0xff) << 8) \
88 | ((dev & 0x1f) << 3) | (fn & 0x7))
89
90 /*
91 * ACPI table definitions
92 *
93 * These data structures are laid over the table to parse the important values
94 * out of it.
95 */
96
97 /*
98 * structure describing one IOMMU in the ACPI table. Typically followed by one
99 * or more ivhd_entrys.
100 */
101 struct ivhd_header {
102 u8 type;
103 u8 flags;
104 u16 length;
105 u16 devid;
106 u16 cap_ptr;
107 u64 mmio_phys;
108 u16 pci_seg;
109 u16 info;
110 u32 efr_attr;
111
112 /* Following only valid on IVHD type 11h and 40h */
113 u64 efr_reg; /* Exact copy of MMIO_EXT_FEATURES */
114 u64 efr_reg2;
115 } __attribute__((packed));
116
117 /*
118 * A device entry describing which devices a specific IOMMU translates and
119 * which requestor ids they use.
120 */
121 struct ivhd_entry {
122 u8 type;
123 u16 devid;
124 u8 flags;
125 struct_group(ext_hid,
126 u32 ext;
127 u32 hidh;
128 );
129 u64 cid;
130 u8 uidf;
131 u8 uidl;
132 u8 uid;
133 } __attribute__((packed));
134
135 int amd_iommu_evtlog_size = EVTLOG_SIZE_DEF;
136 int amd_iommu_pprlog_size = PPRLOG_SIZE_DEF;
137
138 /*
139 * An AMD IOMMU memory definition structure. It defines things like exclusion
140 * ranges for devices and regions that should be unity mapped.
141 */
142 struct ivmd_header {
143 u8 type;
144 u8 flags;
145 u16 length;
146 u16 devid;
147 u16 aux;
148 u16 pci_seg;
149 u8 resv[6];
150 u64 range_start;
151 u64 range_length;
152 } __attribute__((packed));
153
154 bool amd_iommu_dump;
155 static bool amd_iommu_irq_remap __read_mostly;
156
157 enum protection_domain_mode amd_iommu_pgtable = PD_MODE_V1;
158 /* Virtual address size */
159 u8 amd_iommu_hpt_vasize;
160 /* Guest page table level */
161 int amd_iommu_gpt_level = PAGE_MODE_4_LEVEL;
162
163 int amd_iommu_guest_ir = AMD_IOMMU_GUEST_IR_VAPIC;
164 static int amd_iommu_xt_mode = IRQ_REMAP_XAPIC_MODE;
165
166 static bool amd_iommu_detected;
167 static bool amd_iommu_disabled __initdata;
168 static bool amd_iommu_force_enable __initdata;
169 static bool amd_iommu_irtcachedis;
170 static int amd_iommu_target_ivhd_type;
171
172 /* Global EFR and EFR2 registers */
173 u64 amd_iommu_efr;
174 u64 amd_iommu_efr2;
175
176 /* Host (v1) page table is not supported*/
177 bool amd_iommu_hatdis;
178
179 /* SNP is enabled on the system? */
180 bool amd_iommu_snp_en;
181 EXPORT_SYMBOL(amd_iommu_snp_en);
182
183 /* SNP page mode 0 support */
184 bool amd_iommu_snp_mode0_sup;
185
186 LIST_HEAD(amd_iommu_pci_seg_list); /* list of all PCI segments */
187 LIST_HEAD(amd_iommu_list); /* list of all AMD IOMMUs in the system */
188 LIST_HEAD(amd_ivhd_dev_flags_list); /* list of all IVHD device entry settings */
189
190 /* Number of IOMMUs present in the system */
191 static int amd_iommus_present;
192
193 /* IOMMUs have a non-present cache? */
194 bool amd_iommu_np_cache __read_mostly;
195 bool amd_iommu_iotlb_sup __read_mostly = true;
196
197 static bool amd_iommu_pc_present __read_mostly;
198 bool amdr_ivrs_remap_support __read_mostly;
199
200 bool amd_iommu_force_isolation __read_mostly;
201
202 unsigned long amd_iommu_pgsize_bitmap __ro_after_init = AMD_IOMMU_PGSIZES;
203
204 enum iommu_init_state {
205 IOMMU_START_STATE,
206 IOMMU_IVRS_DETECTED,
207 IOMMU_ACPI_FINISHED,
208 IOMMU_ENABLED,
209 IOMMU_PCI_INIT,
210 IOMMU_INTERRUPTS_EN,
211 IOMMU_INITIALIZED,
212 IOMMU_NOT_FOUND,
213 IOMMU_INIT_ERROR,
214 IOMMU_CMDLINE_DISABLED,
215 };
216
217 /* Early ioapic and hpet maps from kernel command line */
218 #define EARLY_MAP_SIZE 4
219 static struct devid_map __initdata early_ioapic_map[EARLY_MAP_SIZE];
220 static struct devid_map __initdata early_hpet_map[EARLY_MAP_SIZE];
221 static struct acpihid_map_entry __initdata early_acpihid_map[EARLY_MAP_SIZE];
222
223 static int __initdata early_ioapic_map_size;
224 static int __initdata early_hpet_map_size;
225 static int __initdata early_acpihid_map_size;
226
227 static bool __initdata cmdline_maps;
228
229 static enum iommu_init_state init_state = IOMMU_START_STATE;
230
231 static int amd_iommu_enable_interrupts(void);
232 static void init_device_table_dma(struct amd_iommu_pci_seg *pci_seg);
233
234 static bool amd_iommu_pre_enabled = true;
235
236 static u32 amd_iommu_ivinfo __initdata;
237
translation_pre_enabled(struct amd_iommu * iommu)238 bool translation_pre_enabled(struct amd_iommu *iommu)
239 {
240 return (iommu->flags & AMD_IOMMU_FLAG_TRANS_PRE_ENABLED);
241 }
242
clear_translation_pre_enabled(struct amd_iommu * iommu)243 static void clear_translation_pre_enabled(struct amd_iommu *iommu)
244 {
245 iommu->flags &= ~AMD_IOMMU_FLAG_TRANS_PRE_ENABLED;
246 }
247
init_translation_status(struct amd_iommu * iommu)248 static void init_translation_status(struct amd_iommu *iommu)
249 {
250 u64 ctrl;
251
252 ctrl = readq(iommu->mmio_base + MMIO_CONTROL_OFFSET);
253 if (ctrl & (1<<CONTROL_IOMMU_EN))
254 iommu->flags |= AMD_IOMMU_FLAG_TRANS_PRE_ENABLED;
255 }
256
amd_iommu_get_num_iommus(void)257 int amd_iommu_get_num_iommus(void)
258 {
259 return amd_iommus_present;
260 }
261
amd_iommu_ht_range_ignore(void)262 bool amd_iommu_ht_range_ignore(void)
263 {
264 return check_feature2(FEATURE_HT_RANGE_IGNORE);
265 }
266
267 /*
268 * Iterate through all the IOMMUs to get common EFR
269 * masks among all IOMMUs and warn if found inconsistency.
270 */
get_global_efr(void)271 static __init void get_global_efr(void)
272 {
273 struct amd_iommu *iommu;
274
275 for_each_iommu(iommu) {
276 u64 tmp = iommu->features;
277 u64 tmp2 = iommu->features2;
278
279 if (list_is_first(&iommu->list, &amd_iommu_list)) {
280 amd_iommu_efr = tmp;
281 amd_iommu_efr2 = tmp2;
282 continue;
283 }
284
285 if (amd_iommu_efr == tmp &&
286 amd_iommu_efr2 == tmp2)
287 continue;
288
289 pr_err(FW_BUG
290 "Found inconsistent EFR/EFR2 %#llx,%#llx (global %#llx,%#llx) on iommu%d (%04x:%02x:%02x.%01x).\n",
291 tmp, tmp2, amd_iommu_efr, amd_iommu_efr2,
292 iommu->index, iommu->pci_seg->id,
293 PCI_BUS_NUM(iommu->devid), PCI_SLOT(iommu->devid),
294 PCI_FUNC(iommu->devid));
295
296 amd_iommu_efr &= tmp;
297 amd_iommu_efr2 &= tmp2;
298 }
299
300 pr_info("Using global IVHD EFR:%#llx, EFR2:%#llx\n", amd_iommu_efr, amd_iommu_efr2);
301 }
302
303 /*
304 * For IVHD type 0x11/0x40, EFR is also available via IVHD.
305 * Default to IVHD EFR since it is available sooner
306 * (i.e. before PCI init).
307 */
early_iommu_features_init(struct amd_iommu * iommu,struct ivhd_header * h)308 static void __init early_iommu_features_init(struct amd_iommu *iommu,
309 struct ivhd_header *h)
310 {
311 if (amd_iommu_ivinfo & IOMMU_IVINFO_EFRSUP) {
312 iommu->features = h->efr_reg;
313 iommu->features2 = h->efr_reg2;
314 }
315 if (amd_iommu_ivinfo & IOMMU_IVINFO_DMA_REMAP)
316 amdr_ivrs_remap_support = true;
317 }
318
319 /* Access to l1 and l2 indexed register spaces */
320
iommu_read_l1(struct amd_iommu * iommu,u16 l1,u8 address)321 static u32 iommu_read_l1(struct amd_iommu *iommu, u16 l1, u8 address)
322 {
323 u32 val;
324
325 pci_write_config_dword(iommu->dev, 0xf8, (address | l1 << 16));
326 pci_read_config_dword(iommu->dev, 0xfc, &val);
327 return val;
328 }
329
iommu_write_l1(struct amd_iommu * iommu,u16 l1,u8 address,u32 val)330 static void iommu_write_l1(struct amd_iommu *iommu, u16 l1, u8 address, u32 val)
331 {
332 pci_write_config_dword(iommu->dev, 0xf8, (address | l1 << 16 | 1 << 31));
333 pci_write_config_dword(iommu->dev, 0xfc, val);
334 pci_write_config_dword(iommu->dev, 0xf8, (address | l1 << 16));
335 }
336
iommu_read_l2(struct amd_iommu * iommu,u8 address)337 static u32 iommu_read_l2(struct amd_iommu *iommu, u8 address)
338 {
339 u32 val;
340
341 pci_write_config_dword(iommu->dev, 0xf0, address);
342 pci_read_config_dword(iommu->dev, 0xf4, &val);
343 return val;
344 }
345
iommu_write_l2(struct amd_iommu * iommu,u8 address,u32 val)346 static void iommu_write_l2(struct amd_iommu *iommu, u8 address, u32 val)
347 {
348 pci_write_config_dword(iommu->dev, 0xf0, (address | 1 << 8));
349 pci_write_config_dword(iommu->dev, 0xf4, val);
350 }
351
352 /****************************************************************************
353 *
354 * AMD IOMMU MMIO register space handling functions
355 *
356 * These functions are used to program the IOMMU device registers in
357 * MMIO space required for that driver.
358 *
359 ****************************************************************************/
360
iommu_set_cwwb_range(struct amd_iommu * iommu)361 static void iommu_set_cwwb_range(struct amd_iommu *iommu)
362 {
363 u64 start = iommu_virt_to_phys((void *)iommu->cmd_sem);
364 u64 entry = start & PM_ADDR_MASK;
365
366 if (!check_feature(FEATURE_SNP))
367 return;
368
369 /* Note:
370 * Re-purpose Exclusion base/limit registers for Completion wait
371 * write-back base/limit.
372 */
373 memcpy_toio(iommu->mmio_base + MMIO_EXCL_BASE_OFFSET,
374 &entry, sizeof(entry));
375
376 /* Note:
377 * Default to 4 Kbytes, which can be specified by setting base
378 * address equal to the limit address.
379 */
380 memcpy_toio(iommu->mmio_base + MMIO_EXCL_LIMIT_OFFSET,
381 &entry, sizeof(entry));
382 }
383
384 /* Programs the physical address of the device table into the IOMMU hardware */
iommu_set_device_table(struct amd_iommu * iommu)385 static void iommu_set_device_table(struct amd_iommu *iommu)
386 {
387 u64 entry;
388 u32 dev_table_size = iommu->pci_seg->dev_table_size;
389 void *dev_table = (void *)get_dev_table(iommu);
390
391 BUG_ON(iommu->mmio_base == NULL);
392
393 if (is_kdump_kernel())
394 return;
395
396 entry = iommu_virt_to_phys(dev_table);
397 entry |= (dev_table_size >> 12) - 1;
398 memcpy_toio(iommu->mmio_base + MMIO_DEV_TABLE_OFFSET,
399 &entry, sizeof(entry));
400 }
401
iommu_feature_set(struct amd_iommu * iommu,u64 val,u64 mask,u8 shift)402 static void iommu_feature_set(struct amd_iommu *iommu, u64 val, u64 mask, u8 shift)
403 {
404 u64 ctrl;
405
406 ctrl = readq(iommu->mmio_base + MMIO_CONTROL_OFFSET);
407 mask <<= shift;
408 ctrl &= ~mask;
409 ctrl |= (val << shift) & mask;
410 writeq(ctrl, iommu->mmio_base + MMIO_CONTROL_OFFSET);
411 }
412
413 /* Generic functions to enable/disable certain features of the IOMMU. */
iommu_feature_enable(struct amd_iommu * iommu,u8 bit)414 void iommu_feature_enable(struct amd_iommu *iommu, u8 bit)
415 {
416 iommu_feature_set(iommu, 1ULL, 1ULL, bit);
417 }
418
iommu_feature_disable(struct amd_iommu * iommu,u8 bit)419 static void iommu_feature_disable(struct amd_iommu *iommu, u8 bit)
420 {
421 iommu_feature_set(iommu, 0ULL, 1ULL, bit);
422 }
423
424 /* Function to enable the hardware */
iommu_enable(struct amd_iommu * iommu)425 static void iommu_enable(struct amd_iommu *iommu)
426 {
427 iommu_feature_enable(iommu, CONTROL_IOMMU_EN);
428 }
429
iommu_disable(struct amd_iommu * iommu)430 static void iommu_disable(struct amd_iommu *iommu)
431 {
432 if (!iommu->mmio_base)
433 return;
434
435 /* Disable command buffer */
436 iommu_feature_disable(iommu, CONTROL_CMDBUF_EN);
437
438 /* Disable event logging and event interrupts */
439 iommu_feature_disable(iommu, CONTROL_EVT_INT_EN);
440 iommu_feature_disable(iommu, CONTROL_EVT_LOG_EN);
441
442 /* Disable IOMMU GA_LOG */
443 iommu_feature_disable(iommu, CONTROL_GALOG_EN);
444 iommu_feature_disable(iommu, CONTROL_GAINT_EN);
445
446 /* Disable IOMMU PPR logging */
447 iommu_feature_disable(iommu, CONTROL_PPRLOG_EN);
448 iommu_feature_disable(iommu, CONTROL_PPRINT_EN);
449
450 /* Disable IOMMU hardware itself */
451 iommu_feature_disable(iommu, CONTROL_IOMMU_EN);
452
453 /* Clear IRTE cache disabling bit */
454 iommu_feature_disable(iommu, CONTROL_IRTCACHEDIS);
455 }
456
457 /*
458 * mapping and unmapping functions for the IOMMU MMIO space. Each AMD IOMMU in
459 * the system has one.
460 */
iommu_map_mmio_space(u64 address,u64 end)461 static u8 __iomem * __init iommu_map_mmio_space(u64 address, u64 end)
462 {
463 if (!request_mem_region(address, end, "amd_iommu")) {
464 pr_err("Can not reserve memory region %llx-%llx for mmio\n",
465 address, end);
466 pr_err("This is a BIOS bug. Please contact your hardware vendor\n");
467 return NULL;
468 }
469
470 return (u8 __iomem *)ioremap(address, end);
471 }
472
iommu_unmap_mmio_space(struct amd_iommu * iommu)473 static void __init iommu_unmap_mmio_space(struct amd_iommu *iommu)
474 {
475 if (iommu->mmio_base)
476 iounmap(iommu->mmio_base);
477 release_mem_region(iommu->mmio_phys, iommu->mmio_phys_end);
478 }
479
get_ivhd_header_size(struct ivhd_header * h)480 static inline u32 get_ivhd_header_size(struct ivhd_header *h)
481 {
482 u32 size = 0;
483
484 switch (h->type) {
485 case 0x10:
486 size = 24;
487 break;
488 case 0x11:
489 case 0x40:
490 size = 40;
491 break;
492 }
493 return size;
494 }
495
496 /****************************************************************************
497 *
498 * The functions below belong to the first pass of AMD IOMMU ACPI table
499 * parsing. In this pass we try to find out the highest device id this
500 * code has to handle. Upon this information the size of the shared data
501 * structures is determined later.
502 *
503 ****************************************************************************/
504
505 /*
506 * This function calculates the length of a given IVHD entry
507 */
ivhd_entry_length(u8 * ivhd)508 static inline int ivhd_entry_length(u8 *ivhd)
509 {
510 u32 type = ((struct ivhd_entry *)ivhd)->type;
511
512 if (type < 0x80) {
513 return 0x04 << (*ivhd >> 6);
514 } else if (type == IVHD_DEV_ACPI_HID) {
515 /* For ACPI_HID, offset 21 is uid len */
516 return *((u8 *)ivhd + 21) + 22;
517 }
518 return 0;
519 }
520
521 /*
522 * After reading the highest device id from the IOMMU PCI capability header
523 * this function looks if there is a higher device id defined in the ACPI table
524 */
find_last_devid_from_ivhd(struct ivhd_header * h)525 static int __init find_last_devid_from_ivhd(struct ivhd_header *h)
526 {
527 u8 *p = (void *)h, *end = (void *)h;
528 struct ivhd_entry *dev;
529 int last_devid = -EINVAL;
530
531 u32 ivhd_size = get_ivhd_header_size(h);
532
533 if (!ivhd_size) {
534 pr_err("Unsupported IVHD type %#x\n", h->type);
535 return -EINVAL;
536 }
537
538 p += ivhd_size;
539 end += h->length;
540
541 while (p < end) {
542 dev = (struct ivhd_entry *)p;
543 switch (dev->type) {
544 case IVHD_DEV_ALL:
545 /* Use maximum BDF value for DEV_ALL */
546 return 0xffff;
547 case IVHD_DEV_SELECT:
548 case IVHD_DEV_RANGE_END:
549 case IVHD_DEV_ALIAS:
550 case IVHD_DEV_EXT_SELECT:
551 /* all the above subfield types refer to device ids */
552 if (dev->devid > last_devid)
553 last_devid = dev->devid;
554 break;
555 default:
556 break;
557 }
558 p += ivhd_entry_length(p);
559 }
560
561 WARN_ON(p != end);
562
563 return last_devid;
564 }
565
check_ivrs_checksum(struct acpi_table_header * table)566 static int __init check_ivrs_checksum(struct acpi_table_header *table)
567 {
568 int i;
569 u8 checksum = 0, *p = (u8 *)table;
570
571 for (i = 0; i < table->length; ++i)
572 checksum += p[i];
573 if (checksum != 0) {
574 /* ACPI table corrupt */
575 pr_err(FW_BUG "IVRS invalid checksum\n");
576 return -ENODEV;
577 }
578
579 return 0;
580 }
581
582 /*
583 * Iterate over all IVHD entries in the ACPI table and find the highest device
584 * id which we need to handle. This is the first of three functions which parse
585 * the ACPI table. So we check the checksum here.
586 */
find_last_devid_acpi(struct acpi_table_header * table,u16 pci_seg)587 static int __init find_last_devid_acpi(struct acpi_table_header *table, u16 pci_seg)
588 {
589 u8 *p = (u8 *)table, *end = (u8 *)table;
590 struct ivhd_header *h;
591 int last_devid, last_bdf = 0;
592
593 p += IVRS_HEADER_LENGTH;
594
595 end += table->length;
596 while (p < end) {
597 h = (struct ivhd_header *)p;
598 if (h->pci_seg == pci_seg &&
599 h->type == amd_iommu_target_ivhd_type) {
600 last_devid = find_last_devid_from_ivhd(h);
601
602 if (last_devid < 0)
603 return -EINVAL;
604 if (last_devid > last_bdf)
605 last_bdf = last_devid;
606 }
607 p += h->length;
608 }
609 WARN_ON(p != end);
610
611 return last_bdf;
612 }
613
614 /****************************************************************************
615 *
616 * The following functions belong to the code path which parses the ACPI table
617 * the second time. In this ACPI parsing iteration we allocate IOMMU specific
618 * data structures, initialize the per PCI segment device/alias/rlookup table
619 * and also basically initialize the hardware.
620 *
621 ****************************************************************************/
622
623 /* Allocate per PCI segment device table */
alloc_dev_table(struct amd_iommu_pci_seg * pci_seg)624 static inline int __init alloc_dev_table(struct amd_iommu_pci_seg *pci_seg)
625 {
626 pci_seg->dev_table = iommu_alloc_pages_sz(GFP_KERNEL | GFP_DMA32,
627 pci_seg->dev_table_size);
628 if (!pci_seg->dev_table)
629 return -ENOMEM;
630
631 return 0;
632 }
633
free_dev_table(struct amd_iommu_pci_seg * pci_seg)634 static inline void free_dev_table(struct amd_iommu_pci_seg *pci_seg)
635 {
636 if (is_kdump_kernel())
637 memunmap((void *)pci_seg->dev_table);
638 else
639 iommu_free_pages(pci_seg->dev_table);
640 pci_seg->dev_table = NULL;
641 }
642
643 /* Allocate per PCI segment IOMMU rlookup table. */
alloc_rlookup_table(struct amd_iommu_pci_seg * pci_seg)644 static inline int __init alloc_rlookup_table(struct amd_iommu_pci_seg *pci_seg)
645 {
646 pci_seg->rlookup_table = kvzalloc_objs(*pci_seg->rlookup_table,
647 pci_seg->last_bdf + 1);
648 if (pci_seg->rlookup_table == NULL)
649 return -ENOMEM;
650
651 return 0;
652 }
653
free_rlookup_table(struct amd_iommu_pci_seg * pci_seg)654 static inline void free_rlookup_table(struct amd_iommu_pci_seg *pci_seg)
655 {
656 kvfree(pci_seg->rlookup_table);
657 pci_seg->rlookup_table = NULL;
658 }
659
alloc_irq_lookup_table(struct amd_iommu_pci_seg * pci_seg)660 static inline int __init alloc_irq_lookup_table(struct amd_iommu_pci_seg *pci_seg)
661 {
662 pci_seg->irq_lookup_table = kvzalloc_objs(*pci_seg->irq_lookup_table,
663 pci_seg->last_bdf + 1);
664 if (pci_seg->irq_lookup_table == NULL)
665 return -ENOMEM;
666
667 return 0;
668 }
669
free_irq_lookup_table(struct amd_iommu_pci_seg * pci_seg)670 static inline void free_irq_lookup_table(struct amd_iommu_pci_seg *pci_seg)
671 {
672 kvfree(pci_seg->irq_lookup_table);
673 pci_seg->irq_lookup_table = NULL;
674 }
675
alloc_alias_table(struct amd_iommu_pci_seg * pci_seg)676 static int __init alloc_alias_table(struct amd_iommu_pci_seg *pci_seg)
677 {
678 int i;
679
680 pci_seg->alias_table = kvmalloc_objs(*pci_seg->alias_table,
681 pci_seg->last_bdf + 1);
682 if (!pci_seg->alias_table)
683 return -ENOMEM;
684
685 /*
686 * let all alias entries point to itself
687 */
688 for (i = 0; i <= pci_seg->last_bdf; ++i)
689 pci_seg->alias_table[i] = i;
690
691 return 0;
692 }
693
free_alias_table(struct amd_iommu_pci_seg * pci_seg)694 static void __init free_alias_table(struct amd_iommu_pci_seg *pci_seg)
695 {
696 kvfree(pci_seg->alias_table);
697 pci_seg->alias_table = NULL;
698 }
699
iommu_memremap(unsigned long paddr,size_t size)700 static inline void *iommu_memremap(unsigned long paddr, size_t size)
701 {
702 phys_addr_t phys;
703
704 if (!paddr)
705 return NULL;
706
707 /*
708 * Obtain true physical address in kdump kernel when SME is enabled.
709 * Currently, previous kernel with SME enabled and kdump kernel
710 * with SME support disabled is not supported.
711 */
712 phys = __sme_clr(paddr);
713
714 if (cc_platform_has(CC_ATTR_HOST_MEM_ENCRYPT))
715 return (__force void *)ioremap_encrypted(phys, size);
716 else
717 return memremap(phys, size, MEMREMAP_WB);
718 }
719
720 /*
721 * Allocates the command buffer. This buffer is per AMD IOMMU. We can
722 * write commands to that buffer later and the IOMMU will execute them
723 * asynchronously
724 */
alloc_command_buffer(struct amd_iommu * iommu)725 static int __init alloc_command_buffer(struct amd_iommu *iommu)
726 {
727 iommu->cmd_buf = iommu_alloc_pages_sz(GFP_KERNEL, CMD_BUFFER_SIZE);
728
729 return iommu->cmd_buf ? 0 : -ENOMEM;
730 }
731
732 /*
733 * Interrupt handler has processed all pending events and adjusted head
734 * and tail pointer. Reset overflow mask and restart logging again.
735 */
amd_iommu_restart_log(struct amd_iommu * iommu,const char * evt_type,u8 cntrl_intr,u8 cntrl_log,u32 status_run_mask,u32 status_overflow_mask)736 void amd_iommu_restart_log(struct amd_iommu *iommu, const char *evt_type,
737 u8 cntrl_intr, u8 cntrl_log,
738 u32 status_run_mask, u32 status_overflow_mask)
739 {
740 u32 status;
741
742 status = readl(iommu->mmio_base + MMIO_STATUS_OFFSET);
743 if (status & status_run_mask)
744 return;
745
746 pr_info_ratelimited("IOMMU %s log restarting\n", evt_type);
747
748 iommu_feature_disable(iommu, cntrl_log);
749 iommu_feature_disable(iommu, cntrl_intr);
750
751 writel(status_overflow_mask, iommu->mmio_base + MMIO_STATUS_OFFSET);
752
753 iommu_feature_enable(iommu, cntrl_intr);
754 iommu_feature_enable(iommu, cntrl_log);
755 }
756
757 /*
758 * This function restarts event logging in case the IOMMU experienced
759 * an event log buffer overflow.
760 */
amd_iommu_restart_event_logging(struct amd_iommu * iommu)761 void amd_iommu_restart_event_logging(struct amd_iommu *iommu)
762 {
763 amd_iommu_restart_log(iommu, "Event", CONTROL_EVT_INT_EN,
764 CONTROL_EVT_LOG_EN, MMIO_STATUS_EVT_RUN_MASK,
765 MMIO_STATUS_EVT_OVERFLOW_MASK);
766 }
767
768 /*
769 * This function restarts event logging in case the IOMMU experienced
770 * GA log overflow.
771 */
amd_iommu_restart_ga_log(struct amd_iommu * iommu)772 void amd_iommu_restart_ga_log(struct amd_iommu *iommu)
773 {
774 amd_iommu_restart_log(iommu, "GA", CONTROL_GAINT_EN,
775 CONTROL_GALOG_EN, MMIO_STATUS_GALOG_RUN_MASK,
776 MMIO_STATUS_GALOG_OVERFLOW_MASK);
777 }
778
779 /*
780 * This function resets the command buffer if the IOMMU stopped fetching
781 * commands from it.
782 */
amd_iommu_reset_cmd_buffer(struct amd_iommu * iommu)783 static void amd_iommu_reset_cmd_buffer(struct amd_iommu *iommu)
784 {
785 iommu_feature_disable(iommu, CONTROL_CMDBUF_EN);
786
787 writel(0x00, iommu->mmio_base + MMIO_CMD_HEAD_OFFSET);
788 writel(0x00, iommu->mmio_base + MMIO_CMD_TAIL_OFFSET);
789 iommu->cmd_buf_head = 0;
790 iommu->cmd_buf_tail = 0;
791
792 iommu_feature_enable(iommu, CONTROL_CMDBUF_EN);
793 }
794
795 /*
796 * This function writes the command buffer address to the hardware and
797 * enables it.
798 */
iommu_enable_command_buffer(struct amd_iommu * iommu)799 static void iommu_enable_command_buffer(struct amd_iommu *iommu)
800 {
801 u64 entry;
802
803 BUG_ON(iommu->cmd_buf == NULL);
804
805 if (!is_kdump_kernel()) {
806 /*
807 * Command buffer is re-used for kdump kernel and setting
808 * of MMIO register is not required.
809 */
810 entry = iommu_virt_to_phys(iommu->cmd_buf);
811 entry |= MMIO_CMD_SIZE_512;
812 memcpy_toio(iommu->mmio_base + MMIO_CMD_BUF_OFFSET,
813 &entry, sizeof(entry));
814 }
815
816 amd_iommu_reset_cmd_buffer(iommu);
817 }
818
819 /*
820 * This function disables the command buffer
821 */
iommu_disable_command_buffer(struct amd_iommu * iommu)822 static void iommu_disable_command_buffer(struct amd_iommu *iommu)
823 {
824 iommu_feature_disable(iommu, CONTROL_CMDBUF_EN);
825 }
826
free_command_buffer(struct amd_iommu * iommu)827 static void __init free_command_buffer(struct amd_iommu *iommu)
828 {
829 iommu_free_pages(iommu->cmd_buf);
830 }
831
iommu_alloc_4k_pages(struct amd_iommu * iommu,gfp_t gfp,size_t size)832 void *__init iommu_alloc_4k_pages(struct amd_iommu *iommu, gfp_t gfp,
833 size_t size)
834 {
835 int nid = iommu->dev ? dev_to_node(&iommu->dev->dev) : NUMA_NO_NODE;
836 void *buf;
837
838 size = PAGE_ALIGN(size);
839 buf = iommu_alloc_pages_node_sz(nid, gfp, size);
840 if (!buf)
841 return NULL;
842 if (check_feature(FEATURE_SNP) &&
843 set_memory_4k((unsigned long)buf, size / PAGE_SIZE)) {
844 iommu_free_pages(buf);
845 return NULL;
846 }
847
848 return buf;
849 }
850
851 /* allocates the memory where the IOMMU will log its events to */
alloc_event_buffer(void)852 static int __init alloc_event_buffer(void)
853 {
854 struct amd_iommu *iommu;
855
856 for_each_iommu(iommu) {
857 iommu->evt_buf = iommu_alloc_4k_pages(iommu, GFP_KERNEL,
858 amd_iommu_evtlog_size);
859 if (!iommu->evt_buf)
860 return -ENOMEM;
861 }
862
863 return 0;
864 }
865
iommu_enable_event_buffer(void)866 static void iommu_enable_event_buffer(void)
867 {
868 struct amd_iommu *iommu;
869 u64 entry;
870
871 for_each_iommu(iommu) {
872 BUG_ON(iommu->evt_buf == NULL);
873
874 if (!is_kdump_kernel()) {
875 /*
876 * Event buffer is re-used for kdump kernel and setting
877 * of MMIO register is not required.
878 */
879 entry = iommu_virt_to_phys(iommu->evt_buf);
880 entry |= (amd_iommu_evtlog_size == EVTLOG_SIZE_DEF) ?
881 EVTLOG_LEN_MASK_DEF : EVTLOG_LEN_MASK_MAX;
882
883 memcpy_toio(iommu->mmio_base + MMIO_EVT_BUF_OFFSET,
884 &entry, sizeof(entry));
885 }
886
887 /* set head and tail to zero manually */
888 writel(0x00, iommu->mmio_base + MMIO_EVT_HEAD_OFFSET);
889 writel(0x00, iommu->mmio_base + MMIO_EVT_TAIL_OFFSET);
890
891 iommu_feature_enable(iommu, CONTROL_EVT_LOG_EN);
892 }
893 }
894
895 /*
896 * This function disables the event log buffer
897 */
iommu_disable_event_buffer(struct amd_iommu * iommu)898 static void iommu_disable_event_buffer(struct amd_iommu *iommu)
899 {
900 iommu_feature_disable(iommu, CONTROL_EVT_LOG_EN);
901 }
902
free_event_buffer(struct amd_iommu * iommu)903 static void __init free_event_buffer(struct amd_iommu *iommu)
904 {
905 iommu_free_pages(iommu->evt_buf);
906 }
907
free_ga_log(struct amd_iommu * iommu)908 static void free_ga_log(struct amd_iommu *iommu)
909 {
910 #ifdef CONFIG_IRQ_REMAP
911 iommu_free_pages(iommu->ga_log);
912 iommu->ga_log = NULL;
913 iommu_free_pages(iommu->ga_log_tail);
914 iommu->ga_log_tail = NULL;
915 #endif
916 }
917
918 #ifdef CONFIG_IRQ_REMAP
iommu_ga_log_enable(struct amd_iommu * iommu)919 static int iommu_ga_log_enable(struct amd_iommu *iommu)
920 {
921 u32 status, i;
922 u64 entry;
923
924 if (!iommu->ga_log)
925 return -EINVAL;
926
927 entry = iommu_virt_to_phys(iommu->ga_log) | GA_LOG_SIZE_512;
928 memcpy_toio(iommu->mmio_base + MMIO_GA_LOG_BASE_OFFSET,
929 &entry, sizeof(entry));
930 entry = (iommu_virt_to_phys(iommu->ga_log_tail) &
931 (BIT_ULL(52)-1)) & ~7ULL;
932 memcpy_toio(iommu->mmio_base + MMIO_GA_LOG_TAIL_OFFSET,
933 &entry, sizeof(entry));
934 writel(0x00, iommu->mmio_base + MMIO_GA_HEAD_OFFSET);
935 writel(0x00, iommu->mmio_base + MMIO_GA_TAIL_OFFSET);
936
937
938 iommu_feature_enable(iommu, CONTROL_GAINT_EN);
939 iommu_feature_enable(iommu, CONTROL_GALOG_EN);
940
941 for (i = 0; i < MMIO_STATUS_TIMEOUT; ++i) {
942 status = readl(iommu->mmio_base + MMIO_STATUS_OFFSET);
943 if (status & (MMIO_STATUS_GALOG_RUN_MASK))
944 break;
945 udelay(10);
946 }
947
948 if (WARN_ON(i >= MMIO_STATUS_TIMEOUT))
949 return -EINVAL;
950
951 return 0;
952 }
953
iommu_init_ga_log(struct amd_iommu * iommu)954 static int iommu_init_ga_log(struct amd_iommu *iommu)
955 {
956 int nid = iommu->dev ? dev_to_node(&iommu->dev->dev) : NUMA_NO_NODE;
957
958 if (WARN_ON_ONCE(!AMD_IOMMU_GUEST_IR_VAPIC(amd_iommu_guest_ir)))
959 return -EINVAL;
960
961 if (iommu->ga_log && iommu->ga_log_tail)
962 return 0;
963
964 iommu->ga_log = iommu_alloc_pages_node_sz(nid, GFP_KERNEL, GA_LOG_SIZE);
965 if (!iommu->ga_log)
966 goto err_out;
967
968 iommu->ga_log_tail = iommu_alloc_pages_node_sz(nid, GFP_KERNEL, 8);
969 if (!iommu->ga_log_tail)
970 goto err_out;
971
972 return 0;
973 err_out:
974 free_ga_log(iommu);
975 return -EINVAL;
976 }
977 #endif /* CONFIG_IRQ_REMAP */
978
alloc_cwwb_sem(struct amd_iommu * iommu)979 static int __init alloc_cwwb_sem(struct amd_iommu *iommu)
980 {
981 iommu->cmd_sem = iommu_alloc_4k_pages(iommu, GFP_KERNEL, 1);
982 if (!iommu->cmd_sem)
983 return -ENOMEM;
984 iommu->cmd_sem_paddr = iommu_virt_to_phys((void *)iommu->cmd_sem);
985 return 0;
986 }
987
remap_event_buffer(void)988 static int __init remap_event_buffer(void)
989 {
990 struct amd_iommu *iommu;
991 u64 paddr;
992
993 pr_info_once("Re-using event buffer from the previous kernel\n");
994 for_each_iommu(iommu) {
995 paddr = readq(iommu->mmio_base + MMIO_EVT_BUF_OFFSET) & PM_ADDR_MASK;
996 iommu->evt_buf = iommu_memremap(paddr, amd_iommu_evtlog_size);
997 if (!iommu->evt_buf)
998 return -ENOMEM;
999 }
1000
1001 return 0;
1002 }
1003
remap_command_buffer(struct amd_iommu * iommu)1004 static int __init remap_command_buffer(struct amd_iommu *iommu)
1005 {
1006 u64 paddr;
1007
1008 pr_info_once("Re-using command buffer from the previous kernel\n");
1009 paddr = readq(iommu->mmio_base + MMIO_CMD_BUF_OFFSET) & PM_ADDR_MASK;
1010 iommu->cmd_buf = iommu_memremap(paddr, CMD_BUFFER_SIZE);
1011
1012 return iommu->cmd_buf ? 0 : -ENOMEM;
1013 }
1014
remap_or_alloc_cwwb_sem(struct amd_iommu * iommu)1015 static int __init remap_or_alloc_cwwb_sem(struct amd_iommu *iommu)
1016 {
1017 u64 paddr;
1018
1019 if (check_feature(FEATURE_SNP)) {
1020 /*
1021 * When SNP is enabled, the exclusion base register is used for the
1022 * completion wait buffer (CWB) address. Read and re-use it.
1023 */
1024 pr_info_once("Re-using CWB buffers from the previous kernel\n");
1025 paddr = readq(iommu->mmio_base + MMIO_EXCL_BASE_OFFSET) & PM_ADDR_MASK;
1026 iommu->cmd_sem = iommu_memremap(paddr, PAGE_SIZE);
1027 if (!iommu->cmd_sem)
1028 return -ENOMEM;
1029 iommu->cmd_sem_paddr = paddr;
1030 } else {
1031 return alloc_cwwb_sem(iommu);
1032 }
1033
1034 return 0;
1035 }
1036
alloc_iommu_buffers(struct amd_iommu * iommu)1037 static int __init alloc_iommu_buffers(struct amd_iommu *iommu)
1038 {
1039 int ret;
1040
1041 /*
1042 * Reuse/Remap the previous kernel's allocated completion wait
1043 * command and event buffers for kdump boot.
1044 */
1045 if (is_kdump_kernel()) {
1046 ret = remap_or_alloc_cwwb_sem(iommu);
1047 if (ret)
1048 return ret;
1049
1050 ret = remap_command_buffer(iommu);
1051 if (ret)
1052 return ret;
1053 } else {
1054 ret = alloc_cwwb_sem(iommu);
1055 if (ret)
1056 return ret;
1057
1058 ret = alloc_command_buffer(iommu);
1059 if (ret)
1060 return ret;
1061 }
1062
1063 return 0;
1064 }
1065
free_cwwb_sem(struct amd_iommu * iommu)1066 static void __init free_cwwb_sem(struct amd_iommu *iommu)
1067 {
1068 if (iommu->cmd_sem)
1069 iommu_free_pages((void *)iommu->cmd_sem);
1070 }
unmap_cwwb_sem(struct amd_iommu * iommu)1071 static void __init unmap_cwwb_sem(struct amd_iommu *iommu)
1072 {
1073 if (iommu->cmd_sem) {
1074 if (check_feature(FEATURE_SNP))
1075 memunmap((void *)iommu->cmd_sem);
1076 else
1077 iommu_free_pages((void *)iommu->cmd_sem);
1078 }
1079 }
1080
unmap_command_buffer(struct amd_iommu * iommu)1081 static void __init unmap_command_buffer(struct amd_iommu *iommu)
1082 {
1083 memunmap((void *)iommu->cmd_buf);
1084 }
1085
unmap_event_buffer(struct amd_iommu * iommu)1086 static void __init unmap_event_buffer(struct amd_iommu *iommu)
1087 {
1088 memunmap(iommu->evt_buf);
1089 }
1090
free_iommu_buffers(struct amd_iommu * iommu)1091 static void __init free_iommu_buffers(struct amd_iommu *iommu)
1092 {
1093 if (is_kdump_kernel()) {
1094 unmap_cwwb_sem(iommu);
1095 unmap_command_buffer(iommu);
1096 unmap_event_buffer(iommu);
1097 } else {
1098 free_cwwb_sem(iommu);
1099 free_command_buffer(iommu);
1100 free_event_buffer(iommu);
1101 }
1102 }
1103
iommu_enable_xt(struct amd_iommu * iommu)1104 static void iommu_enable_xt(struct amd_iommu *iommu)
1105 {
1106 #ifdef CONFIG_IRQ_REMAP
1107 /*
1108 * XT mode (32-bit APIC destination ID) requires
1109 * GA mode (128-bit IRTE support) as a prerequisite.
1110 */
1111 if (AMD_IOMMU_GUEST_IR_GA(amd_iommu_guest_ir) &&
1112 amd_iommu_xt_mode == IRQ_REMAP_X2APIC_MODE)
1113 iommu_feature_enable(iommu, CONTROL_XT_EN);
1114 #endif /* CONFIG_IRQ_REMAP */
1115 }
1116
iommu_enable_gt(struct amd_iommu * iommu)1117 static void iommu_enable_gt(struct amd_iommu *iommu)
1118 {
1119 if (!check_feature(FEATURE_GT))
1120 return;
1121
1122 iommu_feature_enable(iommu, CONTROL_GT_EN);
1123
1124 /*
1125 * This feature needs to be enabled prior to a call
1126 * to iommu_snp_enable(). Since this function is called
1127 * in early_enable_iommu(), it is safe to enable here.
1128 */
1129 if (check_feature2(FEATURE_GCR3TRPMODE))
1130 iommu_feature_enable(iommu, CONTROL_GCR3TRPMODE);
1131 }
1132
1133 /* sets a specific bit in the device table entry. */
set_dte_bit(struct dev_table_entry * dte,u8 bit)1134 static void set_dte_bit(struct dev_table_entry *dte, u8 bit)
1135 {
1136 int i = (bit >> 6) & 0x03;
1137 int _bit = bit & 0x3f;
1138
1139 dte->data[i] |= (1UL << _bit);
1140 }
1141
__reuse_device_table(struct amd_iommu * iommu)1142 static bool __reuse_device_table(struct amd_iommu *iommu)
1143 {
1144 struct amd_iommu_pci_seg *pci_seg = iommu->pci_seg;
1145 struct dev_table_entry *old_dev_tbl_entry;
1146 u32 lo, hi, old_devtb_size, devid;
1147 phys_addr_t old_devtb_phys;
1148 u16 dom_id;
1149 bool dte_v;
1150 u64 entry;
1151
1152 /* Each IOMMU use separate device table with the same size */
1153 lo = readl(iommu->mmio_base + MMIO_DEV_TABLE_OFFSET);
1154 hi = readl(iommu->mmio_base + MMIO_DEV_TABLE_OFFSET + 4);
1155 entry = (((u64) hi) << 32) + lo;
1156
1157 old_devtb_size = ((entry & ~PAGE_MASK) + 1) << 12;
1158 if (old_devtb_size != pci_seg->dev_table_size) {
1159 pr_err("The device table size of IOMMU:%d is not expected!\n",
1160 iommu->index);
1161 return false;
1162 }
1163
1164 /*
1165 * When SME is enabled in the first kernel, the entry includes the
1166 * memory encryption mask(sme_me_mask), we must remove the memory
1167 * encryption mask to obtain the true physical address in kdump kernel.
1168 */
1169 old_devtb_phys = __sme_clr(entry) & PAGE_MASK;
1170
1171 if (old_devtb_phys >= 0x100000000ULL) {
1172 pr_err("The address of old device table is above 4G, not trustworthy!\n");
1173 return false;
1174 }
1175
1176 /*
1177 * Re-use the previous kernel's device table for kdump.
1178 */
1179 pci_seg->old_dev_tbl_cpy = iommu_memremap(old_devtb_phys, pci_seg->dev_table_size);
1180 if (pci_seg->old_dev_tbl_cpy == NULL) {
1181 pr_err("Failed to remap memory for reusing old device table!\n");
1182 return false;
1183 }
1184
1185 for (devid = 0; devid <= pci_seg->last_bdf; devid++) {
1186 old_dev_tbl_entry = &pci_seg->old_dev_tbl_cpy[devid];
1187 dte_v = FIELD_GET(DTE_FLAG_V, old_dev_tbl_entry->data[0]);
1188 dom_id = FIELD_GET(DTE_DOMID_MASK, old_dev_tbl_entry->data[1]);
1189
1190 if (!dte_v || !dom_id)
1191 continue;
1192 /*
1193 * ID reservation can fail with -ENOSPC when there
1194 * are multiple devices present in the same domain,
1195 * hence check only for -ENOMEM.
1196 */
1197 if (amd_iommu_pdom_id_reserve(dom_id, GFP_KERNEL) == -ENOMEM)
1198 return false;
1199 }
1200
1201 return true;
1202 }
1203
reuse_device_table(void)1204 static bool reuse_device_table(void)
1205 {
1206 struct amd_iommu *iommu;
1207 struct amd_iommu_pci_seg *pci_seg;
1208
1209 if (!amd_iommu_pre_enabled)
1210 return false;
1211
1212 pr_warn("Translation is already enabled - trying to reuse translation structures\n");
1213
1214 /*
1215 * All IOMMUs within PCI segment shares common device table.
1216 * Hence reuse device table only once per PCI segment.
1217 */
1218 for_each_pci_segment(pci_seg) {
1219 for_each_iommu(iommu) {
1220 if (pci_seg->id != iommu->pci_seg->id)
1221 continue;
1222 if (!__reuse_device_table(iommu))
1223 return false;
1224 break;
1225 }
1226 }
1227
1228 return true;
1229 }
1230
amd_iommu_get_ivhd_dte_flags(u16 segid,u16 devid)1231 struct dev_table_entry *amd_iommu_get_ivhd_dte_flags(u16 segid, u16 devid)
1232 {
1233 struct ivhd_dte_flags *e;
1234 unsigned int best_len = UINT_MAX;
1235 struct dev_table_entry *dte = NULL;
1236
1237 for_each_ivhd_dte_flags(e) {
1238 /*
1239 * Need to go through the whole list to find the smallest range,
1240 * which contains the devid.
1241 */
1242 if ((e->segid == segid) &&
1243 (e->devid_first <= devid) && (devid <= e->devid_last)) {
1244 unsigned int len = e->devid_last - e->devid_first;
1245
1246 if (len < best_len) {
1247 dte = &(e->dte);
1248 best_len = len;
1249 }
1250 }
1251 }
1252 return dte;
1253 }
1254
search_ivhd_dte_flags(u16 segid,u16 first,u16 last)1255 static bool search_ivhd_dte_flags(u16 segid, u16 first, u16 last)
1256 {
1257 struct ivhd_dte_flags *e;
1258
1259 for_each_ivhd_dte_flags(e) {
1260 if ((e->segid == segid) &&
1261 (e->devid_first == first) &&
1262 (e->devid_last == last))
1263 return true;
1264 }
1265 return false;
1266 }
1267
1268 /*
1269 * This function takes the device specific flags read from the ACPI
1270 * table and sets up the device table entry with that information
1271 */
1272 static void __init
set_dev_entry_from_acpi_range(struct amd_iommu * iommu,u16 first,u16 last,u32 flags,u32 ext_flags)1273 set_dev_entry_from_acpi_range(struct amd_iommu *iommu, u16 first, u16 last,
1274 u32 flags, u32 ext_flags)
1275 {
1276 int i;
1277 struct dev_table_entry dte = {};
1278
1279 /* Parse IVHD DTE setting flags and store information */
1280 if (flags) {
1281 struct ivhd_dte_flags *d;
1282
1283 if (search_ivhd_dte_flags(iommu->pci_seg->id, first, last))
1284 return;
1285
1286 d = kzalloc_obj(struct ivhd_dte_flags);
1287 if (!d)
1288 return;
1289
1290 pr_debug("%s: devid range %#x:%#x\n", __func__, first, last);
1291
1292 if (flags & ACPI_DEVFLAG_INITPASS)
1293 set_dte_bit(&dte, DEV_ENTRY_INIT_PASS);
1294 if (flags & ACPI_DEVFLAG_EXTINT)
1295 set_dte_bit(&dte, DEV_ENTRY_EINT_PASS);
1296 if (flags & ACPI_DEVFLAG_NMI)
1297 set_dte_bit(&dte, DEV_ENTRY_NMI_PASS);
1298 if (flags & ACPI_DEVFLAG_SYSMGT1)
1299 set_dte_bit(&dte, DEV_ENTRY_SYSMGT1);
1300 if (flags & ACPI_DEVFLAG_SYSMGT2)
1301 set_dte_bit(&dte, DEV_ENTRY_SYSMGT2);
1302 if (flags & ACPI_DEVFLAG_LINT0)
1303 set_dte_bit(&dte, DEV_ENTRY_LINT0_PASS);
1304 if (flags & ACPI_DEVFLAG_LINT1)
1305 set_dte_bit(&dte, DEV_ENTRY_LINT1_PASS);
1306
1307 /* Apply erratum 63, which needs info in initial_dte */
1308 if (FIELD_GET(DTE_DATA1_SYSMGT_MASK, dte.data[1]) == 0x1)
1309 dte.data[0] |= DTE_FLAG_IW;
1310
1311 memcpy(&d->dte, &dte, sizeof(dte));
1312 d->segid = iommu->pci_seg->id;
1313 d->devid_first = first;
1314 d->devid_last = last;
1315 list_add_tail(&d->list, &amd_ivhd_dev_flags_list);
1316 }
1317
1318 for (i = first; i <= last; i++) {
1319 if (flags) {
1320 struct dev_table_entry *dev_table = get_dev_table(iommu);
1321
1322 memcpy(&dev_table[i], &dte, sizeof(dte));
1323 }
1324 amd_iommu_set_rlookup_table(iommu, i);
1325 }
1326 }
1327
set_dev_entry_from_acpi(struct amd_iommu * iommu,u16 devid,u32 flags,u32 ext_flags)1328 static void __init set_dev_entry_from_acpi(struct amd_iommu *iommu,
1329 u16 devid, u32 flags, u32 ext_flags)
1330 {
1331 set_dev_entry_from_acpi_range(iommu, devid, devid, flags, ext_flags);
1332 }
1333
add_special_device(u8 type,u8 id,u32 * devid,bool cmd_line)1334 int __init add_special_device(u8 type, u8 id, u32 *devid, bool cmd_line)
1335 {
1336 struct devid_map *entry;
1337 struct list_head *list;
1338
1339 if (type == IVHD_SPECIAL_IOAPIC)
1340 list = &ioapic_map;
1341 else if (type == IVHD_SPECIAL_HPET)
1342 list = &hpet_map;
1343 else
1344 return -EINVAL;
1345
1346 list_for_each_entry(entry, list, list) {
1347 if (!(entry->id == id && entry->cmd_line))
1348 continue;
1349
1350 pr_info("Command-line override present for %s id %d - ignoring\n",
1351 type == IVHD_SPECIAL_IOAPIC ? "IOAPIC" : "HPET", id);
1352
1353 *devid = entry->devid;
1354
1355 return 0;
1356 }
1357
1358 entry = kzalloc_obj(*entry);
1359 if (!entry)
1360 return -ENOMEM;
1361
1362 entry->id = id;
1363 entry->devid = *devid;
1364 entry->cmd_line = cmd_line;
1365
1366 list_add_tail(&entry->list, list);
1367
1368 return 0;
1369 }
1370
add_acpi_hid_device(u8 * hid,u8 * uid,u32 * devid,bool cmd_line)1371 static int __init add_acpi_hid_device(u8 *hid, u8 *uid, u32 *devid,
1372 bool cmd_line)
1373 {
1374 struct acpihid_map_entry *entry;
1375 struct list_head *list = &acpihid_map;
1376
1377 list_for_each_entry(entry, list, list) {
1378 if (strcmp(entry->hid, hid) ||
1379 (*uid && *entry->uid && strcmp(entry->uid, uid)) ||
1380 !entry->cmd_line)
1381 continue;
1382
1383 pr_info("Command-line override for hid:%s uid:%s\n",
1384 hid, uid);
1385 *devid = entry->devid;
1386 return 0;
1387 }
1388
1389 entry = kzalloc_obj(*entry);
1390 if (!entry)
1391 return -ENOMEM;
1392
1393 memcpy(entry->uid, uid, strlen(uid));
1394 memcpy(entry->hid, hid, strlen(hid));
1395 entry->devid = *devid;
1396 entry->cmd_line = cmd_line;
1397 entry->root_devid = (entry->devid & (~0x7));
1398
1399 pr_info("%s, add hid:%s, uid:%s, rdevid:%#x\n",
1400 entry->cmd_line ? "cmd" : "ivrs",
1401 entry->hid, entry->uid, entry->root_devid);
1402
1403 list_add_tail(&entry->list, list);
1404 return 0;
1405 }
1406
add_early_maps(void)1407 static int __init add_early_maps(void)
1408 {
1409 int i, ret;
1410
1411 for (i = 0; i < early_ioapic_map_size; ++i) {
1412 ret = add_special_device(IVHD_SPECIAL_IOAPIC,
1413 early_ioapic_map[i].id,
1414 &early_ioapic_map[i].devid,
1415 early_ioapic_map[i].cmd_line);
1416 if (ret)
1417 return ret;
1418 }
1419
1420 for (i = 0; i < early_hpet_map_size; ++i) {
1421 ret = add_special_device(IVHD_SPECIAL_HPET,
1422 early_hpet_map[i].id,
1423 &early_hpet_map[i].devid,
1424 early_hpet_map[i].cmd_line);
1425 if (ret)
1426 return ret;
1427 }
1428
1429 for (i = 0; i < early_acpihid_map_size; ++i) {
1430 ret = add_acpi_hid_device(early_acpihid_map[i].hid,
1431 early_acpihid_map[i].uid,
1432 &early_acpihid_map[i].devid,
1433 early_acpihid_map[i].cmd_line);
1434 if (ret)
1435 return ret;
1436 }
1437
1438 return 0;
1439 }
1440
1441 /*
1442 * Takes a pointer to an AMD IOMMU entry in the ACPI table and
1443 * initializes the hardware and our data structures with it.
1444 */
init_iommu_from_acpi(struct amd_iommu * iommu,struct ivhd_header * h)1445 static int __init init_iommu_from_acpi(struct amd_iommu *iommu,
1446 struct ivhd_header *h)
1447 {
1448 u8 *p = (u8 *)h;
1449 u8 *end = p, flags = 0;
1450 u16 devid = 0, devid_start = 0, devid_to = 0, seg_id;
1451 u32 dev_i, ext_flags = 0;
1452 bool alias = false;
1453 struct ivhd_entry *e;
1454 struct amd_iommu_pci_seg *pci_seg = iommu->pci_seg;
1455 u32 ivhd_size;
1456 int ret;
1457
1458
1459 ret = add_early_maps();
1460 if (ret)
1461 return ret;
1462
1463 amd_iommu_apply_ivrs_quirks();
1464
1465 /*
1466 * First save the recommended feature enable bits from ACPI
1467 */
1468 iommu->acpi_flags = h->flags;
1469
1470 /*
1471 * Done. Now parse the device entries
1472 */
1473 ivhd_size = get_ivhd_header_size(h);
1474 if (!ivhd_size) {
1475 pr_err("Unsupported IVHD type %#x\n", h->type);
1476 return -EINVAL;
1477 }
1478
1479 p += ivhd_size;
1480
1481 end += h->length;
1482
1483
1484 while (p < end) {
1485 e = (struct ivhd_entry *)p;
1486 seg_id = pci_seg->id;
1487
1488 switch (e->type) {
1489 case IVHD_DEV_ALL:
1490
1491 DUMP_printk(" DEV_ALL\t\t\tsetting: %#02x\n", e->flags);
1492 set_dev_entry_from_acpi_range(iommu, 0, pci_seg->last_bdf, e->flags, 0);
1493 break;
1494 case IVHD_DEV_SELECT:
1495
1496 DUMP_printk(" DEV_SELECT\t\t\tdevid: %04x:%02x:%02x.%x flags: %#02x\n",
1497 seg_id, PCI_BUS_NUM(e->devid),
1498 PCI_SLOT(e->devid),
1499 PCI_FUNC(e->devid),
1500 e->flags);
1501
1502 devid = e->devid;
1503 set_dev_entry_from_acpi(iommu, devid, e->flags, 0);
1504 break;
1505 case IVHD_DEV_SELECT_RANGE_START:
1506
1507 DUMP_printk(" DEV_SELECT_RANGE_START\tdevid: %04x:%02x:%02x.%x flags: %#02x\n",
1508 seg_id, PCI_BUS_NUM(e->devid),
1509 PCI_SLOT(e->devid),
1510 PCI_FUNC(e->devid),
1511 e->flags);
1512
1513 devid_start = e->devid;
1514 flags = e->flags;
1515 ext_flags = 0;
1516 alias = false;
1517 break;
1518 case IVHD_DEV_ALIAS:
1519
1520 DUMP_printk(" DEV_ALIAS\t\t\tdevid: %04x:%02x:%02x.%x flags: %#02x devid_to: %02x:%02x.%x\n",
1521 seg_id, PCI_BUS_NUM(e->devid),
1522 PCI_SLOT(e->devid),
1523 PCI_FUNC(e->devid),
1524 e->flags,
1525 PCI_BUS_NUM(e->ext >> 8),
1526 PCI_SLOT(e->ext >> 8),
1527 PCI_FUNC(e->ext >> 8));
1528
1529 devid = e->devid;
1530 devid_to = e->ext >> 8;
1531 set_dev_entry_from_acpi(iommu, devid , e->flags, 0);
1532 set_dev_entry_from_acpi(iommu, devid_to, e->flags, 0);
1533 pci_seg->alias_table[devid] = devid_to;
1534 break;
1535 case IVHD_DEV_ALIAS_RANGE:
1536
1537 DUMP_printk(" DEV_ALIAS_RANGE\t\tdevid: %04x:%02x:%02x.%x flags: %#02x devid_to: %04x:%02x:%02x.%x\n",
1538 seg_id, PCI_BUS_NUM(e->devid),
1539 PCI_SLOT(e->devid),
1540 PCI_FUNC(e->devid),
1541 e->flags,
1542 seg_id, PCI_BUS_NUM(e->ext >> 8),
1543 PCI_SLOT(e->ext >> 8),
1544 PCI_FUNC(e->ext >> 8));
1545
1546 devid_start = e->devid;
1547 flags = e->flags;
1548 devid_to = e->ext >> 8;
1549 ext_flags = 0;
1550 alias = true;
1551 break;
1552 case IVHD_DEV_EXT_SELECT:
1553
1554 DUMP_printk(" DEV_EXT_SELECT\t\tdevid: %04x:%02x:%02x.%x flags: %#02x ext: %08x\n",
1555 seg_id, PCI_BUS_NUM(e->devid),
1556 PCI_SLOT(e->devid),
1557 PCI_FUNC(e->devid),
1558 e->flags, e->ext);
1559
1560 devid = e->devid;
1561 set_dev_entry_from_acpi(iommu, devid, e->flags,
1562 e->ext);
1563 break;
1564 case IVHD_DEV_EXT_SELECT_RANGE:
1565
1566 DUMP_printk(" DEV_EXT_SELECT_RANGE\tdevid: %04x:%02x:%02x.%x flags: %#02x ext: %08x\n",
1567 seg_id, PCI_BUS_NUM(e->devid),
1568 PCI_SLOT(e->devid),
1569 PCI_FUNC(e->devid),
1570 e->flags, e->ext);
1571
1572 devid_start = e->devid;
1573 flags = e->flags;
1574 ext_flags = e->ext;
1575 alias = false;
1576 break;
1577 case IVHD_DEV_RANGE_END:
1578
1579 DUMP_printk(" DEV_RANGE_END\t\tdevid: %04x:%02x:%02x.%x\n",
1580 seg_id, PCI_BUS_NUM(e->devid),
1581 PCI_SLOT(e->devid),
1582 PCI_FUNC(e->devid));
1583
1584 devid = e->devid;
1585 if (alias) {
1586 for (dev_i = devid_start; dev_i <= devid; ++dev_i)
1587 pci_seg->alias_table[dev_i] = devid_to;
1588 set_dev_entry_from_acpi(iommu, devid_to, flags, ext_flags);
1589 }
1590 set_dev_entry_from_acpi_range(iommu, devid_start, devid, flags, ext_flags);
1591 break;
1592 case IVHD_DEV_SPECIAL: {
1593 u8 handle, type;
1594 const char *var;
1595 u32 devid;
1596 int ret;
1597
1598 handle = e->ext & 0xff;
1599 devid = PCI_SEG_DEVID_TO_SBDF(seg_id, (e->ext >> 8));
1600 type = (e->ext >> 24) & 0xff;
1601
1602 if (type == IVHD_SPECIAL_IOAPIC)
1603 var = "IOAPIC";
1604 else if (type == IVHD_SPECIAL_HPET)
1605 var = "HPET";
1606 else
1607 var = "UNKNOWN";
1608
1609 DUMP_printk(" DEV_SPECIAL(%s[%d])\t\tdevid: %04x:%02x:%02x.%x, flags: %#02x\n",
1610 var, (int)handle,
1611 seg_id, PCI_BUS_NUM(devid),
1612 PCI_SLOT(devid),
1613 PCI_FUNC(devid),
1614 e->flags);
1615
1616 ret = add_special_device(type, handle, &devid, false);
1617 if (ret)
1618 return ret;
1619
1620 /*
1621 * add_special_device might update the devid in case a
1622 * command-line override is present. So call
1623 * set_dev_entry_from_acpi after add_special_device.
1624 */
1625 set_dev_entry_from_acpi(iommu, devid, e->flags, 0);
1626
1627 break;
1628 }
1629 case IVHD_DEV_ACPI_HID: {
1630 u32 devid;
1631 u8 hid[ACPIHID_HID_LEN];
1632 u8 uid[ACPIHID_UID_LEN];
1633 int ret;
1634
1635 if (h->type != 0x40) {
1636 pr_err(FW_BUG "Invalid IVHD device type %#x\n",
1637 e->type);
1638 break;
1639 }
1640
1641 BUILD_BUG_ON(sizeof(e->ext_hid) != ACPIHID_HID_LEN - 1);
1642 memcpy(hid, &e->ext_hid, ACPIHID_HID_LEN - 1);
1643 hid[ACPIHID_HID_LEN - 1] = '\0';
1644
1645 if (!(*hid)) {
1646 pr_err(FW_BUG "Invalid HID.\n");
1647 break;
1648 }
1649
1650 uid[0] = '\0';
1651 switch (e->uidf) {
1652 case UID_NOT_PRESENT:
1653
1654 if (e->uidl != 0)
1655 pr_warn(FW_BUG "Invalid UID length.\n");
1656
1657 break;
1658 case UID_IS_INTEGER:
1659
1660 sprintf(uid, "%d", e->uid);
1661
1662 break;
1663 case UID_IS_CHARACTER:
1664
1665 memcpy(uid, &e->uid, e->uidl);
1666 uid[e->uidl] = '\0';
1667
1668 break;
1669 default:
1670 break;
1671 }
1672
1673 devid = PCI_SEG_DEVID_TO_SBDF(seg_id, e->devid);
1674 DUMP_printk(" DEV_ACPI_HID(%s[%s])\t\tdevid: %04x:%02x:%02x.%x, flags: %#02x\n",
1675 hid, uid, seg_id,
1676 PCI_BUS_NUM(devid),
1677 PCI_SLOT(devid),
1678 PCI_FUNC(devid),
1679 e->flags);
1680
1681 flags = e->flags;
1682
1683 ret = add_acpi_hid_device(hid, uid, &devid, false);
1684 if (ret)
1685 return ret;
1686
1687 /*
1688 * add_special_device might update the devid in case a
1689 * command-line override is present. So call
1690 * set_dev_entry_from_acpi after add_special_device.
1691 */
1692 set_dev_entry_from_acpi(iommu, devid, e->flags, 0);
1693
1694 break;
1695 }
1696 default:
1697 break;
1698 }
1699
1700 p += ivhd_entry_length(p);
1701 }
1702
1703 return 0;
1704 }
1705
1706 /* Allocate PCI segment data structure */
alloc_pci_segment(u16 id,struct acpi_table_header * ivrs_base)1707 static struct amd_iommu_pci_seg *__init alloc_pci_segment(u16 id,
1708 struct acpi_table_header *ivrs_base)
1709 {
1710 struct amd_iommu_pci_seg *pci_seg;
1711 int last_bdf;
1712
1713 /*
1714 * First parse ACPI tables to find the largest Bus/Dev/Func we need to
1715 * handle in this PCI segment. Upon this information the shared data
1716 * structures for the PCI segments in the system will be allocated.
1717 */
1718 last_bdf = find_last_devid_acpi(ivrs_base, id);
1719 if (last_bdf < 0)
1720 return NULL;
1721
1722 pci_seg = kzalloc_obj(struct amd_iommu_pci_seg);
1723 if (pci_seg == NULL)
1724 return NULL;
1725
1726 pci_seg->last_bdf = last_bdf;
1727 DUMP_printk("PCI segment : 0x%0x, last bdf : 0x%04x\n", id, last_bdf);
1728 pci_seg->dev_table_size =
1729 max(roundup_pow_of_two((last_bdf + 1) * DEV_TABLE_ENTRY_SIZE),
1730 SZ_4K);
1731
1732 pci_seg->id = id;
1733 init_llist_head(&pci_seg->dev_data_list);
1734 INIT_LIST_HEAD(&pci_seg->unity_map);
1735 list_add_tail(&pci_seg->list, &amd_iommu_pci_seg_list);
1736
1737 if (alloc_dev_table(pci_seg))
1738 goto err_free_pci_seg;
1739 if (alloc_alias_table(pci_seg))
1740 goto err_free_dev_table;
1741 if (alloc_rlookup_table(pci_seg))
1742 goto err_free_alias_table;
1743
1744 return pci_seg;
1745
1746 err_free_alias_table:
1747 free_alias_table(pci_seg);
1748 err_free_dev_table:
1749 free_dev_table(pci_seg);
1750 err_free_pci_seg:
1751 list_del(&pci_seg->list);
1752 kfree(pci_seg);
1753 return NULL;
1754 }
1755
get_pci_segment(u16 id,struct acpi_table_header * ivrs_base)1756 static struct amd_iommu_pci_seg *__init get_pci_segment(u16 id,
1757 struct acpi_table_header *ivrs_base)
1758 {
1759 struct amd_iommu_pci_seg *pci_seg;
1760
1761 for_each_pci_segment(pci_seg) {
1762 if (pci_seg->id == id)
1763 return pci_seg;
1764 }
1765
1766 return alloc_pci_segment(id, ivrs_base);
1767 }
1768
free_pci_segments(void)1769 static void __init free_pci_segments(void)
1770 {
1771 struct amd_iommu_pci_seg *pci_seg, *next;
1772
1773 for_each_pci_segment_safe(pci_seg, next) {
1774 list_del(&pci_seg->list);
1775 free_irq_lookup_table(pci_seg);
1776 free_rlookup_table(pci_seg);
1777 free_alias_table(pci_seg);
1778 free_dev_table(pci_seg);
1779 kfree(pci_seg);
1780 }
1781 }
1782
free_sysfs(struct amd_iommu * iommu)1783 static void __init free_sysfs(struct amd_iommu *iommu)
1784 {
1785 if (iommu->iommu.dev) {
1786 iommu_device_unregister(&iommu->iommu);
1787 iommu_device_sysfs_remove(&iommu->iommu);
1788 }
1789 }
1790
free_iommu_one(struct amd_iommu * iommu)1791 static void __init free_iommu_one(struct amd_iommu *iommu)
1792 {
1793 free_sysfs(iommu);
1794 free_iommu_buffers(iommu);
1795 amd_iommu_free_ppr_log(iommu);
1796 free_ga_log(iommu);
1797 iommu_unmap_mmio_space(iommu);
1798 amd_iommu_iopf_uninit(iommu);
1799 }
1800
free_iommu_all(void)1801 static void __init free_iommu_all(void)
1802 {
1803 struct amd_iommu *iommu, *next;
1804
1805 for_each_iommu_safe(iommu, next) {
1806 list_del(&iommu->list);
1807 free_iommu_one(iommu);
1808 kfree(iommu);
1809 }
1810 }
1811
1812 /*
1813 * Family15h Model 10h-1fh erratum 746 (IOMMU Logging May Stall Translations)
1814 * Workaround:
1815 * BIOS should disable L2B micellaneous clock gating by setting
1816 * L2_L2B_CK_GATE_CONTROL[CKGateL2BMiscDisable](D0F2xF4_x90[2]) = 1b
1817 */
amd_iommu_erratum_746_workaround(struct amd_iommu * iommu)1818 static void amd_iommu_erratum_746_workaround(struct amd_iommu *iommu)
1819 {
1820 u32 value;
1821
1822 if ((boot_cpu_data.x86 != 0x15) ||
1823 (boot_cpu_data.x86_model < 0x10) ||
1824 (boot_cpu_data.x86_model > 0x1f))
1825 return;
1826
1827 pci_write_config_dword(iommu->dev, 0xf0, 0x90);
1828 pci_read_config_dword(iommu->dev, 0xf4, &value);
1829
1830 if (value & BIT(2))
1831 return;
1832
1833 /* Select NB indirect register 0x90 and enable writing */
1834 pci_write_config_dword(iommu->dev, 0xf0, 0x90 | (1 << 8));
1835
1836 pci_write_config_dword(iommu->dev, 0xf4, value | 0x4);
1837 pci_info(iommu->dev, "Applying erratum 746 workaround\n");
1838
1839 /* Clear the enable writing bit */
1840 pci_write_config_dword(iommu->dev, 0xf0, 0x90);
1841 }
1842
1843 /*
1844 * Family15h Model 30h-3fh (IOMMU Mishandles ATS Write Permission)
1845 * Workaround:
1846 * BIOS should enable ATS write permission check by setting
1847 * L2_DEBUG_3[AtsIgnoreIWDis](D0F2xF4_x47[0]) = 1b
1848 */
amd_iommu_ats_write_check_workaround(struct amd_iommu * iommu)1849 static void amd_iommu_ats_write_check_workaround(struct amd_iommu *iommu)
1850 {
1851 u32 value;
1852
1853 if ((boot_cpu_data.x86 != 0x15) ||
1854 (boot_cpu_data.x86_model < 0x30) ||
1855 (boot_cpu_data.x86_model > 0x3f))
1856 return;
1857
1858 /* Test L2_DEBUG_3[AtsIgnoreIWDis] == 1 */
1859 value = iommu_read_l2(iommu, 0x47);
1860
1861 if (value & BIT(0))
1862 return;
1863
1864 /* Set L2_DEBUG_3[AtsIgnoreIWDis] = 1 */
1865 iommu_write_l2(iommu, 0x47, value | BIT(0));
1866
1867 pci_info(iommu->dev, "Applying ATS write check workaround\n");
1868 }
1869
1870 /*
1871 * This function glues the initialization function for one IOMMU
1872 * together and also allocates the command buffer and programs the
1873 * hardware. It does NOT enable the IOMMU. This is done afterwards.
1874 */
init_iommu_one(struct amd_iommu * iommu,struct ivhd_header * h,struct acpi_table_header * ivrs_base)1875 static int __init init_iommu_one(struct amd_iommu *iommu, struct ivhd_header *h,
1876 struct acpi_table_header *ivrs_base)
1877 {
1878 struct amd_iommu_pci_seg *pci_seg;
1879
1880 pci_seg = get_pci_segment(h->pci_seg, ivrs_base);
1881 if (pci_seg == NULL)
1882 return -ENOMEM;
1883 iommu->pci_seg = pci_seg;
1884
1885 raw_spin_lock_init(&iommu->lock);
1886 iommu->cmd_sem_val = 0;
1887
1888 /* Add IOMMU to internal data structures */
1889 list_add_tail(&iommu->list, &amd_iommu_list);
1890 iommu->index = amd_iommus_present++;
1891
1892 if (unlikely(iommu->index >= MAX_IOMMUS)) {
1893 WARN(1, "System has more IOMMUs than supported by this driver\n");
1894 return -ENOSYS;
1895 }
1896
1897 /*
1898 * Copy data from ACPI table entry to the iommu struct
1899 */
1900 iommu->devid = h->devid;
1901 iommu->cap_ptr = h->cap_ptr;
1902 iommu->mmio_phys = h->mmio_phys;
1903
1904 switch (h->type) {
1905 case 0x10:
1906 /* Check if IVHD EFR contains proper max banks/counters */
1907 if ((h->efr_attr != 0) &&
1908 ((h->efr_attr & (0xF << 13)) != 0) &&
1909 ((h->efr_attr & (0x3F << 17)) != 0))
1910 iommu->mmio_phys_end = MMIO_REG_END_OFFSET;
1911 else
1912 iommu->mmio_phys_end = MMIO_CNTR_CONF_OFFSET;
1913
1914 /* GAM requires GA mode. */
1915 if ((h->efr_attr & (0x1 << IOMMU_FEAT_GASUP_SHIFT)) == 0)
1916 amd_iommu_guest_ir = AMD_IOMMU_GUEST_IR_LEGACY;
1917 break;
1918 case 0x11:
1919 case 0x40:
1920 if (h->efr_reg & (1 << 9))
1921 iommu->mmio_phys_end = MMIO_REG_END_OFFSET;
1922 else
1923 iommu->mmio_phys_end = MMIO_CNTR_CONF_OFFSET;
1924
1925 if (h->efr_attr & BIT(IOMMU_IVHD_ATTR_HATDIS_SHIFT)) {
1926 pr_warn_once("Host Address Translation is not supported.\n");
1927 amd_iommu_hatdis = true;
1928 }
1929
1930 /* XT and GAM require GA mode. */
1931 if ((h->efr_reg & (0x1 << IOMMU_EFR_GASUP_SHIFT)) == 0) {
1932 amd_iommu_guest_ir = AMD_IOMMU_GUEST_IR_LEGACY;
1933 break;
1934 } else {
1935 if (h->efr_reg & BIT(IOMMU_EFR_XTSUP_SHIFT))
1936 amd_iommu_xt_mode = IRQ_REMAP_X2APIC_MODE;
1937 }
1938
1939 early_iommu_features_init(iommu, h);
1940
1941 break;
1942 default:
1943 return -EINVAL;
1944 }
1945
1946 iommu->mmio_base = iommu_map_mmio_space(iommu->mmio_phys,
1947 iommu->mmio_phys_end);
1948 if (!iommu->mmio_base)
1949 return -ENOMEM;
1950
1951 return init_iommu_from_acpi(iommu, h);
1952 }
1953
init_iommu_one_late(struct amd_iommu * iommu)1954 static int __init init_iommu_one_late(struct amd_iommu *iommu)
1955 {
1956 int ret;
1957
1958 ret = alloc_iommu_buffers(iommu);
1959 if (ret)
1960 return ret;
1961
1962 iommu->int_enabled = false;
1963
1964 init_translation_status(iommu);
1965 if (translation_pre_enabled(iommu) && !is_kdump_kernel()) {
1966 iommu_disable(iommu);
1967 clear_translation_pre_enabled(iommu);
1968 pr_warn("Translation was enabled for IOMMU:%d but we are not in kdump mode\n",
1969 iommu->index);
1970 }
1971 if (amd_iommu_pre_enabled)
1972 amd_iommu_pre_enabled = translation_pre_enabled(iommu);
1973
1974 if (amd_iommu_irq_remap) {
1975 ret = amd_iommu_create_irq_domain(iommu);
1976 if (ret)
1977 return ret;
1978 }
1979
1980 /*
1981 * Make sure IOMMU is not considered to translate itself. The IVRS
1982 * table tells us so, but this is a lie!
1983 */
1984 iommu->pci_seg->rlookup_table[iommu->devid] = NULL;
1985
1986 return 0;
1987 }
1988
1989 /**
1990 * get_highest_supported_ivhd_type - Look up the appropriate IVHD type
1991 * @ivrs: Pointer to the IVRS header
1992 *
1993 * This function search through all IVDB of the maximum supported IVHD
1994 */
get_highest_supported_ivhd_type(struct acpi_table_header * ivrs)1995 static u8 get_highest_supported_ivhd_type(struct acpi_table_header *ivrs)
1996 {
1997 u8 *base = (u8 *)ivrs;
1998 struct ivhd_header *ivhd = (struct ivhd_header *)
1999 (base + IVRS_HEADER_LENGTH);
2000 u8 last_type = ivhd->type;
2001 u16 devid = ivhd->devid;
2002
2003 while (((u8 *)ivhd - base < ivrs->length) &&
2004 (ivhd->type <= ACPI_IVHD_TYPE_MAX_SUPPORTED)) {
2005 u8 *p = (u8 *) ivhd;
2006
2007 if (ivhd->devid == devid)
2008 last_type = ivhd->type;
2009 ivhd = (struct ivhd_header *)(p + ivhd->length);
2010 }
2011
2012 return last_type;
2013 }
2014
2015 /*
2016 * Iterates over all IOMMU entries in the ACPI table, allocates the
2017 * IOMMU structure and initializes it with init_iommu_one()
2018 */
init_iommu_all(struct acpi_table_header * table)2019 static int __init init_iommu_all(struct acpi_table_header *table)
2020 {
2021 u8 *p = (u8 *)table, *end = (u8 *)table;
2022 struct ivhd_header *h;
2023 struct amd_iommu *iommu;
2024 int ret;
2025
2026 end += table->length;
2027 p += IVRS_HEADER_LENGTH;
2028
2029 /* Phase 1: Process all IVHD blocks */
2030 while (p < end) {
2031 h = (struct ivhd_header *)p;
2032 if (*p == amd_iommu_target_ivhd_type) {
2033
2034 DUMP_printk("device: %04x:%02x:%02x.%01x cap: %04x "
2035 "flags: %01x info %04x\n",
2036 h->pci_seg, PCI_BUS_NUM(h->devid),
2037 PCI_SLOT(h->devid), PCI_FUNC(h->devid),
2038 h->cap_ptr, h->flags, h->info);
2039 DUMP_printk(" mmio-addr: %016llx\n",
2040 h->mmio_phys);
2041
2042 iommu = kzalloc_obj(struct amd_iommu);
2043 if (iommu == NULL)
2044 return -ENOMEM;
2045
2046 ret = init_iommu_one(iommu, h, table);
2047 if (ret)
2048 return ret;
2049 }
2050 p += h->length;
2051
2052 }
2053 WARN_ON(p != end);
2054
2055 /* Phase 2 : Early feature support check */
2056 get_global_efr();
2057
2058 /* Phase 3 : Enabling IOMMU features */
2059 for_each_iommu(iommu) {
2060 ret = init_iommu_one_late(iommu);
2061 if (ret)
2062 return ret;
2063 }
2064
2065 return 0;
2066 }
2067
init_iommu_perf_ctr(struct amd_iommu * iommu)2068 static void init_iommu_perf_ctr(struct amd_iommu *iommu)
2069 {
2070 u64 val;
2071 struct pci_dev *pdev = iommu->dev;
2072
2073 if (!check_feature(FEATURE_PC))
2074 return;
2075
2076 amd_iommu_pc_present = true;
2077
2078 pci_info(pdev, "IOMMU performance counters supported\n");
2079
2080 val = readl(iommu->mmio_base + MMIO_CNTR_CONF_OFFSET);
2081 iommu->max_banks = (u8) ((val >> 12) & 0x3f);
2082 iommu->max_counters = (u8) ((val >> 7) & 0xf);
2083
2084 return;
2085 }
2086
amd_iommu_show_cap(struct device * dev,struct device_attribute * attr,char * buf)2087 static ssize_t amd_iommu_show_cap(struct device *dev,
2088 struct device_attribute *attr,
2089 char *buf)
2090 {
2091 struct amd_iommu *iommu = dev_to_amd_iommu(dev);
2092 return sysfs_emit(buf, "%x\n", iommu->cap);
2093 }
2094 static DEVICE_ATTR(cap, S_IRUGO, amd_iommu_show_cap, NULL);
2095
amd_iommu_show_features(struct device * dev,struct device_attribute * attr,char * buf)2096 static ssize_t amd_iommu_show_features(struct device *dev,
2097 struct device_attribute *attr,
2098 char *buf)
2099 {
2100 return sysfs_emit(buf, "%llx:%llx\n", amd_iommu_efr, amd_iommu_efr2);
2101 }
2102 static DEVICE_ATTR(features, S_IRUGO, amd_iommu_show_features, NULL);
2103
2104 static struct attribute *amd_iommu_attrs[] = {
2105 &dev_attr_cap.attr,
2106 &dev_attr_features.attr,
2107 NULL,
2108 };
2109
2110 static struct attribute_group amd_iommu_group = {
2111 .name = "amd-iommu",
2112 .attrs = amd_iommu_attrs,
2113 };
2114
2115 static const struct attribute_group *amd_iommu_groups[] = {
2116 &amd_iommu_group,
2117 NULL,
2118 };
2119
2120 /*
2121 * Note: IVHD 0x11 and 0x40 also contains exact copy
2122 * of the IOMMU Extended Feature Register [MMIO Offset 0030h].
2123 * Default to EFR in IVHD since it is available sooner (i.e. before PCI init).
2124 */
late_iommu_features_init(struct amd_iommu * iommu)2125 static void __init late_iommu_features_init(struct amd_iommu *iommu)
2126 {
2127 u64 features, features2;
2128
2129 if (!(iommu->cap & (1 << IOMMU_CAP_EFR)))
2130 return;
2131
2132 /* read extended feature bits */
2133 features = readq(iommu->mmio_base + MMIO_EXT_FEATURES);
2134 features2 = readq(iommu->mmio_base + MMIO_EXT_FEATURES2);
2135
2136 if (!amd_iommu_efr) {
2137 amd_iommu_efr = features;
2138 amd_iommu_efr2 = features2;
2139 return;
2140 }
2141
2142 /*
2143 * Sanity check and warn if EFR values from
2144 * IVHD and MMIO conflict.
2145 */
2146 if (features != amd_iommu_efr ||
2147 features2 != amd_iommu_efr2) {
2148 pr_warn(FW_WARN
2149 "EFR mismatch. Use IVHD EFR (%#llx : %#llx), EFR2 (%#llx : %#llx).\n",
2150 features, amd_iommu_efr,
2151 features2, amd_iommu_efr2);
2152 }
2153 }
2154
iommu_init_pci(struct amd_iommu * iommu)2155 static int __init iommu_init_pci(struct amd_iommu *iommu)
2156 {
2157 int cap_ptr = iommu->cap_ptr;
2158 int ret;
2159
2160 iommu->dev = pci_get_domain_bus_and_slot(iommu->pci_seg->id,
2161 PCI_BUS_NUM(iommu->devid),
2162 iommu->devid & 0xff);
2163 if (!iommu->dev)
2164 return -ENODEV;
2165
2166 /* ACPI _PRT won't have an IRQ for IOMMU */
2167 iommu->dev->irq_managed = 1;
2168
2169 pci_read_config_dword(iommu->dev, cap_ptr + MMIO_CAP_HDR_OFFSET,
2170 &iommu->cap);
2171
2172 if (!(iommu->cap & (1 << IOMMU_CAP_IOTLB)))
2173 amd_iommu_iotlb_sup = false;
2174
2175 late_iommu_features_init(iommu);
2176
2177 if (check_feature(FEATURE_GT)) {
2178 int glxval;
2179 u64 pasmax;
2180
2181 pasmax = FIELD_GET(FEATURE_PASMAX, amd_iommu_efr);
2182 iommu->iommu.max_pasids = (1 << (pasmax + 1)) - 1;
2183
2184 BUG_ON(iommu->iommu.max_pasids & ~PASID_MASK);
2185
2186 glxval = FIELD_GET(FEATURE_GLX, amd_iommu_efr);
2187
2188 if (amd_iommu_max_glx_val == -1)
2189 amd_iommu_max_glx_val = glxval;
2190 else
2191 amd_iommu_max_glx_val = min(amd_iommu_max_glx_val, glxval);
2192
2193 iommu_enable_gt(iommu);
2194 }
2195
2196 if (check_feature(FEATURE_PPR) && amd_iommu_alloc_ppr_log(iommu))
2197 return -ENOMEM;
2198
2199 if (iommu->cap & (1UL << IOMMU_CAP_NPCACHE)) {
2200 pr_info("Using strict mode due to virtualization\n");
2201 iommu_set_dma_strict();
2202 amd_iommu_np_cache = true;
2203 }
2204
2205 init_iommu_perf_ctr(iommu);
2206
2207 if (is_rd890_iommu(iommu->dev)) {
2208 int i, j;
2209
2210 iommu->root_pdev =
2211 pci_get_domain_bus_and_slot(iommu->pci_seg->id,
2212 iommu->dev->bus->number,
2213 PCI_DEVFN(0, 0));
2214
2215 /*
2216 * Some rd890 systems may not be fully reconfigured by the
2217 * BIOS, so it's necessary for us to store this information so
2218 * it can be reprogrammed on resume
2219 */
2220 pci_read_config_dword(iommu->dev, iommu->cap_ptr + 4,
2221 &iommu->stored_addr_lo);
2222 pci_read_config_dword(iommu->dev, iommu->cap_ptr + 8,
2223 &iommu->stored_addr_hi);
2224
2225 /* Low bit locks writes to configuration space */
2226 iommu->stored_addr_lo &= ~1;
2227
2228 for (i = 0; i < 6; i++)
2229 for (j = 0; j < 0x12; j++)
2230 iommu->stored_l1[i][j] = iommu_read_l1(iommu, i, j);
2231
2232 for (i = 0; i < 0x83; i++)
2233 iommu->stored_l2[i] = iommu_read_l2(iommu, i);
2234 }
2235
2236 amd_iommu_erratum_746_workaround(iommu);
2237 amd_iommu_ats_write_check_workaround(iommu);
2238
2239 ret = iommu_device_sysfs_add(&iommu->iommu, &iommu->dev->dev,
2240 amd_iommu_groups, "ivhd%d", iommu->index);
2241 if (ret)
2242 return ret;
2243
2244 /*
2245 * Allocate per IOMMU IOPF queue here so that in attach device path,
2246 * PRI capable device can be added to IOPF queue
2247 */
2248 if (amd_iommu_gt_ppr_supported()) {
2249 ret = amd_iommu_iopf_init(iommu);
2250 if (ret)
2251 return ret;
2252 }
2253
2254 ret = iommu_device_register(&iommu->iommu, &amd_iommu_ops, NULL);
2255 if (ret || amd_iommu_pgtable == PD_MODE_NONE) {
2256 /*
2257 * Remove sysfs if DMA translation is not supported by the
2258 * IOMMU. Do not return an error to enable IRQ remapping
2259 * in state_next(), DTE[V, TV] must eventually be set to 0.
2260 */
2261 iommu_device_sysfs_remove(&iommu->iommu);
2262 }
2263
2264 return pci_enable_device(iommu->dev);
2265 }
2266
print_iommu_info(void)2267 static void print_iommu_info(void)
2268 {
2269 int i;
2270 static const char * const feat_str[] = {
2271 "PreF", "PPR", "X2APIC", "NX", "GT", "[5]",
2272 "IA", "GA", "HE", "PC"
2273 };
2274
2275 if (amd_iommu_efr) {
2276 pr_info("Extended features (%#llx, %#llx):", amd_iommu_efr, amd_iommu_efr2);
2277
2278 for (i = 0; i < ARRAY_SIZE(feat_str); ++i) {
2279 if (check_feature(1ULL << i))
2280 pr_cont(" %s", feat_str[i]);
2281 }
2282
2283 if (check_feature(FEATURE_GAM_VAPIC))
2284 pr_cont(" GA_vAPIC");
2285
2286 if (check_feature(FEATURE_SNP))
2287 pr_cont(" SNP");
2288
2289 if (check_feature2(FEATURE_SEVSNPIO_SUP))
2290 pr_cont(" SEV-TIO");
2291
2292 pr_cont("\n");
2293 }
2294
2295 if (irq_remapping_enabled) {
2296 pr_info("Interrupt remapping enabled\n");
2297 if (amd_iommu_xt_mode == IRQ_REMAP_X2APIC_MODE)
2298 pr_info("X2APIC enabled\n");
2299 }
2300 if (amd_iommu_pgtable == PD_MODE_V2) {
2301 pr_info("V2 page table enabled (Paging mode : %d level)\n",
2302 amd_iommu_gpt_level);
2303 }
2304 }
2305
amd_iommu_init_pci(void)2306 static int __init amd_iommu_init_pci(void)
2307 {
2308 struct amd_iommu *iommu;
2309 struct amd_iommu_pci_seg *pci_seg;
2310 int ret;
2311
2312 /* Init global identity domain before registering IOMMU */
2313 amd_iommu_init_identity_domain();
2314
2315 for_each_iommu(iommu) {
2316 ret = iommu_init_pci(iommu);
2317 if (ret) {
2318 pr_err("IOMMU%d: Failed to initialize IOMMU Hardware (error=%d)!\n",
2319 iommu->index, ret);
2320 goto out;
2321 }
2322 /* Need to setup range after PCI init */
2323 iommu_set_cwwb_range(iommu);
2324 }
2325
2326 /*
2327 * Order is important here to make sure any unity map requirements are
2328 * fulfilled. The unity mappings are created and written to the device
2329 * table during the iommu_init_pci() call.
2330 *
2331 * After that we call init_device_table_dma() to make sure any
2332 * uninitialized DTE will block DMA, and in the end we flush the caches
2333 * of all IOMMUs to make sure the changes to the device table are
2334 * active.
2335 */
2336 for_each_pci_segment(pci_seg)
2337 init_device_table_dma(pci_seg);
2338
2339 for_each_iommu(iommu)
2340 amd_iommu_flush_all_caches(iommu);
2341
2342 print_iommu_info();
2343
2344 out:
2345 return ret;
2346 }
2347
2348 /****************************************************************************
2349 *
2350 * The following functions initialize the MSI interrupts for all IOMMUs
2351 * in the system. It's a bit challenging because there could be multiple
2352 * IOMMUs per PCI BDF but we can call pci_enable_msi(x) only once per
2353 * pci_dev.
2354 *
2355 ****************************************************************************/
2356
iommu_setup_msi(struct amd_iommu * iommu)2357 static int iommu_setup_msi(struct amd_iommu *iommu)
2358 {
2359 int r;
2360
2361 r = pci_enable_msi(iommu->dev);
2362 if (r)
2363 return r;
2364
2365 r = request_threaded_irq(iommu->dev->irq, NULL, amd_iommu_int_thread,
2366 IRQF_ONESHOT, "AMD-Vi", iommu);
2367 if (r) {
2368 pci_disable_msi(iommu->dev);
2369 return r;
2370 }
2371
2372 return 0;
2373 }
2374
2375 union intcapxt {
2376 u64 capxt;
2377 struct {
2378 u64 reserved_0 : 2,
2379 dest_mode_logical : 1,
2380 reserved_1 : 5,
2381 destid_0_23 : 24,
2382 vector : 8,
2383 reserved_2 : 16,
2384 destid_24_31 : 8;
2385 };
2386 } __attribute__ ((packed));
2387
2388
2389 static struct irq_chip intcapxt_controller;
2390
intcapxt_irqdomain_activate(struct irq_domain * domain,struct irq_data * irqd,bool reserve)2391 static int intcapxt_irqdomain_activate(struct irq_domain *domain,
2392 struct irq_data *irqd, bool reserve)
2393 {
2394 return 0;
2395 }
2396
intcapxt_irqdomain_deactivate(struct irq_domain * domain,struct irq_data * irqd)2397 static void intcapxt_irqdomain_deactivate(struct irq_domain *domain,
2398 struct irq_data *irqd)
2399 {
2400 }
2401
2402
intcapxt_irqdomain_alloc(struct irq_domain * domain,unsigned int virq,unsigned int nr_irqs,void * arg)2403 static int intcapxt_irqdomain_alloc(struct irq_domain *domain, unsigned int virq,
2404 unsigned int nr_irqs, void *arg)
2405 {
2406 struct irq_alloc_info *info = arg;
2407 int i, ret;
2408
2409 if (!info || info->type != X86_IRQ_ALLOC_TYPE_AMDVI)
2410 return -EINVAL;
2411
2412 ret = irq_domain_alloc_irqs_parent(domain, virq, nr_irqs, arg);
2413 if (ret < 0)
2414 return ret;
2415
2416 for (i = virq; i < virq + nr_irqs; i++) {
2417 struct irq_data *irqd = irq_domain_get_irq_data(domain, i);
2418
2419 irqd->chip = &intcapxt_controller;
2420 irqd->hwirq = info->hwirq;
2421 irqd->chip_data = info->data;
2422 __irq_set_handler(i, handle_edge_irq, 0, "edge");
2423 }
2424
2425 return ret;
2426 }
2427
intcapxt_irqdomain_free(struct irq_domain * domain,unsigned int virq,unsigned int nr_irqs)2428 static void intcapxt_irqdomain_free(struct irq_domain *domain, unsigned int virq,
2429 unsigned int nr_irqs)
2430 {
2431 irq_domain_free_irqs_top(domain, virq, nr_irqs);
2432 }
2433
2434
intcapxt_unmask_irq(struct irq_data * irqd)2435 static void intcapxt_unmask_irq(struct irq_data *irqd)
2436 {
2437 struct amd_iommu *iommu = irqd->chip_data;
2438 struct irq_cfg *cfg = irqd_cfg(irqd);
2439 union intcapxt xt;
2440
2441 xt.capxt = 0ULL;
2442 xt.dest_mode_logical = apic->dest_mode_logical;
2443 xt.vector = cfg->vector;
2444 xt.destid_0_23 = cfg->dest_apicid & GENMASK(23, 0);
2445 xt.destid_24_31 = cfg->dest_apicid >> 24;
2446
2447 writeq(xt.capxt, iommu->mmio_base + irqd->hwirq);
2448 }
2449
intcapxt_mask_irq(struct irq_data * irqd)2450 static void intcapxt_mask_irq(struct irq_data *irqd)
2451 {
2452 struct amd_iommu *iommu = irqd->chip_data;
2453
2454 writeq(0, iommu->mmio_base + irqd->hwirq);
2455 }
2456
2457
intcapxt_set_affinity(struct irq_data * irqd,const struct cpumask * mask,bool force)2458 static int intcapxt_set_affinity(struct irq_data *irqd,
2459 const struct cpumask *mask, bool force)
2460 {
2461 struct irq_data *parent = irqd->parent_data;
2462 int ret;
2463
2464 ret = parent->chip->irq_set_affinity(parent, mask, force);
2465 if (ret < 0 || ret == IRQ_SET_MASK_OK_DONE)
2466 return ret;
2467 return 0;
2468 }
2469
intcapxt_set_wake(struct irq_data * irqd,unsigned int on)2470 static int intcapxt_set_wake(struct irq_data *irqd, unsigned int on)
2471 {
2472 return on ? -EOPNOTSUPP : 0;
2473 }
2474
2475 static struct irq_chip intcapxt_controller = {
2476 .name = "IOMMU-MSI",
2477 .irq_unmask = intcapxt_unmask_irq,
2478 .irq_mask = intcapxt_mask_irq,
2479 .irq_ack = irq_chip_ack_parent,
2480 .irq_retrigger = irq_chip_retrigger_hierarchy,
2481 .irq_set_affinity = intcapxt_set_affinity,
2482 .irq_set_wake = intcapxt_set_wake,
2483 .flags = IRQCHIP_MASK_ON_SUSPEND | IRQCHIP_MOVE_DEFERRED,
2484 };
2485
2486 static const struct irq_domain_ops intcapxt_domain_ops = {
2487 .alloc = intcapxt_irqdomain_alloc,
2488 .free = intcapxt_irqdomain_free,
2489 .activate = intcapxt_irqdomain_activate,
2490 .deactivate = intcapxt_irqdomain_deactivate,
2491 };
2492
2493
2494 static struct irq_domain *iommu_irqdomain;
2495
iommu_get_irqdomain(void)2496 static struct irq_domain *iommu_get_irqdomain(void)
2497 {
2498 struct fwnode_handle *fn;
2499
2500 /* No need for locking here (yet) as the init is single-threaded */
2501 if (iommu_irqdomain)
2502 return iommu_irqdomain;
2503
2504 fn = irq_domain_alloc_named_fwnode("AMD-Vi-MSI");
2505 if (!fn)
2506 return NULL;
2507
2508 iommu_irqdomain = irq_domain_create_hierarchy(x86_vector_domain, 0, 0,
2509 fn, &intcapxt_domain_ops,
2510 NULL);
2511 if (!iommu_irqdomain)
2512 irq_domain_free_fwnode(fn);
2513
2514 return iommu_irqdomain;
2515 }
2516
__iommu_setup_intcapxt(struct amd_iommu * iommu,const char * devname,int hwirq,irq_handler_t thread_fn)2517 static int __iommu_setup_intcapxt(struct amd_iommu *iommu, const char *devname,
2518 int hwirq, irq_handler_t thread_fn)
2519 {
2520 struct irq_domain *domain;
2521 struct irq_alloc_info info;
2522 int irq, ret;
2523 int node = dev_to_node(&iommu->dev->dev);
2524
2525 domain = iommu_get_irqdomain();
2526 if (!domain)
2527 return -ENXIO;
2528
2529 init_irq_alloc_info(&info, NULL);
2530 info.type = X86_IRQ_ALLOC_TYPE_AMDVI;
2531 info.data = iommu;
2532 info.hwirq = hwirq;
2533
2534 irq = irq_domain_alloc_irqs(domain, 1, node, &info);
2535 if (irq < 0) {
2536 irq_domain_remove(domain);
2537 return irq;
2538 }
2539
2540 ret = request_threaded_irq(irq, NULL, thread_fn, IRQF_ONESHOT, devname,
2541 iommu);
2542 if (ret) {
2543 irq_domain_free_irqs(irq, 1);
2544 irq_domain_remove(domain);
2545 return ret;
2546 }
2547
2548 return 0;
2549 }
2550
iommu_setup_intcapxt(struct amd_iommu * iommu)2551 static int iommu_setup_intcapxt(struct amd_iommu *iommu)
2552 {
2553 int ret;
2554
2555 snprintf(iommu->evt_irq_name, sizeof(iommu->evt_irq_name),
2556 "AMD-Vi%d-Evt", iommu->index);
2557 ret = __iommu_setup_intcapxt(iommu, iommu->evt_irq_name,
2558 MMIO_INTCAPXT_EVT_OFFSET,
2559 amd_iommu_int_thread_evtlog);
2560 if (ret)
2561 return ret;
2562
2563 snprintf(iommu->ppr_irq_name, sizeof(iommu->ppr_irq_name),
2564 "AMD-Vi%d-PPR", iommu->index);
2565 ret = __iommu_setup_intcapxt(iommu, iommu->ppr_irq_name,
2566 MMIO_INTCAPXT_PPR_OFFSET,
2567 amd_iommu_int_thread_pprlog);
2568 if (ret)
2569 return ret;
2570
2571 #ifdef CONFIG_IRQ_REMAP
2572 snprintf(iommu->ga_irq_name, sizeof(iommu->ga_irq_name),
2573 "AMD-Vi%d-GA", iommu->index);
2574 ret = __iommu_setup_intcapxt(iommu, iommu->ga_irq_name,
2575 MMIO_INTCAPXT_GALOG_OFFSET,
2576 amd_iommu_int_thread_galog);
2577 #endif
2578
2579 return ret;
2580 }
2581
iommu_init_irq(struct amd_iommu * iommu)2582 static int iommu_init_irq(struct amd_iommu *iommu)
2583 {
2584 int ret;
2585
2586 if (iommu->int_enabled)
2587 goto enable_faults;
2588
2589 if (amd_iommu_xt_mode == IRQ_REMAP_X2APIC_MODE)
2590 ret = iommu_setup_intcapxt(iommu);
2591 else if (iommu->dev->msi_cap)
2592 ret = iommu_setup_msi(iommu);
2593 else
2594 ret = -ENODEV;
2595
2596 if (ret)
2597 return ret;
2598
2599 iommu->int_enabled = true;
2600 enable_faults:
2601
2602 if (amd_iommu_xt_mode == IRQ_REMAP_X2APIC_MODE)
2603 iommu_feature_enable(iommu, CONTROL_INTCAPXT_EN);
2604
2605 iommu_feature_enable(iommu, CONTROL_EVT_INT_EN);
2606
2607 return 0;
2608 }
2609
2610 /****************************************************************************
2611 *
2612 * The next functions belong to the third pass of parsing the ACPI
2613 * table. In this last pass the memory mapping requirements are
2614 * gathered (like exclusion and unity mapping ranges).
2615 *
2616 ****************************************************************************/
2617
free_unity_maps(void)2618 static void __init free_unity_maps(void)
2619 {
2620 struct unity_map_entry *entry, *next;
2621 struct amd_iommu_pci_seg *p, *pci_seg;
2622
2623 for_each_pci_segment_safe(pci_seg, p) {
2624 list_for_each_entry_safe(entry, next, &pci_seg->unity_map, list) {
2625 list_del(&entry->list);
2626 kfree(entry);
2627 }
2628 }
2629 }
2630
2631 /* called for unity map ACPI definition */
init_unity_map_range(struct ivmd_header * m,struct acpi_table_header * ivrs_base)2632 static int __init init_unity_map_range(struct ivmd_header *m,
2633 struct acpi_table_header *ivrs_base)
2634 {
2635 struct unity_map_entry *e = NULL;
2636 struct amd_iommu_pci_seg *pci_seg;
2637 char *s;
2638
2639 pci_seg = get_pci_segment(m->pci_seg, ivrs_base);
2640 if (pci_seg == NULL)
2641 return -ENOMEM;
2642
2643 e = kzalloc_obj(*e);
2644 if (e == NULL)
2645 return -ENOMEM;
2646
2647 switch (m->type) {
2648 default:
2649 kfree(e);
2650 return 0;
2651 case ACPI_IVMD_TYPE:
2652 s = "IVMD_TYPEi\t\t\t";
2653 e->devid_start = e->devid_end = m->devid;
2654 break;
2655 case ACPI_IVMD_TYPE_ALL:
2656 s = "IVMD_TYPE_ALL\t\t";
2657 e->devid_start = 0;
2658 e->devid_end = pci_seg->last_bdf;
2659 break;
2660 case ACPI_IVMD_TYPE_RANGE:
2661 s = "IVMD_TYPE_RANGE\t\t";
2662 e->devid_start = m->devid;
2663 e->devid_end = m->aux;
2664 break;
2665 }
2666 e->address_start = PAGE_ALIGN(m->range_start);
2667 e->address_end = e->address_start + PAGE_ALIGN(m->range_length);
2668 e->prot = m->flags >> 1;
2669
2670 /*
2671 * Treat per-device exclusion ranges as r/w unity-mapped regions
2672 * since some buggy BIOSes might lead to the overwritten exclusion
2673 * range (exclusion_start and exclusion_length members). This
2674 * happens when there are multiple exclusion ranges (IVMD entries)
2675 * defined in ACPI table.
2676 */
2677 if (m->flags & IVMD_FLAG_EXCL_RANGE)
2678 e->prot = (IVMD_FLAG_IW | IVMD_FLAG_IR) >> 1;
2679
2680 DUMP_printk("%s devid_start: %04x:%02x:%02x.%x devid_end: "
2681 "%04x:%02x:%02x.%x range_start: %016llx range_end: %016llx"
2682 " flags: %x\n", s, m->pci_seg,
2683 PCI_BUS_NUM(e->devid_start), PCI_SLOT(e->devid_start),
2684 PCI_FUNC(e->devid_start), m->pci_seg,
2685 PCI_BUS_NUM(e->devid_end),
2686 PCI_SLOT(e->devid_end), PCI_FUNC(e->devid_end),
2687 e->address_start, e->address_end, m->flags);
2688
2689 list_add_tail(&e->list, &pci_seg->unity_map);
2690
2691 return 0;
2692 }
2693
2694 /* iterates over all memory definitions we find in the ACPI table */
init_memory_definitions(struct acpi_table_header * table)2695 static int __init init_memory_definitions(struct acpi_table_header *table)
2696 {
2697 u8 *p = (u8 *)table, *end = (u8 *)table;
2698 struct ivmd_header *m;
2699
2700 end += table->length;
2701 p += IVRS_HEADER_LENGTH;
2702
2703 while (p < end) {
2704 m = (struct ivmd_header *)p;
2705 if (m->flags & (IVMD_FLAG_UNITY_MAP | IVMD_FLAG_EXCL_RANGE))
2706 init_unity_map_range(m, table);
2707
2708 p += m->length;
2709 }
2710
2711 return 0;
2712 }
2713
2714 /*
2715 * Init the device table to not allow DMA access for devices
2716 */
init_device_table_dma(struct amd_iommu_pci_seg * pci_seg)2717 static void init_device_table_dma(struct amd_iommu_pci_seg *pci_seg)
2718 {
2719 u32 devid;
2720 struct dev_table_entry *dev_table = pci_seg->dev_table;
2721
2722 if (!dev_table || amd_iommu_pgtable == PD_MODE_NONE)
2723 return;
2724
2725 for (devid = 0; devid <= pci_seg->last_bdf; ++devid) {
2726 set_dte_bit(&dev_table[devid], DEV_ENTRY_VALID);
2727 if (!amd_iommu_snp_en)
2728 set_dte_bit(&dev_table[devid], DEV_ENTRY_TRANSLATION);
2729 }
2730 }
2731
uninit_device_table_dma(struct amd_iommu_pci_seg * pci_seg)2732 static void __init uninit_device_table_dma(struct amd_iommu_pci_seg *pci_seg)
2733 {
2734 u32 devid;
2735 struct dev_table_entry *dev_table = pci_seg->dev_table;
2736
2737 if (dev_table == NULL)
2738 return;
2739
2740 for (devid = 0; devid <= pci_seg->last_bdf; ++devid) {
2741 dev_table[devid].data[0] = 0ULL;
2742 dev_table[devid].data[1] = 0ULL;
2743 }
2744 }
2745
init_device_table(void)2746 static void init_device_table(void)
2747 {
2748 struct amd_iommu_pci_seg *pci_seg;
2749 u32 devid;
2750
2751 if (!amd_iommu_irq_remap)
2752 return;
2753
2754 for_each_pci_segment(pci_seg) {
2755 for (devid = 0; devid <= pci_seg->last_bdf; ++devid)
2756 set_dte_bit(&pci_seg->dev_table[devid], DEV_ENTRY_IRQ_TBL_EN);
2757 }
2758 }
2759
iommu_init_flags(struct amd_iommu * iommu)2760 static void iommu_init_flags(struct amd_iommu *iommu)
2761 {
2762 iommu->acpi_flags & IVHD_FLAG_HT_TUN_EN_MASK ?
2763 iommu_feature_enable(iommu, CONTROL_HT_TUN_EN) :
2764 iommu_feature_disable(iommu, CONTROL_HT_TUN_EN);
2765
2766 iommu->acpi_flags & IVHD_FLAG_PASSPW_EN_MASK ?
2767 iommu_feature_enable(iommu, CONTROL_PASSPW_EN) :
2768 iommu_feature_disable(iommu, CONTROL_PASSPW_EN);
2769
2770 iommu->acpi_flags & IVHD_FLAG_RESPASSPW_EN_MASK ?
2771 iommu_feature_enable(iommu, CONTROL_RESPASSPW_EN) :
2772 iommu_feature_disable(iommu, CONTROL_RESPASSPW_EN);
2773
2774 iommu->acpi_flags & IVHD_FLAG_ISOC_EN_MASK ?
2775 iommu_feature_enable(iommu, CONTROL_ISOC_EN) :
2776 iommu_feature_disable(iommu, CONTROL_ISOC_EN);
2777
2778 /*
2779 * make IOMMU memory accesses cache coherent
2780 */
2781 iommu_feature_enable(iommu, CONTROL_COHERENT_EN);
2782
2783 /* Set IOTLB invalidation timeout to 1s */
2784 iommu_feature_set(iommu, CTRL_INV_TO_1S, CTRL_INV_TO_MASK, CONTROL_INV_TIMEOUT);
2785
2786 /* Enable Enhanced Peripheral Page Request Handling */
2787 if (check_feature(FEATURE_EPHSUP))
2788 iommu_feature_enable(iommu, CONTROL_EPH_EN);
2789 }
2790
iommu_apply_resume_quirks(struct amd_iommu * iommu)2791 static void iommu_apply_resume_quirks(struct amd_iommu *iommu)
2792 {
2793 int i, j;
2794 u32 ioc_feature_control;
2795 struct pci_dev *pdev = iommu->root_pdev;
2796
2797 /* RD890 BIOSes may not have completely reconfigured the iommu */
2798 if (!is_rd890_iommu(iommu->dev) || !pdev)
2799 return;
2800
2801 /*
2802 * First, we need to ensure that the iommu is enabled. This is
2803 * controlled by a register in the northbridge
2804 */
2805
2806 /* Select Northbridge indirect register 0x75 and enable writing */
2807 pci_write_config_dword(pdev, 0x60, 0x75 | (1 << 7));
2808 pci_read_config_dword(pdev, 0x64, &ioc_feature_control);
2809
2810 /* Enable the iommu */
2811 if (!(ioc_feature_control & 0x1))
2812 pci_write_config_dword(pdev, 0x64, ioc_feature_control | 1);
2813
2814 /* Restore the iommu BAR */
2815 pci_write_config_dword(iommu->dev, iommu->cap_ptr + 4,
2816 iommu->stored_addr_lo);
2817 pci_write_config_dword(iommu->dev, iommu->cap_ptr + 8,
2818 iommu->stored_addr_hi);
2819
2820 /* Restore the l1 indirect regs for each of the 6 l1s */
2821 for (i = 0; i < 6; i++)
2822 for (j = 0; j < 0x12; j++)
2823 iommu_write_l1(iommu, i, j, iommu->stored_l1[i][j]);
2824
2825 /* Restore the l2 indirect regs */
2826 for (i = 0; i < 0x83; i++)
2827 iommu_write_l2(iommu, i, iommu->stored_l2[i]);
2828
2829 /* Lock PCI setup registers */
2830 pci_write_config_dword(iommu->dev, iommu->cap_ptr + 4,
2831 iommu->stored_addr_lo | 1);
2832 }
2833
iommu_enable_ga(struct amd_iommu * iommu)2834 static void iommu_enable_ga(struct amd_iommu *iommu)
2835 {
2836 #ifdef CONFIG_IRQ_REMAP
2837 switch (amd_iommu_guest_ir) {
2838 case AMD_IOMMU_GUEST_IR_VAPIC:
2839 case AMD_IOMMU_GUEST_IR_LEGACY_GA:
2840 iommu_feature_enable(iommu, CONTROL_GA_EN);
2841 iommu->irte_ops = &irte_128_ops;
2842 break;
2843 default:
2844 iommu->irte_ops = &irte_32_ops;
2845 break;
2846 }
2847 #endif
2848 }
2849
iommu_disable_irtcachedis(struct amd_iommu * iommu)2850 static void iommu_disable_irtcachedis(struct amd_iommu *iommu)
2851 {
2852 iommu_feature_disable(iommu, CONTROL_IRTCACHEDIS);
2853 }
2854
iommu_enable_irtcachedis(struct amd_iommu * iommu)2855 static void iommu_enable_irtcachedis(struct amd_iommu *iommu)
2856 {
2857 u64 ctrl;
2858
2859 if (!amd_iommu_irtcachedis)
2860 return;
2861
2862 /*
2863 * Note:
2864 * The support for IRTCacheDis feature is dertermined by
2865 * checking if the bit is writable.
2866 */
2867 iommu_feature_enable(iommu, CONTROL_IRTCACHEDIS);
2868 ctrl = readq(iommu->mmio_base + MMIO_CONTROL_OFFSET);
2869 ctrl &= (1ULL << CONTROL_IRTCACHEDIS);
2870 if (ctrl)
2871 iommu->irtcachedis_enabled = true;
2872 pr_info("iommu%d (%#06x) : IRT cache is %s\n",
2873 iommu->index, iommu->devid,
2874 iommu->irtcachedis_enabled ? "disabled" : "enabled");
2875 }
2876
iommu_enable_2k_int(struct amd_iommu * iommu)2877 static void iommu_enable_2k_int(struct amd_iommu *iommu)
2878 {
2879 if (!FEATURE_NUM_INT_REMAP_SUP_2K(amd_iommu_efr2))
2880 return;
2881
2882 iommu_feature_set(iommu,
2883 CONTROL_NUM_INT_REMAP_MODE_2K,
2884 CONTROL_NUM_INT_REMAP_MODE_MASK,
2885 CONTROL_NUM_INT_REMAP_MODE);
2886 }
2887
early_enable_iommu(struct amd_iommu * iommu)2888 static void early_enable_iommu(struct amd_iommu *iommu)
2889 {
2890 iommu_disable(iommu);
2891 iommu_init_flags(iommu);
2892 iommu_set_device_table(iommu);
2893 iommu_enable_command_buffer(iommu);
2894 iommu_enable_gt(iommu);
2895 iommu_enable_ga(iommu);
2896 iommu_enable_xt(iommu);
2897 iommu_enable_irtcachedis(iommu);
2898 iommu_enable_2k_int(iommu);
2899 iommu_enable(iommu);
2900 amd_iommu_flush_all_caches(iommu);
2901 }
2902
2903 /*
2904 * This function finally enables all IOMMUs found in the system after
2905 * they have been initialized.
2906 *
2907 * Or if in kdump kernel and IOMMUs are all pre-enabled, try to reuse
2908 * the old content of device table entries. Not this case or reuse failed,
2909 * just continue as normal kernel does.
2910 */
early_enable_iommus(void)2911 static void early_enable_iommus(void)
2912 {
2913 struct amd_iommu *iommu;
2914 struct amd_iommu_pci_seg *pci_seg;
2915
2916 if (!reuse_device_table()) {
2917 /*
2918 * If come here because of failure in reusing device table from old
2919 * kernel with all IOMMUs enabled, print error message and try to
2920 * free allocated old_dev_tbl_cpy.
2921 */
2922 if (amd_iommu_pre_enabled) {
2923 pr_err("Failed to reuse DEV table from previous kernel.\n");
2924 /*
2925 * Bail out early if unable to remap/reuse DEV table from
2926 * previous kernel if SNP enabled as IOMMU commands will
2927 * time out without DEV table and cause kdump boot panic.
2928 */
2929 BUG_ON(check_feature(FEATURE_SNP));
2930 }
2931
2932 for_each_pci_segment(pci_seg) {
2933 if (pci_seg->old_dev_tbl_cpy != NULL) {
2934 memunmap((void *)pci_seg->old_dev_tbl_cpy);
2935 pci_seg->old_dev_tbl_cpy = NULL;
2936 }
2937 }
2938
2939 for_each_iommu(iommu) {
2940 clear_translation_pre_enabled(iommu);
2941 early_enable_iommu(iommu);
2942 }
2943 } else {
2944 pr_info("Reused DEV table from previous kernel.\n");
2945
2946 for_each_pci_segment(pci_seg) {
2947 iommu_free_pages(pci_seg->dev_table);
2948 pci_seg->dev_table = pci_seg->old_dev_tbl_cpy;
2949 }
2950
2951 for_each_iommu(iommu) {
2952 iommu_disable_command_buffer(iommu);
2953 iommu_disable_event_buffer(iommu);
2954 iommu_disable_irtcachedis(iommu);
2955 iommu_enable_command_buffer(iommu);
2956 iommu_enable_ga(iommu);
2957 iommu_enable_xt(iommu);
2958 iommu_enable_irtcachedis(iommu);
2959 iommu_enable_2k_int(iommu);
2960 iommu_set_device_table(iommu);
2961 amd_iommu_flush_all_caches(iommu);
2962 }
2963 }
2964 }
2965
enable_iommus_ppr(void)2966 static void enable_iommus_ppr(void)
2967 {
2968 struct amd_iommu *iommu;
2969
2970 if (!amd_iommu_gt_ppr_supported())
2971 return;
2972
2973 for_each_iommu(iommu)
2974 amd_iommu_enable_ppr_log(iommu);
2975 }
2976
enable_iommus_vapic(void)2977 static void enable_iommus_vapic(void)
2978 {
2979 #ifdef CONFIG_IRQ_REMAP
2980 u32 status, i;
2981 struct amd_iommu *iommu;
2982
2983 for_each_iommu(iommu) {
2984 /*
2985 * Disable GALog if already running. It could have been enabled
2986 * in the previous boot before kdump.
2987 */
2988 status = readl(iommu->mmio_base + MMIO_STATUS_OFFSET);
2989 if (!(status & MMIO_STATUS_GALOG_RUN_MASK))
2990 continue;
2991
2992 iommu_feature_disable(iommu, CONTROL_GALOG_EN);
2993 iommu_feature_disable(iommu, CONTROL_GAINT_EN);
2994
2995 /*
2996 * Need to set and poll check the GALOGRun bit to zero before
2997 * we can set/ modify GA Log registers safely.
2998 */
2999 for (i = 0; i < MMIO_STATUS_TIMEOUT; ++i) {
3000 status = readl(iommu->mmio_base + MMIO_STATUS_OFFSET);
3001 if (!(status & MMIO_STATUS_GALOG_RUN_MASK))
3002 break;
3003 udelay(10);
3004 }
3005
3006 if (WARN_ON(i >= MMIO_STATUS_TIMEOUT))
3007 return;
3008 }
3009
3010 if (!AMD_IOMMU_GUEST_IR_VAPIC(amd_iommu_guest_ir))
3011 return;
3012
3013 if (!check_feature(FEATURE_GAM_VAPIC)) {
3014 amd_iommu_guest_ir = AMD_IOMMU_GUEST_IR_LEGACY_GA;
3015 return;
3016 }
3017
3018 if (amd_iommu_snp_en &&
3019 !FEATURE_SNPAVICSUP_GAM(amd_iommu_efr2)) {
3020 pr_warn("Force to disable Virtual APIC due to SNP\n");
3021 amd_iommu_guest_ir = AMD_IOMMU_GUEST_IR_LEGACY_GA;
3022 return;
3023 }
3024
3025 /* Enabling GAM and SNPAVIC support */
3026 for_each_iommu(iommu) {
3027 if (iommu_init_ga_log(iommu) ||
3028 iommu_ga_log_enable(iommu))
3029 return;
3030
3031 iommu_feature_enable(iommu, CONTROL_GAM_EN);
3032 if (amd_iommu_snp_en)
3033 iommu_feature_enable(iommu, CONTROL_SNPAVIC_EN);
3034 }
3035
3036 amd_iommu_irq_ops.capability |= (1 << IRQ_POSTING_CAP);
3037 pr_info("Virtual APIC enabled\n");
3038 #endif
3039 }
3040
disable_iommus(void)3041 static void disable_iommus(void)
3042 {
3043 struct amd_iommu *iommu;
3044
3045 for_each_iommu(iommu)
3046 iommu_disable(iommu);
3047
3048 #ifdef CONFIG_IRQ_REMAP
3049 if (AMD_IOMMU_GUEST_IR_VAPIC(amd_iommu_guest_ir))
3050 amd_iommu_irq_ops.capability &= ~(1 << IRQ_POSTING_CAP);
3051 #endif
3052 }
3053
3054 /*
3055 * Suspend/Resume support
3056 * disable suspend until real resume implemented
3057 */
3058
amd_iommu_resume(void * data)3059 static void amd_iommu_resume(void *data)
3060 {
3061 struct amd_iommu *iommu;
3062
3063 for_each_iommu(iommu)
3064 iommu_apply_resume_quirks(iommu);
3065
3066 /* re-load the hardware */
3067 for_each_iommu(iommu)
3068 early_enable_iommu(iommu);
3069
3070 iommu_enable_event_buffer();
3071 amd_iommu_enable_interrupts();
3072 }
3073
amd_iommu_suspend(void * data)3074 static int amd_iommu_suspend(void *data)
3075 {
3076 /* disable IOMMUs to go out of the way for BIOS */
3077 disable_iommus();
3078
3079 return 0;
3080 }
3081
3082 static const struct syscore_ops amd_iommu_syscore_ops = {
3083 .suspend = amd_iommu_suspend,
3084 .resume = amd_iommu_resume,
3085 };
3086
3087 static struct syscore amd_iommu_syscore = {
3088 .ops = &amd_iommu_syscore_ops,
3089 };
3090
free_iommu_resources(void)3091 static void __init free_iommu_resources(void)
3092 {
3093 free_iommu_all();
3094 free_pci_segments();
3095 }
3096
check_sb_ioapic(int devid)3097 static bool __init check_sb_ioapic(int devid)
3098 {
3099 u8 bus = PCI_BUS_NUM(devid);
3100 u8 devfn = devid & 0xff;
3101 u16 val;
3102
3103 val = read_pci_config_16(bus, PCI_SLOT(devfn), PCI_FUNC(devfn),
3104 PCI_CLASS_DEVICE);
3105
3106 /*
3107 * The SB IOAPIC is integrated into the FCH (Southbridge), which is
3108 * exposed as an SMBus or ISA bridge in PCI config space.
3109 */
3110 return val == PCI_CLASS_SERIAL_SMBUS || val == PCI_CLASS_BRIDGE_ISA;
3111 }
3112
3113 /*
3114 * The Southbridge IOAPIC is assigned a GSI Base of 0 (handling interrupts
3115 * 0 through 23).
3116 */
get_sb_ioapic_id(void)3117 static int __init get_sb_ioapic_id(void)
3118 {
3119 int idx = mp_find_ioapic(0);
3120
3121 if (idx < 0)
3122 return -ENODEV;
3123
3124 return mpc_ioapic_id(idx);
3125 }
3126
check_ioapic_information(void)3127 static bool __init check_ioapic_information(void)
3128 {
3129 const char *fw_bug = FW_BUG;
3130 bool ret, has_sb_ioapic;
3131 int idx, sb_apicid;
3132
3133 has_sb_ioapic = false;
3134 ret = true;
3135
3136 /*
3137 * If we have map overrides on the kernel command line the
3138 * messages in this function might not describe firmware bugs
3139 * anymore - so be careful
3140 */
3141 if (cmdline_maps)
3142 fw_bug = "";
3143
3144 sb_apicid = get_sb_ioapic_id();
3145 if (sb_apicid < 0) {
3146 /*
3147 * Lack of SB IOAPIC registration is not a firmware bug,
3148 * e.g. kernel booted with noapic or noacpi.
3149 */
3150 fw_bug = "";
3151 goto out;
3152 }
3153
3154 for (idx = 0; idx < nr_ioapics; idx++) {
3155 int devid, id = mpc_ioapic_id(idx);
3156
3157 devid = get_ioapic_devid(id);
3158 if (devid < 0) {
3159 pr_err("%s: IOAPIC[%d] not in IVRS table\n",
3160 fw_bug, id);
3161 ret = false;
3162 } else if (id == sb_apicid && check_sb_ioapic(devid)) {
3163 has_sb_ioapic = true;
3164 }
3165 }
3166 out:
3167 if (!has_sb_ioapic) {
3168 /*
3169 * We expect the SB IOAPIC to be listed in the IVRS
3170 * table. The system timer is connected to the SB IOAPIC
3171 * and if we don't have it in the list the system will
3172 * panic at boot time. This situation usually happens
3173 * when the BIOS is buggy and provides us the wrong
3174 * device id for the IOAPIC in the system.
3175 */
3176 pr_err("%s: No southbridge IOAPIC found\n", fw_bug);
3177 ret = false;
3178 }
3179
3180 if (!ret)
3181 pr_err("Disabling interrupt remapping\n");
3182
3183 return ret;
3184 }
3185
free_dma_resources(void)3186 static void __init free_dma_resources(void)
3187 {
3188 amd_iommu_pdom_id_destroy();
3189 free_unity_maps();
3190 }
3191
ivinfo_init(void * ivrs)3192 static void __init ivinfo_init(void *ivrs)
3193 {
3194 amd_iommu_ivinfo = *((u32 *)(ivrs + IOMMU_IVINFO_OFFSET));
3195 }
3196
3197 /*
3198 * This is the hardware init function for AMD IOMMU in the system.
3199 * This function is called either from amd_iommu_init or from the interrupt
3200 * remapping setup code.
3201 *
3202 * This function basically parses the ACPI table for AMD IOMMU (IVRS)
3203 * four times:
3204 *
3205 * 1 pass) Discover the most comprehensive IVHD type to use.
3206 *
3207 * 2 pass) Find the highest PCI device id the driver has to handle.
3208 * Upon this information the size of the data structures is
3209 * determined that needs to be allocated.
3210 *
3211 * 3 pass) Initialize the data structures just allocated with the
3212 * information in the ACPI table about available AMD IOMMUs
3213 * in the system. It also maps the PCI devices in the
3214 * system to specific IOMMUs
3215 *
3216 * 4 pass) After the basic data structures are allocated and
3217 * initialized we update them with information about memory
3218 * remapping requirements parsed out of the ACPI table in
3219 * this last pass.
3220 *
3221 * After everything is set up the IOMMUs are enabled and the necessary
3222 * hotplug and suspend notifiers are registered.
3223 */
early_amd_iommu_init(void)3224 static int __init early_amd_iommu_init(void)
3225 {
3226 struct acpi_table_header *ivrs_base;
3227 int ret;
3228 acpi_status status;
3229 u8 efr_hats, max_vasize;
3230
3231 if (!amd_iommu_detected)
3232 return -ENODEV;
3233
3234 status = acpi_get_table("IVRS", 0, &ivrs_base);
3235 if (status == AE_NOT_FOUND)
3236 return -ENODEV;
3237 else if (ACPI_FAILURE(status)) {
3238 const char *err = acpi_format_exception(status);
3239 pr_err("IVRS table error: %s\n", err);
3240 return -EINVAL;
3241 }
3242
3243 if (!boot_cpu_has(X86_FEATURE_CX16)) {
3244 pr_err("Failed to initialize. The CMPXCHG16B feature is required.\n");
3245 ret = -EINVAL;
3246 goto out;
3247 }
3248
3249 /*
3250 * Validate checksum here so we don't need to do it when
3251 * we actually parse the table
3252 */
3253 ret = check_ivrs_checksum(ivrs_base);
3254 if (ret)
3255 goto out;
3256
3257 ivinfo_init(ivrs_base);
3258
3259 max_vasize = FIELD_GET(IOMMU_IVINFO_VASIZE, amd_iommu_ivinfo);
3260 if (!max_vasize)
3261 max_vasize = 64;
3262
3263 amd_iommu_target_ivhd_type = get_highest_supported_ivhd_type(ivrs_base);
3264 DUMP_printk("Using IVHD type %#x\n", amd_iommu_target_ivhd_type);
3265
3266 /*
3267 * now the data structures are allocated and basically initialized
3268 * start the real acpi table scan
3269 */
3270 ret = init_iommu_all(ivrs_base);
3271 if (ret)
3272 goto out;
3273
3274 /* 5 level guest page table */
3275 if (cpu_feature_enabled(X86_FEATURE_LA57) &&
3276 FIELD_GET(FEATURE_GATS, amd_iommu_efr) == GUEST_PGTABLE_5_LEVEL)
3277 amd_iommu_gpt_level = PAGE_MODE_5_LEVEL;
3278
3279 efr_hats = FIELD_GET(FEATURE_HATS, amd_iommu_efr);
3280 if (efr_hats != 0x3) {
3281 /*
3282 * efr[HATS] bits specify the maximum host translation level
3283 * supported, with LEVEL 4 being initial max level.
3284 */
3285 amd_iommu_hpt_vasize = min_t(unsigned int, max_vasize,
3286 (efr_hats + PAGE_MODE_4_LEVEL - 1) * 9 + 21);
3287 } else {
3288 pr_warn_once(FW_BUG "Disable host address translation due to invalid translation level (%#x).\n",
3289 efr_hats);
3290 amd_iommu_hatdis = true;
3291 }
3292
3293 if (amd_iommu_pgtable == PD_MODE_V2) {
3294 if (!amd_iommu_v2_pgtbl_supported()) {
3295 pr_warn("Cannot enable v2 page table for DMA-API. Fallback to v1.\n");
3296 amd_iommu_pgtable = PD_MODE_V1;
3297 }
3298 }
3299
3300 if (amd_iommu_hatdis) {
3301 /*
3302 * Host (v1) page table is not available. Attempt to use
3303 * Guest (v2) page table.
3304 */
3305 if (amd_iommu_v2_pgtbl_supported())
3306 amd_iommu_pgtable = PD_MODE_V2;
3307 else
3308 amd_iommu_pgtable = PD_MODE_NONE;
3309 }
3310
3311 /* Disable any previously enabled IOMMUs */
3312 if (!is_kdump_kernel() || amd_iommu_disabled)
3313 disable_iommus();
3314
3315 if (amd_iommu_irq_remap)
3316 amd_iommu_irq_remap = check_ioapic_information();
3317
3318 if (amd_iommu_irq_remap) {
3319 struct amd_iommu_pci_seg *pci_seg;
3320 ret = -ENOMEM;
3321 for_each_pci_segment(pci_seg) {
3322 if (alloc_irq_lookup_table(pci_seg))
3323 goto out;
3324 }
3325 }
3326
3327 ret = init_memory_definitions(ivrs_base);
3328 if (ret)
3329 goto out;
3330
3331 /* init the device table */
3332 init_device_table();
3333
3334 out:
3335 /* Don't leak any ACPI memory */
3336 acpi_put_table(ivrs_base);
3337
3338 return ret;
3339 }
3340
amd_iommu_enable_interrupts(void)3341 static int amd_iommu_enable_interrupts(void)
3342 {
3343 struct amd_iommu *iommu;
3344 int ret = 0;
3345
3346 for_each_iommu(iommu) {
3347 ret = iommu_init_irq(iommu);
3348 if (ret)
3349 goto out;
3350 }
3351
3352 /*
3353 * Interrupt handler is ready to process interrupts. Enable
3354 * PPR and GA log interrupt for all IOMMUs.
3355 */
3356 enable_iommus_vapic();
3357 enable_iommus_ppr();
3358
3359 out:
3360 return ret;
3361 }
3362
detect_ivrs(void)3363 static bool __init detect_ivrs(void)
3364 {
3365 struct acpi_table_header *ivrs_base;
3366 acpi_status status;
3367 int i;
3368
3369 status = acpi_get_table("IVRS", 0, &ivrs_base);
3370 if (status == AE_NOT_FOUND)
3371 return false;
3372 else if (ACPI_FAILURE(status)) {
3373 const char *err = acpi_format_exception(status);
3374 pr_err("IVRS table error: %s\n", err);
3375 return false;
3376 }
3377
3378 acpi_put_table(ivrs_base);
3379
3380 if (amd_iommu_force_enable)
3381 goto out;
3382
3383 /* Don't use IOMMU if there is Stoney Ridge graphics */
3384 for (i = 0; i < 32; i++) {
3385 u32 pci_id;
3386
3387 pci_id = read_pci_config(0, i, 0, 0);
3388 if ((pci_id & 0xffff) == 0x1002 && (pci_id >> 16) == 0x98e4) {
3389 pr_info("Disable IOMMU on Stoney Ridge\n");
3390 return false;
3391 }
3392 }
3393
3394 out:
3395 /* Make sure ACS will be enabled during PCI probe */
3396 pci_request_acs();
3397
3398 return true;
3399 }
3400
iommu_snp_enable(void)3401 static __init void iommu_snp_enable(void)
3402 {
3403 #ifdef CONFIG_KVM_AMD_SEV
3404 if (!cc_platform_has(CC_ATTR_HOST_SEV_SNP))
3405 return;
3406
3407 /* SNP support required IOMMU to be ON */
3408 if (no_iommu) {
3409 pr_warn("SNP: IOMMU disabled, SNP cannot be supported.\n");
3410 goto disable_snp;
3411 }
3412
3413 amd_iommu_snp_mode0_sup = check_feature2(FEATURE_SNP_PAGE_MODE0_SUP);
3414 /*
3415 * If SNP page mode 0 is not enabled, then SNP support requires that IOMMU
3416 * must be configured with V1 page table (DTE[Mode] != 0).
3417 */
3418 if (!amd_iommu_snp_mode0_sup) {
3419 if (iommu_default_passthrough()) {
3420 pr_warn("SNP: IOMMU configured in passthrough mode, SNP cannot be supported.\n");
3421 goto disable_snp;
3422 }
3423
3424 if (amd_iommu_pgtable != PD_MODE_V1) {
3425 pr_warn("SNP: IOMMU is configured with V2 page table mode, SNP cannot be supported.\n");
3426 goto disable_snp;
3427 }
3428 }
3429
3430 amd_iommu_snp_en = check_feature(FEATURE_SNP);
3431 if (!amd_iommu_snp_en) {
3432 pr_warn("SNP: IOMMU SNP feature not enabled, SNP cannot be supported.\n");
3433 goto disable_snp;
3434 }
3435
3436 /*
3437 * Enable host SNP support once SNP support is checked on IOMMU.
3438 */
3439 if (snp_rmptable_init()) {
3440 pr_warn("SNP: RMP initialization failed, SNP cannot be supported.\n");
3441 goto disable_snp;
3442 }
3443
3444 pr_info("IOMMU SNP support enabled.\n");
3445 return;
3446
3447 disable_snp:
3448 cc_platform_clear(CC_ATTR_HOST_SEV_SNP);
3449 #endif
3450 }
3451
amd_iommu_apply_erratum_snp(void)3452 static void amd_iommu_apply_erratum_snp(void)
3453 {
3454 #ifdef CONFIG_KVM_AMD_SEV
3455 if (!amd_iommu_snp_en)
3456 return;
3457
3458 /* Errata fix for Family 0x19 */
3459 if (boot_cpu_data.x86 != 0x19)
3460 return;
3461
3462 /* Set event log buffer size to max */
3463 amd_iommu_evtlog_size = EVTLOG_SIZE_MAX;
3464 pr_info("Applying erratum: Increase Event log size to 0x%x\n",
3465 amd_iommu_evtlog_size);
3466
3467 /*
3468 * Set PPR log buffer size to max.
3469 * (Family 0x19, model < 0x10 doesn't support PPR when SNP is enabled).
3470 */
3471 if (boot_cpu_data.x86_model >= 0x10) {
3472 amd_iommu_pprlog_size = PPRLOG_SIZE_MAX;
3473 pr_info("Applying erratum: Increase PPR log size to 0x%x\n",
3474 amd_iommu_pprlog_size);
3475 }
3476 #endif
3477 }
3478
3479 /****************************************************************************
3480 *
3481 * AMD IOMMU Initialization State Machine
3482 *
3483 ****************************************************************************/
3484
state_next(void)3485 static int __init state_next(void)
3486 {
3487 int ret = 0;
3488
3489 switch (init_state) {
3490 case IOMMU_START_STATE:
3491 if (!detect_ivrs()) {
3492 init_state = IOMMU_NOT_FOUND;
3493 ret = -ENODEV;
3494 } else {
3495 init_state = IOMMU_IVRS_DETECTED;
3496 }
3497 break;
3498 case IOMMU_IVRS_DETECTED:
3499 if (amd_iommu_disabled) {
3500 init_state = IOMMU_CMDLINE_DISABLED;
3501 ret = -EINVAL;
3502 } else {
3503 ret = early_amd_iommu_init();
3504 init_state = ret ? IOMMU_INIT_ERROR : IOMMU_ACPI_FINISHED;
3505 }
3506 break;
3507 case IOMMU_ACPI_FINISHED:
3508 early_enable_iommus();
3509 x86_platform.iommu_shutdown = disable_iommus;
3510 init_state = IOMMU_ENABLED;
3511 break;
3512 case IOMMU_ENABLED:
3513 register_syscore(&amd_iommu_syscore);
3514 iommu_snp_enable();
3515
3516 amd_iommu_apply_erratum_snp();
3517
3518 /* Allocate/enable event log buffer */
3519 if (is_kdump_kernel())
3520 ret = remap_event_buffer();
3521 else
3522 ret = alloc_event_buffer();
3523
3524 if (ret) {
3525 init_state = IOMMU_INIT_ERROR;
3526 break;
3527 }
3528 iommu_enable_event_buffer();
3529
3530 ret = amd_iommu_init_pci();
3531 init_state = ret ? IOMMU_INIT_ERROR : IOMMU_PCI_INIT;
3532 break;
3533 case IOMMU_PCI_INIT:
3534 ret = amd_iommu_enable_interrupts();
3535 init_state = ret ? IOMMU_INIT_ERROR : IOMMU_INTERRUPTS_EN;
3536 break;
3537 case IOMMU_INTERRUPTS_EN:
3538 init_state = IOMMU_INITIALIZED;
3539 break;
3540 case IOMMU_INITIALIZED:
3541 /* Nothing to do */
3542 break;
3543 case IOMMU_NOT_FOUND:
3544 case IOMMU_INIT_ERROR:
3545 case IOMMU_CMDLINE_DISABLED:
3546 /* Error states => do nothing */
3547 ret = -EINVAL;
3548 break;
3549 default:
3550 /* Unknown state */
3551 BUG();
3552 }
3553
3554 if (ret) {
3555 free_dma_resources();
3556 if (!irq_remapping_enabled) {
3557 disable_iommus();
3558 free_iommu_resources();
3559 } else {
3560 struct amd_iommu *iommu;
3561 struct amd_iommu_pci_seg *pci_seg;
3562
3563 for_each_pci_segment(pci_seg)
3564 uninit_device_table_dma(pci_seg);
3565
3566 for_each_iommu(iommu)
3567 amd_iommu_flush_all_caches(iommu);
3568 }
3569 }
3570 return ret;
3571 }
3572
iommu_go_to_state(enum iommu_init_state state)3573 static int __init iommu_go_to_state(enum iommu_init_state state)
3574 {
3575 int ret = -EINVAL;
3576
3577 while (init_state != state) {
3578 if (init_state == IOMMU_NOT_FOUND ||
3579 init_state == IOMMU_INIT_ERROR ||
3580 init_state == IOMMU_CMDLINE_DISABLED)
3581 break;
3582 ret = state_next();
3583 }
3584
3585 /*
3586 * SNP platform initilazation requires IOMMUs to be fully configured.
3587 * If the SNP support on IOMMUs has NOT been checked, simply mark SNP
3588 * as unsupported. If the SNP support on IOMMUs has been checked and
3589 * host SNP support enabled but RMP enforcement has not been enabled
3590 * in IOMMUs, then the system is in a half-baked state, but can limp
3591 * along as all memory should be Hypervisor-Owned in the RMP. WARN,
3592 * but leave SNP as "supported" to avoid confusing the kernel.
3593 */
3594 if (ret && cc_platform_has(CC_ATTR_HOST_SEV_SNP) &&
3595 !WARN_ON_ONCE(amd_iommu_snp_en))
3596 cc_platform_clear(CC_ATTR_HOST_SEV_SNP);
3597
3598 return ret;
3599 }
3600
3601 #ifdef CONFIG_IRQ_REMAP
amd_iommu_prepare(void)3602 int __init amd_iommu_prepare(void)
3603 {
3604 int ret;
3605
3606 amd_iommu_irq_remap = true;
3607
3608 ret = iommu_go_to_state(IOMMU_ACPI_FINISHED);
3609 if (ret) {
3610 amd_iommu_irq_remap = false;
3611 return ret;
3612 }
3613
3614 return amd_iommu_irq_remap ? 0 : -ENODEV;
3615 }
3616
amd_iommu_enable(void)3617 int __init amd_iommu_enable(void)
3618 {
3619 int ret;
3620
3621 ret = iommu_go_to_state(IOMMU_ENABLED);
3622 if (ret)
3623 return ret;
3624
3625 irq_remapping_enabled = 1;
3626 return amd_iommu_xt_mode;
3627 }
3628
amd_iommu_disable(void)3629 void amd_iommu_disable(void)
3630 {
3631 amd_iommu_suspend(NULL);
3632 }
3633
amd_iommu_reenable(int mode)3634 int amd_iommu_reenable(int mode)
3635 {
3636 amd_iommu_resume(NULL);
3637
3638 return 0;
3639 }
3640
amd_iommu_enable_faulting(unsigned int cpu)3641 int amd_iommu_enable_faulting(unsigned int cpu)
3642 {
3643 /* We enable MSI later when PCI is initialized */
3644 return 0;
3645 }
3646 #endif
3647
3648 /*
3649 * This is the core init function for AMD IOMMU hardware in the system.
3650 * This function is called from the generic x86 DMA layer initialization
3651 * code.
3652 */
amd_iommu_init(void)3653 static int __init amd_iommu_init(void)
3654 {
3655 int ret;
3656
3657 ret = iommu_go_to_state(IOMMU_INITIALIZED);
3658 #ifdef CONFIG_GART_IOMMU
3659 if (ret && list_empty(&amd_iommu_list)) {
3660 /*
3661 * We failed to initialize the AMD IOMMU - try fallback
3662 * to GART if possible.
3663 */
3664 gart_iommu_init();
3665 }
3666 #endif
3667
3668 if (!ret)
3669 amd_iommu_debugfs_setup();
3670
3671 return ret;
3672 }
3673
amd_iommu_sme_check(void)3674 static bool amd_iommu_sme_check(void)
3675 {
3676 if (!cc_platform_has(CC_ATTR_HOST_MEM_ENCRYPT) ||
3677 (boot_cpu_data.x86 != 0x17))
3678 return true;
3679
3680 /* For Fam17h, a specific level of support is required */
3681 if (boot_cpu_data.microcode >= 0x08001205)
3682 return true;
3683
3684 if ((boot_cpu_data.microcode >= 0x08001126) &&
3685 (boot_cpu_data.microcode <= 0x080011ff))
3686 return true;
3687
3688 pr_notice("IOMMU not currently supported when SME is active\n");
3689
3690 return false;
3691 }
3692
3693 /****************************************************************************
3694 *
3695 * Early detect code. This code runs at IOMMU detection time in the DMA
3696 * layer. It just looks if there is an IVRS ACPI table to detect AMD
3697 * IOMMUs
3698 *
3699 ****************************************************************************/
amd_iommu_detect(void)3700 void __init amd_iommu_detect(void)
3701 {
3702 int ret;
3703
3704 if (no_iommu || (iommu_detected && !gart_iommu_aperture))
3705 goto disable_snp;
3706
3707 if (!amd_iommu_sme_check())
3708 goto disable_snp;
3709
3710 ret = iommu_go_to_state(IOMMU_IVRS_DETECTED);
3711 if (ret)
3712 goto disable_snp;
3713
3714 amd_iommu_detected = true;
3715 iommu_detected = 1;
3716 x86_init.iommu.iommu_init = amd_iommu_init;
3717 return;
3718
3719 disable_snp:
3720 if (cc_platform_has(CC_ATTR_HOST_SEV_SNP))
3721 cc_platform_clear(CC_ATTR_HOST_SEV_SNP);
3722 }
3723
3724 /****************************************************************************
3725 *
3726 * Parsing functions for the AMD IOMMU specific kernel command line
3727 * options.
3728 *
3729 ****************************************************************************/
3730
parse_amd_iommu_dump(char * str)3731 static int __init parse_amd_iommu_dump(char *str)
3732 {
3733 amd_iommu_dump = true;
3734
3735 return 1;
3736 }
3737
parse_amd_iommu_intr(char * str)3738 static int __init parse_amd_iommu_intr(char *str)
3739 {
3740 for (; *str; ++str) {
3741 if (strncmp(str, "legacy", 6) == 0) {
3742 amd_iommu_guest_ir = AMD_IOMMU_GUEST_IR_LEGACY_GA;
3743 break;
3744 }
3745 if (strncmp(str, "vapic", 5) == 0) {
3746 amd_iommu_guest_ir = AMD_IOMMU_GUEST_IR_VAPIC;
3747 break;
3748 }
3749 }
3750 return 1;
3751 }
3752
parse_amd_iommu_options(char * str)3753 static int __init parse_amd_iommu_options(char *str)
3754 {
3755 if (!str)
3756 return -EINVAL;
3757
3758 while (*str) {
3759 if (strncmp(str, "fullflush", 9) == 0) {
3760 pr_warn("amd_iommu=fullflush deprecated; use iommu.strict=1 instead\n");
3761 iommu_set_dma_strict();
3762 } else if (strncmp(str, "force_enable", 12) == 0) {
3763 amd_iommu_force_enable = true;
3764 } else if (strncmp(str, "off", 3) == 0) {
3765 amd_iommu_disabled = true;
3766 } else if (strncmp(str, "force_isolation", 15) == 0) {
3767 amd_iommu_force_isolation = true;
3768 } else if (strncmp(str, "pgtbl_v1", 8) == 0) {
3769 amd_iommu_pgtable = PD_MODE_V1;
3770 } else if (strncmp(str, "pgtbl_v2", 8) == 0) {
3771 amd_iommu_pgtable = PD_MODE_V2;
3772 } else if (strncmp(str, "irtcachedis", 11) == 0) {
3773 amd_iommu_irtcachedis = true;
3774 } else if (strncmp(str, "nohugepages", 11) == 0) {
3775 pr_info("Restricting V1 page-sizes to 4KiB");
3776 amd_iommu_pgsize_bitmap = AMD_IOMMU_PGSIZES_4K;
3777 } else if (strncmp(str, "v2_pgsizes_only", 15) == 0) {
3778 pr_info("Restricting V1 page-sizes to 4KiB/2MiB/1GiB");
3779 amd_iommu_pgsize_bitmap = AMD_IOMMU_PGSIZES_V2;
3780 } else {
3781 pr_notice("Unknown option - '%s'\n", str);
3782 }
3783
3784 str += strcspn(str, ",");
3785 while (*str == ',')
3786 str++;
3787 }
3788
3789 return 1;
3790 }
3791
parse_ivrs_ioapic(char * str)3792 static int __init parse_ivrs_ioapic(char *str)
3793 {
3794 u32 seg = 0, bus, dev, fn;
3795 int id, i;
3796 u32 devid;
3797
3798 if (sscanf(str, "=%d@%x:%x.%x", &id, &bus, &dev, &fn) == 4 ||
3799 sscanf(str, "=%d@%x:%x:%x.%x", &id, &seg, &bus, &dev, &fn) == 5)
3800 goto found;
3801
3802 if (sscanf(str, "[%d]=%x:%x.%x", &id, &bus, &dev, &fn) == 4 ||
3803 sscanf(str, "[%d]=%x:%x:%x.%x", &id, &seg, &bus, &dev, &fn) == 5) {
3804 pr_warn("ivrs_ioapic%s option format deprecated; use ivrs_ioapic=%d@%04x:%02x:%02x.%d instead\n",
3805 str, id, seg, bus, dev, fn);
3806 goto found;
3807 }
3808
3809 pr_err("Invalid command line: ivrs_ioapic%s\n", str);
3810 return 1;
3811
3812 found:
3813 if (early_ioapic_map_size == EARLY_MAP_SIZE) {
3814 pr_err("Early IOAPIC map overflow - ignoring ivrs_ioapic%s\n",
3815 str);
3816 return 1;
3817 }
3818
3819 devid = IVRS_GET_SBDF_ID(seg, bus, dev, fn);
3820
3821 cmdline_maps = true;
3822 i = early_ioapic_map_size++;
3823 early_ioapic_map[i].id = id;
3824 early_ioapic_map[i].devid = devid;
3825 early_ioapic_map[i].cmd_line = true;
3826
3827 return 1;
3828 }
3829
parse_ivrs_hpet(char * str)3830 static int __init parse_ivrs_hpet(char *str)
3831 {
3832 u32 seg = 0, bus, dev, fn;
3833 int id, i;
3834 u32 devid;
3835
3836 if (sscanf(str, "=%d@%x:%x.%x", &id, &bus, &dev, &fn) == 4 ||
3837 sscanf(str, "=%d@%x:%x:%x.%x", &id, &seg, &bus, &dev, &fn) == 5)
3838 goto found;
3839
3840 if (sscanf(str, "[%d]=%x:%x.%x", &id, &bus, &dev, &fn) == 4 ||
3841 sscanf(str, "[%d]=%x:%x:%x.%x", &id, &seg, &bus, &dev, &fn) == 5) {
3842 pr_warn("ivrs_hpet%s option format deprecated; use ivrs_hpet=%d@%04x:%02x:%02x.%d instead\n",
3843 str, id, seg, bus, dev, fn);
3844 goto found;
3845 }
3846
3847 pr_err("Invalid command line: ivrs_hpet%s\n", str);
3848 return 1;
3849
3850 found:
3851 if (early_hpet_map_size == EARLY_MAP_SIZE) {
3852 pr_err("Early HPET map overflow - ignoring ivrs_hpet%s\n",
3853 str);
3854 return 1;
3855 }
3856
3857 devid = IVRS_GET_SBDF_ID(seg, bus, dev, fn);
3858
3859 cmdline_maps = true;
3860 i = early_hpet_map_size++;
3861 early_hpet_map[i].id = id;
3862 early_hpet_map[i].devid = devid;
3863 early_hpet_map[i].cmd_line = true;
3864
3865 return 1;
3866 }
3867
3868 #define ACPIID_LEN (ACPIHID_UID_LEN + ACPIHID_HID_LEN)
3869
parse_ivrs_acpihid(char * str)3870 static int __init parse_ivrs_acpihid(char *str)
3871 {
3872 u32 seg = 0, bus, dev, fn;
3873 char *hid, *uid, *p, *addr;
3874 char acpiid[ACPIID_LEN + 1] = { }; /* size with NULL terminator */
3875 int i;
3876
3877 addr = strchr(str, '@');
3878 if (!addr) {
3879 addr = strchr(str, '=');
3880 if (!addr)
3881 goto not_found;
3882
3883 ++addr;
3884
3885 if (strlen(addr) > ACPIID_LEN)
3886 goto not_found;
3887
3888 if (sscanf(str, "[%x:%x.%x]=%s", &bus, &dev, &fn, acpiid) == 4 ||
3889 sscanf(str, "[%x:%x:%x.%x]=%s", &seg, &bus, &dev, &fn, acpiid) == 5) {
3890 pr_warn("ivrs_acpihid%s option format deprecated; use ivrs_acpihid=%s@%04x:%02x:%02x.%d instead\n",
3891 str, acpiid, seg, bus, dev, fn);
3892 goto found;
3893 }
3894 goto not_found;
3895 }
3896
3897 /* We have the '@', make it the terminator to get just the acpiid */
3898 *addr++ = 0;
3899
3900 if (strlen(str) > ACPIID_LEN)
3901 goto not_found;
3902
3903 if (sscanf(str, "=%s", acpiid) != 1)
3904 goto not_found;
3905
3906 if (sscanf(addr, "%x:%x.%x", &bus, &dev, &fn) == 3 ||
3907 sscanf(addr, "%x:%x:%x.%x", &seg, &bus, &dev, &fn) == 4)
3908 goto found;
3909
3910 not_found:
3911 pr_err("Invalid command line: ivrs_acpihid%s\n", str);
3912 return 1;
3913
3914 found:
3915 if (early_acpihid_map_size == EARLY_MAP_SIZE) {
3916 pr_err("Early ACPI HID map overflow - ignoring ivrs_acpihid%s\n",
3917 str);
3918 return 1;
3919 }
3920
3921 p = acpiid;
3922 hid = strsep(&p, ":");
3923 uid = p;
3924
3925 if (!hid || !(*hid) || !uid) {
3926 pr_err("Invalid command line: hid or uid\n");
3927 return 1;
3928 }
3929
3930 /*
3931 * Ignore leading zeroes after ':', so e.g., AMDI0095:00
3932 * will match AMDI0095:0 in the second strcmp in acpi_dev_hid_uid_match
3933 */
3934 while (*uid == '0' && *(uid + 1))
3935 uid++;
3936
3937 if (strlen(hid) >= ACPIHID_HID_LEN) {
3938 pr_err("Invalid command line: hid is too long\n");
3939 return 1;
3940 } else if (strlen(uid) >= ACPIHID_UID_LEN) {
3941 pr_err("Invalid command line: uid is too long\n");
3942 return 1;
3943 }
3944
3945 i = early_acpihid_map_size++;
3946 memcpy(early_acpihid_map[i].hid, hid, strlen(hid));
3947 memcpy(early_acpihid_map[i].uid, uid, strlen(uid));
3948 early_acpihid_map[i].devid = IVRS_GET_SBDF_ID(seg, bus, dev, fn);
3949 early_acpihid_map[i].cmd_line = true;
3950
3951 return 1;
3952 }
3953
3954 __setup("amd_iommu_dump", parse_amd_iommu_dump);
3955 __setup("amd_iommu=", parse_amd_iommu_options);
3956 __setup("amd_iommu_intr=", parse_amd_iommu_intr);
3957 __setup("ivrs_ioapic", parse_ivrs_ioapic);
3958 __setup("ivrs_hpet", parse_ivrs_hpet);
3959 __setup("ivrs_acpihid", parse_ivrs_acpihid);
3960
amd_iommu_pasid_supported(void)3961 bool amd_iommu_pasid_supported(void)
3962 {
3963 /* CPU page table size should match IOMMU guest page table size */
3964 if (cpu_feature_enabled(X86_FEATURE_LA57) &&
3965 amd_iommu_gpt_level != PAGE_MODE_5_LEVEL)
3966 return false;
3967
3968 if (!amd_iommu_gt_ppr_supported())
3969 return false;
3970
3971 /*
3972 * If SNP page mode 0 is not supported, then DTE[Mode]=0 is prohibited
3973 * on SNP-enabled system (i.e. EFR[SNPSup]=1). IOMMUv2 page table
3974 * cannot be used without setting up IOMMUv1 page table.
3975 */
3976 return (!amd_iommu_snp_en) || (amd_iommu_snp_en && amd_iommu_snp_mode0_sup);
3977 }
3978
get_amd_iommu(unsigned int idx)3979 struct amd_iommu *get_amd_iommu(unsigned int idx)
3980 {
3981 unsigned int i = 0;
3982 struct amd_iommu *iommu;
3983
3984 for_each_iommu(iommu)
3985 if (i++ == idx)
3986 return iommu;
3987 return NULL;
3988 }
3989
3990 /****************************************************************************
3991 *
3992 * IOMMU EFR Performance Counter support functionality. This code allows
3993 * access to the IOMMU PC functionality.
3994 *
3995 ****************************************************************************/
3996
amd_iommu_pc_get_max_banks(unsigned int idx)3997 u8 amd_iommu_pc_get_max_banks(unsigned int idx)
3998 {
3999 struct amd_iommu *iommu = get_amd_iommu(idx);
4000
4001 if (iommu)
4002 return iommu->max_banks;
4003
4004 return 0;
4005 }
4006
amd_iommu_pc_supported(void)4007 bool amd_iommu_pc_supported(void)
4008 {
4009 return amd_iommu_pc_present;
4010 }
4011
amd_iommu_pc_get_max_counters(unsigned int idx)4012 u8 amd_iommu_pc_get_max_counters(unsigned int idx)
4013 {
4014 struct amd_iommu *iommu = get_amd_iommu(idx);
4015
4016 if (iommu)
4017 return iommu->max_counters;
4018
4019 return 0;
4020 }
4021
iommu_pc_get_set_reg(struct amd_iommu * iommu,u8 bank,u8 cntr,u8 fxn,u64 * value,bool is_write)4022 static int iommu_pc_get_set_reg(struct amd_iommu *iommu, u8 bank, u8 cntr,
4023 u8 fxn, u64 *value, bool is_write)
4024 {
4025 u32 offset;
4026 u32 max_offset_lim;
4027
4028 /* Make sure the IOMMU PC resource is available */
4029 if (!amd_iommu_pc_present)
4030 return -ENODEV;
4031
4032 /* Check for valid iommu and pc register indexing */
4033 if (WARN_ON(!iommu || (fxn > 0x28) || (fxn & 7)))
4034 return -ENODEV;
4035
4036 offset = (u32)(((0x40 | bank) << 12) | (cntr << 8) | fxn);
4037
4038 /* Limit the offset to the hw defined mmio region aperture */
4039 max_offset_lim = (u32)(((0x40 | iommu->max_banks) << 12) |
4040 (iommu->max_counters << 8) | 0x28);
4041 if ((offset < MMIO_CNTR_REG_OFFSET) ||
4042 (offset > max_offset_lim))
4043 return -EINVAL;
4044
4045 if (is_write) {
4046 u64 val = *value & GENMASK_ULL(47, 0);
4047
4048 writel((u32)val, iommu->mmio_base + offset);
4049 writel((val >> 32), iommu->mmio_base + offset + 4);
4050 } else {
4051 *value = readl(iommu->mmio_base + offset + 4);
4052 *value <<= 32;
4053 *value |= readl(iommu->mmio_base + offset);
4054 *value &= GENMASK_ULL(47, 0);
4055 }
4056
4057 return 0;
4058 }
4059
amd_iommu_pc_get_reg(struct amd_iommu * iommu,u8 bank,u8 cntr,u8 fxn,u64 * value)4060 int amd_iommu_pc_get_reg(struct amd_iommu *iommu, u8 bank, u8 cntr, u8 fxn, u64 *value)
4061 {
4062 if (!iommu)
4063 return -EINVAL;
4064
4065 return iommu_pc_get_set_reg(iommu, bank, cntr, fxn, value, false);
4066 }
4067
amd_iommu_pc_set_reg(struct amd_iommu * iommu,u8 bank,u8 cntr,u8 fxn,u64 * value)4068 int amd_iommu_pc_set_reg(struct amd_iommu *iommu, u8 bank, u8 cntr, u8 fxn, u64 *value)
4069 {
4070 if (!iommu)
4071 return -EINVAL;
4072
4073 return iommu_pc_get_set_reg(iommu, bank, cntr, fxn, value, true);
4074 }
4075
4076 #ifdef CONFIG_KVM_AMD_SEV
iommu_page_make_shared(void * page)4077 static int iommu_page_make_shared(void *page)
4078 {
4079 unsigned long paddr, pfn;
4080
4081 paddr = iommu_virt_to_phys(page);
4082 /* Cbit maybe set in the paddr */
4083 pfn = __sme_clr(paddr) >> PAGE_SHIFT;
4084
4085 if (!(pfn % PTRS_PER_PMD)) {
4086 int ret, level;
4087 bool assigned;
4088
4089 ret = snp_lookup_rmpentry(pfn, &assigned, &level);
4090 if (ret) {
4091 pr_warn("IOMMU PFN %lx RMP lookup failed, ret %d\n", pfn, ret);
4092 return ret;
4093 }
4094
4095 if (!assigned) {
4096 pr_warn("IOMMU PFN %lx not assigned in RMP table\n", pfn);
4097 return -EINVAL;
4098 }
4099
4100 if (level > PG_LEVEL_4K) {
4101 ret = psmash(pfn);
4102 if (!ret)
4103 goto done;
4104
4105 pr_warn("PSMASH failed for IOMMU PFN %lx huge RMP entry, ret: %d, level: %d\n",
4106 pfn, ret, level);
4107 return ret;
4108 }
4109 }
4110
4111 done:
4112 return rmp_make_shared(pfn, PG_LEVEL_4K);
4113 }
4114
iommu_make_shared(void * va,size_t size)4115 static int iommu_make_shared(void *va, size_t size)
4116 {
4117 void *page;
4118 int ret;
4119
4120 if (!va)
4121 return 0;
4122
4123 for (page = va; page < (va + size); page += PAGE_SIZE) {
4124 ret = iommu_page_make_shared(page);
4125 if (ret)
4126 return ret;
4127 }
4128
4129 return 0;
4130 }
4131
amd_iommu_snp_disable(void)4132 int amd_iommu_snp_disable(void)
4133 {
4134 struct amd_iommu *iommu;
4135 int ret;
4136
4137 if (!amd_iommu_snp_en)
4138 return 0;
4139
4140 for_each_iommu(iommu) {
4141 ret = iommu_make_shared(iommu->evt_buf, amd_iommu_evtlog_size);
4142 if (ret)
4143 return ret;
4144
4145 ret = iommu_make_shared(iommu->ppr_log, amd_iommu_pprlog_size);
4146 if (ret)
4147 return ret;
4148
4149 ret = iommu_make_shared((void *)iommu->cmd_sem, PAGE_SIZE);
4150 if (ret)
4151 return ret;
4152 }
4153
4154 return 0;
4155 }
4156 EXPORT_SYMBOL_GPL(amd_iommu_snp_disable);
4157
amd_iommu_sev_tio_supported(void)4158 bool amd_iommu_sev_tio_supported(void)
4159 {
4160 return check_feature2(FEATURE_SEVSNPIO_SUP);
4161 }
4162 EXPORT_SYMBOL_GPL(amd_iommu_sev_tio_supported);
4163 #endif
4164