1 // SPDX-License-Identifier: GPL-2.0-only 2 /* 3 * Dynamic DMA mapping support for AMD Hammer. 4 * 5 * Use the integrated AGP GART in the Hammer northbridge as an IOMMU for PCI. 6 * This allows to use PCI devices that only support 32bit addresses on systems 7 * with more than 4GB. 8 * 9 * See Documentation/core-api/dma-api-howto.rst for the interface specification. 10 * 11 * Copyright 2002 Andi Kleen, SuSE Labs. 12 */ 13 14 #include <linux/types.h> 15 #include <linux/ctype.h> 16 #include <linux/agp_backend.h> 17 #include <linux/init.h> 18 #include <linux/mm.h> 19 #include <linux/sched.h> 20 #include <linux/sched/debug.h> 21 #include <linux/string.h> 22 #include <linux/spinlock.h> 23 #include <linux/pci.h> 24 #include <linux/topology.h> 25 #include <linux/interrupt.h> 26 #include <linux/bitmap.h> 27 #include <linux/kdebug.h> 28 #include <linux/scatterlist.h> 29 #include <linux/iommu-helper.h> 30 #include <linux/syscore_ops.h> 31 #include <linux/io.h> 32 #include <linux/gfp.h> 33 #include <linux/atomic.h> 34 #include <linux/dma-direct.h> 35 #include <linux/dma-map-ops.h> 36 #include <asm/mtrr.h> 37 #include <asm/proto.h> 38 #include <asm/iommu.h> 39 #include <asm/gart.h> 40 #include <asm/set_memory.h> 41 #include <asm/dma.h> 42 #include <asm/amd/nb.h> 43 #include <asm/x86_init.h> 44 45 static unsigned long iommu_bus_base; /* GART remapping area (physical) */ 46 static unsigned long iommu_size; /* size of remapping area bytes */ 47 static unsigned long iommu_pages; /* .. and in pages */ 48 49 static u32 *iommu_gatt_base; /* Remapping table */ 50 51 /* 52 * If this is disabled the IOMMU will use an optimized flushing strategy 53 * of only flushing when an mapping is reused. With it true the GART is 54 * flushed for every mapping. Problem is that doing the lazy flush seems 55 * to trigger bugs with some popular PCI cards, in particular 3ware (but 56 * has been also seen with Qlogic at least). 57 */ 58 static int iommu_fullflush = 1; 59 60 /* Allocation bitmap for the remapping area: */ 61 static DEFINE_SPINLOCK(iommu_bitmap_lock); 62 /* Guarded by iommu_bitmap_lock: */ 63 static unsigned long *iommu_gart_bitmap; 64 65 static u32 gart_unmapped_entry; 66 67 #define GPTE_VALID 1 68 #define GPTE_COHERENT 2 69 #define GPTE_ENCODE(x) \ 70 (((x) & 0xfffff000) | (((x) >> 32) << 4) | GPTE_VALID | GPTE_COHERENT) 71 #define GPTE_DECODE(x) (((x) & 0xfffff000) | (((u64)(x) & 0xff0) << 28)) 72 73 #ifdef CONFIG_AGP 74 #define AGPEXTERN extern 75 #else 76 #define AGPEXTERN 77 #endif 78 79 /* GART can only remap to physical addresses < 1TB */ 80 #define GART_MAX_PHYS_ADDR (1ULL << 40) 81 82 /* backdoor interface to AGP driver */ 83 AGPEXTERN int agp_memory_reserved; 84 AGPEXTERN __u32 *agp_gatt_table; 85 86 static unsigned long next_bit; /* protected by iommu_bitmap_lock */ 87 static bool need_flush; /* global flush state. set for each gart wrap */ 88 89 static unsigned long alloc_iommu(struct device *dev, int size, 90 unsigned long align_mask) 91 { 92 unsigned long offset, flags; 93 unsigned long boundary_size; 94 unsigned long base_index; 95 96 base_index = ALIGN(iommu_bus_base & dma_get_seg_boundary(dev), 97 PAGE_SIZE) >> PAGE_SHIFT; 98 boundary_size = dma_get_seg_boundary_nr_pages(dev, PAGE_SHIFT); 99 100 spin_lock_irqsave(&iommu_bitmap_lock, flags); 101 offset = iommu_area_alloc(iommu_gart_bitmap, iommu_pages, next_bit, 102 size, base_index, boundary_size, align_mask); 103 if (offset == -1) { 104 need_flush = true; 105 offset = iommu_area_alloc(iommu_gart_bitmap, iommu_pages, 0, 106 size, base_index, boundary_size, 107 align_mask); 108 } 109 if (offset != -1) { 110 next_bit = offset+size; 111 if (next_bit >= iommu_pages) { 112 next_bit = 0; 113 need_flush = true; 114 } 115 } 116 if (iommu_fullflush) 117 need_flush = true; 118 spin_unlock_irqrestore(&iommu_bitmap_lock, flags); 119 120 return offset; 121 } 122 123 static void free_iommu(unsigned long offset, int size) 124 { 125 unsigned long flags; 126 127 spin_lock_irqsave(&iommu_bitmap_lock, flags); 128 bitmap_clear(iommu_gart_bitmap, offset, size); 129 if (offset >= next_bit) 130 next_bit = offset + size; 131 spin_unlock_irqrestore(&iommu_bitmap_lock, flags); 132 } 133 134 /* 135 * Use global flush state to avoid races with multiple flushers. 136 */ 137 static void flush_gart(void) 138 { 139 unsigned long flags; 140 141 spin_lock_irqsave(&iommu_bitmap_lock, flags); 142 if (need_flush) { 143 amd_flush_garts(); 144 need_flush = false; 145 } 146 spin_unlock_irqrestore(&iommu_bitmap_lock, flags); 147 } 148 149 #ifdef CONFIG_IOMMU_LEAK 150 /* Debugging aid for drivers that don't free their IOMMU tables */ 151 static void dump_leak(void) 152 { 153 static int dump; 154 155 if (dump) 156 return; 157 dump = 1; 158 159 show_stack(NULL, NULL, KERN_ERR); 160 debug_dma_dump_mappings(NULL); 161 } 162 #endif 163 164 static void iommu_full(struct device *dev, size_t size, int dir) 165 { 166 /* 167 * Ran out of IOMMU space for this operation. This is very bad. 168 * Unfortunately the drivers cannot handle this operation properly. 169 * Return some non mapped prereserved space in the aperture and 170 * let the Northbridge deal with it. This will result in garbage 171 * in the IO operation. When the size exceeds the prereserved space 172 * memory corruption will occur or random memory will be DMAed 173 * out. Hopefully no network devices use single mappings that big. 174 */ 175 176 dev_err(dev, "PCI-DMA: Out of IOMMU space for %lu bytes\n", size); 177 #ifdef CONFIG_IOMMU_LEAK 178 dump_leak(); 179 #endif 180 } 181 182 static inline int 183 need_iommu(struct device *dev, unsigned long addr, size_t size, unsigned long attrs) 184 { 185 return force_iommu || !dma_capable(dev, addr, size, true, attrs); 186 } 187 188 static inline int 189 nonforced_iommu(struct device *dev, unsigned long addr, size_t size, 190 unsigned long attrs) 191 { 192 return !dma_capable(dev, addr, size, true, attrs); 193 } 194 195 /* Map a single continuous physical area into the IOMMU. 196 * Caller needs to check if the iommu is needed and flush. 197 */ 198 static dma_addr_t dma_map_area(struct device *dev, dma_addr_t phys_mem, 199 size_t size, int dir, unsigned long align_mask, unsigned long attrs) 200 { 201 unsigned long npages = iommu_num_pages(phys_mem, size, PAGE_SIZE); 202 unsigned long iommu_page; 203 int i; 204 205 if (unlikely(phys_mem + size > GART_MAX_PHYS_ADDR)) 206 return DMA_MAPPING_ERROR; 207 208 iommu_page = alloc_iommu(dev, npages, align_mask); 209 if (iommu_page == -1) { 210 if (!nonforced_iommu(dev, phys_mem, size, attrs)) 211 return phys_mem; 212 if (panic_on_overflow) 213 panic("dma_map_area overflow %lu bytes\n", size); 214 iommu_full(dev, size, dir); 215 return DMA_MAPPING_ERROR; 216 } 217 218 for (i = 0; i < npages; i++) { 219 iommu_gatt_base[iommu_page + i] = GPTE_ENCODE(phys_mem); 220 phys_mem += PAGE_SIZE; 221 } 222 return iommu_bus_base + iommu_page*PAGE_SIZE + (phys_mem & ~PAGE_MASK); 223 } 224 225 /* Map a single area into the IOMMU */ 226 static dma_addr_t gart_map_phys(struct device *dev, phys_addr_t paddr, 227 size_t size, enum dma_data_direction dir, 228 unsigned long attrs) 229 { 230 unsigned long bus; 231 232 if (unlikely(attrs & DMA_ATTR_MMIO)) 233 return DMA_MAPPING_ERROR; 234 235 if (!need_iommu(dev, paddr, size, attrs)) 236 return paddr; 237 238 bus = dma_map_area(dev, paddr, size, dir, 0, attrs); 239 flush_gart(); 240 241 return bus; 242 } 243 244 /* 245 * Free a DMA mapping. 246 */ 247 static void gart_unmap_phys(struct device *dev, dma_addr_t dma_addr, 248 size_t size, enum dma_data_direction dir, 249 unsigned long attrs) 250 { 251 unsigned long iommu_page; 252 int npages; 253 int i; 254 255 if (WARN_ON_ONCE(dma_addr == DMA_MAPPING_ERROR)) 256 return; 257 258 /* 259 * This driver will not always use a GART mapping, but might have 260 * created a direct mapping instead. If that is the case there is 261 * nothing to unmap here. 262 */ 263 if (dma_addr < iommu_bus_base || 264 dma_addr >= iommu_bus_base + iommu_size) 265 return; 266 267 iommu_page = (dma_addr - iommu_bus_base)>>PAGE_SHIFT; 268 npages = iommu_num_pages(dma_addr, size, PAGE_SIZE); 269 for (i = 0; i < npages; i++) { 270 iommu_gatt_base[iommu_page + i] = gart_unmapped_entry; 271 } 272 free_iommu(iommu_page, npages); 273 } 274 275 /* 276 * Wrapper for pci_unmap_single working with scatterlists. 277 */ 278 static void gart_unmap_sg(struct device *dev, struct scatterlist *sg, int nents, 279 enum dma_data_direction dir, unsigned long attrs) 280 { 281 struct scatterlist *s; 282 int i; 283 284 for_each_sg(sg, s, nents, i) { 285 if (!s->dma_length || !s->length) 286 break; 287 gart_unmap_phys(dev, s->dma_address, s->dma_length, dir, 0); 288 } 289 } 290 291 /* Fallback for dma_map_sg in case of overflow */ 292 static int dma_map_sg_nonforce(struct device *dev, struct scatterlist *sg, 293 int nents, int dir, unsigned long attrs) 294 { 295 struct scatterlist *s; 296 int i; 297 298 #ifdef CONFIG_IOMMU_DEBUG 299 pr_debug("dma_map_sg overflow\n"); 300 #endif 301 302 for_each_sg(sg, s, nents, i) { 303 unsigned long addr = sg_phys(s); 304 305 if (nonforced_iommu(dev, addr, s->length, attrs)) { 306 addr = dma_map_area(dev, addr, s->length, dir, 0, attrs); 307 if (addr == DMA_MAPPING_ERROR) { 308 if (i > 0) 309 gart_unmap_sg(dev, sg, i, dir, 0); 310 nents = 0; 311 sg[0].dma_length = 0; 312 break; 313 } 314 } 315 s->dma_address = addr; 316 s->dma_length = s->length; 317 } 318 flush_gart(); 319 320 return nents; 321 } 322 323 /* Map multiple scatterlist entries continuous into the first. */ 324 static int __dma_map_cont(struct device *dev, struct scatterlist *start, 325 int nelems, struct scatterlist *sout, 326 unsigned long pages) 327 { 328 unsigned long iommu_start = alloc_iommu(dev, pages, 0); 329 unsigned long iommu_page = iommu_start; 330 struct scatterlist *s; 331 int i; 332 333 if (iommu_start == -1) 334 return -ENOMEM; 335 336 for_each_sg(start, s, nelems, i) { 337 unsigned long pages, addr; 338 unsigned long phys_addr = s->dma_address; 339 340 BUG_ON(s != start && s->offset); 341 if (s == start) { 342 sout->dma_address = iommu_bus_base; 343 sout->dma_address += iommu_page*PAGE_SIZE + s->offset; 344 sout->dma_length = s->length; 345 } else { 346 sout->dma_length += s->length; 347 } 348 349 addr = phys_addr; 350 pages = iommu_num_pages(s->offset, s->length, PAGE_SIZE); 351 while (pages--) { 352 iommu_gatt_base[iommu_page] = GPTE_ENCODE(addr); 353 addr += PAGE_SIZE; 354 iommu_page++; 355 } 356 } 357 BUG_ON(iommu_page - iommu_start != pages); 358 359 return 0; 360 } 361 362 static inline int 363 dma_map_cont(struct device *dev, struct scatterlist *start, int nelems, 364 struct scatterlist *sout, unsigned long pages, int need) 365 { 366 if (!need) { 367 BUG_ON(nelems != 1); 368 sout->dma_address = start->dma_address; 369 sout->dma_length = start->length; 370 return 0; 371 } 372 return __dma_map_cont(dev, start, nelems, sout, pages); 373 } 374 375 /* 376 * DMA map all entries in a scatterlist. 377 * Merge chunks that have page aligned sizes into a continuous mapping. 378 */ 379 static int gart_map_sg(struct device *dev, struct scatterlist *sg, int nents, 380 enum dma_data_direction dir, unsigned long attrs) 381 { 382 struct scatterlist *s, *ps, *start_sg, *sgmap; 383 int need = 0, nextneed, i, out, start, ret; 384 unsigned long pages = 0; 385 unsigned int seg_size; 386 unsigned int max_seg_size; 387 388 if (nents == 0) 389 return -EINVAL; 390 391 out = 0; 392 start = 0; 393 start_sg = sg; 394 sgmap = sg; 395 seg_size = 0; 396 max_seg_size = dma_get_max_seg_size(dev); 397 ps = NULL; /* shut up gcc */ 398 399 for_each_sg(sg, s, nents, i) { 400 dma_addr_t addr = sg_phys(s); 401 402 s->dma_address = addr; 403 BUG_ON(s->length == 0); 404 405 nextneed = need_iommu(dev, addr, s->length, attrs); 406 407 /* Handle the previous not yet processed entries */ 408 if (i > start) { 409 /* 410 * Can only merge when the last chunk ends on a 411 * page boundary and the new one doesn't have an 412 * offset. 413 */ 414 if (!iommu_merge || !nextneed || !need || s->offset || 415 (s->length + seg_size > max_seg_size) || 416 (ps->offset + ps->length) % PAGE_SIZE) { 417 ret = dma_map_cont(dev, start_sg, i - start, 418 sgmap, pages, need); 419 if (ret < 0) 420 goto error; 421 out++; 422 423 seg_size = 0; 424 sgmap = sg_next(sgmap); 425 pages = 0; 426 start = i; 427 start_sg = s; 428 } 429 } 430 431 seg_size += s->length; 432 need = nextneed; 433 pages += iommu_num_pages(s->offset, s->length, PAGE_SIZE); 434 ps = s; 435 } 436 ret = dma_map_cont(dev, start_sg, i - start, sgmap, pages, need); 437 if (ret < 0) 438 goto error; 439 out++; 440 flush_gart(); 441 if (out < nents) { 442 sgmap = sg_next(sgmap); 443 sgmap->dma_length = 0; 444 } 445 return out; 446 447 error: 448 flush_gart(); 449 gart_unmap_sg(dev, sg, out, dir, 0); 450 451 /* When it was forced or merged try again in a dumb way */ 452 if (force_iommu || iommu_merge) { 453 out = dma_map_sg_nonforce(dev, sg, nents, dir, attrs); 454 if (out > 0) 455 return out; 456 } 457 if (panic_on_overflow) 458 panic("dma_map_sg: overflow on %lu pages\n", pages); 459 460 iommu_full(dev, pages << PAGE_SHIFT, dir); 461 return ret; 462 } 463 464 /* allocate and map a coherent mapping */ 465 static void * 466 gart_alloc_coherent(struct device *dev, size_t size, dma_addr_t *dma_addr, 467 gfp_t flag, unsigned long attrs) 468 { 469 void *vaddr; 470 471 vaddr = dma_direct_alloc(dev, size, dma_addr, flag, attrs); 472 if (!vaddr || 473 !force_iommu || dev->coherent_dma_mask <= DMA_BIT_MASK(24)) 474 return vaddr; 475 476 *dma_addr = dma_map_area(dev, virt_to_phys(vaddr), size, 477 DMA_BIDIRECTIONAL, 478 (1UL << get_order(size)) - 1, attrs); 479 flush_gart(); 480 if (unlikely(*dma_addr == DMA_MAPPING_ERROR)) 481 goto out_free; 482 return vaddr; 483 out_free: 484 dma_direct_free(dev, size, vaddr, *dma_addr, attrs); 485 return NULL; 486 } 487 488 /* free a coherent mapping */ 489 static void 490 gart_free_coherent(struct device *dev, size_t size, void *vaddr, 491 dma_addr_t dma_addr, unsigned long attrs) 492 { 493 gart_unmap_phys(dev, dma_addr, size, DMA_BIDIRECTIONAL, 0); 494 dma_direct_free(dev, size, vaddr, dma_addr, attrs); 495 } 496 497 static int no_agp; 498 499 static __init unsigned long check_iommu_size(unsigned long aper, u64 aper_size) 500 { 501 unsigned long a; 502 503 if (!iommu_size) { 504 iommu_size = aper_size; 505 if (!no_agp) 506 iommu_size /= 2; 507 } 508 509 a = aper + iommu_size; 510 iommu_size -= round_up(a, PMD_SIZE) - a; 511 512 if (iommu_size < 64*1024*1024) { 513 pr_warn("PCI-DMA: Warning: Small IOMMU %luMB." 514 " Consider increasing the AGP aperture in BIOS\n", 515 iommu_size >> 20); 516 } 517 518 return iommu_size; 519 } 520 521 static __init unsigned read_aperture(struct pci_dev *dev, u32 *size) 522 { 523 unsigned aper_size = 0, aper_base_32, aper_order; 524 u64 aper_base; 525 526 pci_read_config_dword(dev, AMD64_GARTAPERTUREBASE, &aper_base_32); 527 pci_read_config_dword(dev, AMD64_GARTAPERTURECTL, &aper_order); 528 aper_order = (aper_order >> 1) & 7; 529 530 aper_base = aper_base_32 & 0x7fff; 531 aper_base <<= 25; 532 533 aper_size = (32 * 1024 * 1024) << aper_order; 534 if (aper_base + aper_size > 0x100000000UL || !aper_size) 535 aper_base = 0; 536 537 *size = aper_size; 538 return aper_base; 539 } 540 541 static void enable_gart_translations(void) 542 { 543 int i; 544 545 if (!amd_nb_has_feature(AMD_NB_GART)) 546 return; 547 548 for (i = 0; i < amd_nb_num(); i++) { 549 struct pci_dev *dev = node_to_amd_nb(i)->misc; 550 551 enable_gart_translation(dev, __pa(agp_gatt_table)); 552 } 553 554 /* Flush the GART-TLB to remove stale entries */ 555 amd_flush_garts(); 556 } 557 558 /* 559 * If fix_up_north_bridges is set, the north bridges have to be fixed up on 560 * resume in the same way as they are handled in gart_iommu_hole_init(). 561 */ 562 static bool fix_up_north_bridges; 563 static u32 aperture_order; 564 static u32 aperture_alloc; 565 566 void set_up_gart_resume(u32 aper_order, u32 aper_alloc) 567 { 568 fix_up_north_bridges = true; 569 aperture_order = aper_order; 570 aperture_alloc = aper_alloc; 571 } 572 573 static void gart_fixup_northbridges(void) 574 { 575 int i; 576 577 if (!fix_up_north_bridges) 578 return; 579 580 if (!amd_nb_has_feature(AMD_NB_GART)) 581 return; 582 583 pr_info("PCI-DMA: Restoring GART aperture settings\n"); 584 585 for (i = 0; i < amd_nb_num(); i++) { 586 struct pci_dev *dev = node_to_amd_nb(i)->misc; 587 588 /* 589 * Don't enable translations just yet. That is the next 590 * step. Restore the pre-suspend aperture settings. 591 */ 592 gart_set_size_and_enable(dev, aperture_order); 593 pci_write_config_dword(dev, AMD64_GARTAPERTUREBASE, aperture_alloc >> 25); 594 } 595 } 596 597 static void gart_resume(void *data) 598 { 599 pr_info("PCI-DMA: Resuming GART IOMMU\n"); 600 601 gart_fixup_northbridges(); 602 603 enable_gart_translations(); 604 } 605 606 static const struct syscore_ops gart_syscore_ops = { 607 .resume = gart_resume, 608 609 }; 610 611 static struct syscore gart_syscore = { 612 .ops = &gart_syscore_ops, 613 }; 614 615 /* 616 * Private Northbridge GATT initialization in case we cannot use the 617 * AGP driver for some reason. 618 */ 619 static __init int init_amd_gatt(struct agp_kern_info *info) 620 { 621 unsigned aper_size, gatt_size, new_aper_size; 622 unsigned aper_base, new_aper_base; 623 struct pci_dev *dev; 624 void *gatt; 625 int i; 626 627 pr_info("PCI-DMA: Disabling AGP.\n"); 628 629 aper_size = aper_base = info->aper_size = 0; 630 dev = NULL; 631 for (i = 0; i < amd_nb_num(); i++) { 632 dev = node_to_amd_nb(i)->misc; 633 new_aper_base = read_aperture(dev, &new_aper_size); 634 if (!new_aper_base) 635 goto nommu; 636 637 if (!aper_base) { 638 aper_size = new_aper_size; 639 aper_base = new_aper_base; 640 } 641 if (aper_size != new_aper_size || aper_base != new_aper_base) 642 goto nommu; 643 } 644 if (!aper_base) 645 goto nommu; 646 647 info->aper_base = aper_base; 648 info->aper_size = aper_size >> 20; 649 650 gatt_size = (aper_size >> PAGE_SHIFT) * sizeof(u32); 651 gatt = (void *)__get_free_pages(GFP_KERNEL | __GFP_ZERO, 652 get_order(gatt_size)); 653 if (!gatt) 654 panic("Cannot allocate GATT table"); 655 if (set_memory_uc((unsigned long)gatt, gatt_size >> PAGE_SHIFT)) 656 panic("Could not set GART PTEs to uncacheable pages"); 657 658 agp_gatt_table = gatt; 659 660 register_syscore(&gart_syscore); 661 662 flush_gart(); 663 664 pr_info("PCI-DMA: aperture base @ %x size %u KB\n", 665 aper_base, aper_size>>10); 666 667 return 0; 668 669 nommu: 670 /* Should not happen anymore */ 671 pr_warn("PCI-DMA: More than 4GB of RAM and no IOMMU - falling back to iommu=soft.\n"); 672 return -1; 673 } 674 675 static const struct dma_map_ops gart_dma_ops = { 676 .map_sg = gart_map_sg, 677 .unmap_sg = gart_unmap_sg, 678 .map_phys = gart_map_phys, 679 .unmap_phys = gart_unmap_phys, 680 .alloc = gart_alloc_coherent, 681 .free = gart_free_coherent, 682 .mmap = dma_common_mmap, 683 .get_sgtable = dma_common_get_sgtable, 684 .dma_supported = dma_direct_supported, 685 .get_required_mask = dma_direct_get_required_mask, 686 .alloc_pages_op = dma_direct_alloc_pages, 687 .free_pages = dma_direct_free_pages, 688 }; 689 690 static void gart_iommu_shutdown(void) 691 { 692 struct pci_dev *dev; 693 int i; 694 695 /* don't shutdown it if there is AGP installed */ 696 if (!no_agp) 697 return; 698 699 if (!amd_nb_has_feature(AMD_NB_GART)) 700 return; 701 702 for (i = 0; i < amd_nb_num(); i++) { 703 u32 ctl; 704 705 dev = node_to_amd_nb(i)->misc; 706 pci_read_config_dword(dev, AMD64_GARTAPERTURECTL, &ctl); 707 708 ctl &= ~GARTEN; 709 710 pci_write_config_dword(dev, AMD64_GARTAPERTURECTL, ctl); 711 } 712 } 713 714 int __init gart_iommu_init(void) 715 { 716 struct agp_kern_info info; 717 unsigned long iommu_start; 718 unsigned long aper_base, aper_size; 719 unsigned long start_pfn, end_pfn; 720 unsigned long scratch; 721 722 if (!amd_nb_has_feature(AMD_NB_GART)) 723 return 0; 724 725 #ifndef CONFIG_AGP_AMD64 726 no_agp = 1; 727 #else 728 /* Makefile puts PCI initialization via subsys_initcall first. */ 729 /* Add other AMD AGP bridge drivers here */ 730 no_agp = no_agp || 731 (agp_amd64_init() < 0) || 732 (agp_copy_info(agp_bridge, &info) < 0); 733 #endif 734 735 if (no_iommu || 736 (!force_iommu && max_pfn <= MAX_DMA32_PFN) || 737 !gart_iommu_aperture || 738 (no_agp && init_amd_gatt(&info) < 0)) { 739 if (max_pfn > MAX_DMA32_PFN) { 740 pr_warn("More than 4GB of memory but GART IOMMU not available.\n"); 741 pr_warn("falling back to iommu=soft.\n"); 742 } 743 return 0; 744 } 745 746 /* need to map that range */ 747 aper_size = info.aper_size << 20; 748 aper_base = info.aper_base; 749 end_pfn = (aper_base>>PAGE_SHIFT) + (aper_size>>PAGE_SHIFT); 750 751 start_pfn = PFN_DOWN(aper_base); 752 if (!pfn_range_is_mapped(start_pfn, end_pfn)) 753 init_memory_mapping(start_pfn<<PAGE_SHIFT, end_pfn<<PAGE_SHIFT, 754 PAGE_KERNEL); 755 756 pr_info("PCI-DMA: using GART IOMMU.\n"); 757 iommu_size = check_iommu_size(info.aper_base, aper_size); 758 iommu_pages = iommu_size >> PAGE_SHIFT; 759 760 iommu_gart_bitmap = (void *) __get_free_pages(GFP_KERNEL | __GFP_ZERO, 761 get_order(iommu_pages/8)); 762 if (!iommu_gart_bitmap) 763 panic("Cannot allocate iommu bitmap\n"); 764 765 pr_info("PCI-DMA: Reserving %luMB of IOMMU area in the AGP aperture\n", 766 iommu_size >> 20); 767 768 agp_memory_reserved = iommu_size; 769 iommu_start = aper_size - iommu_size; 770 iommu_bus_base = info.aper_base + iommu_start; 771 iommu_gatt_base = agp_gatt_table + (iommu_start>>PAGE_SHIFT); 772 773 /* 774 * Unmap the IOMMU part of the GART. The alias of the page is 775 * always mapped with cache enabled and there is no full cache 776 * coherency across the GART remapping. The unmapping avoids 777 * automatic prefetches from the CPU allocating cache lines in 778 * there. All CPU accesses are done via the direct mapping to 779 * the backing memory. The GART address is only used by PCI 780 * devices. 781 */ 782 set_memory_np((unsigned long)__va(iommu_bus_base), 783 iommu_size >> PAGE_SHIFT); 784 /* 785 * Tricky. The GART table remaps the physical memory range, 786 * so the CPU won't notice potential aliases and if the memory 787 * is remapped to UC later on, we might surprise the PCI devices 788 * with a stray writeout of a cacheline. So play it sure and 789 * do an explicit, full-scale wbinvd() _after_ having marked all 790 * the pages as Not-Present: 791 */ 792 wbinvd(); 793 794 /* 795 * Now all caches are flushed and we can safely enable 796 * GART hardware. Doing it early leaves the possibility 797 * of stale cache entries that can lead to GART PTE 798 * errors. 799 */ 800 enable_gart_translations(); 801 802 /* 803 * Try to workaround a bug (thanks to BenH): 804 * Set unmapped entries to a scratch page instead of 0. 805 * Any prefetches that hit unmapped entries won't get an bus abort 806 * then. (P2P bridge may be prefetching on DMA reads). 807 */ 808 scratch = get_zeroed_page(GFP_KERNEL); 809 if (!scratch) 810 panic("Cannot allocate iommu scratch page"); 811 gart_unmapped_entry = GPTE_ENCODE(__pa(scratch)); 812 813 flush_gart(); 814 dma_ops = &gart_dma_ops; 815 x86_platform.iommu_shutdown = gart_iommu_shutdown; 816 x86_swiotlb_enable = false; 817 818 return 0; 819 } 820 821 void __init gart_parse_options(char *p) 822 { 823 int arg; 824 825 if (isdigit(*p) && get_option(&p, &arg)) 826 iommu_size = arg; 827 if (!strncmp(p, "fullflush", 9)) 828 iommu_fullflush = 1; 829 if (!strncmp(p, "nofullflush", 11)) 830 iommu_fullflush = 0; 831 if (!strncmp(p, "noagp", 5)) 832 no_agp = 1; 833 if (!strncmp(p, "noaperture", 10)) 834 fix_aperture = 0; 835 /* duplicated from pci-dma.c */ 836 if (!strncmp(p, "force", 5)) 837 gart_iommu_aperture_allowed = 1; 838 if (!strncmp(p, "allowed", 7)) 839 gart_iommu_aperture_allowed = 1; 840 if (!strncmp(p, "memaper", 7)) { 841 fallback_aper_force = 1; 842 p += 7; 843 if (*p == '=') { 844 ++p; 845 if (get_option(&p, &arg)) 846 fallback_aper_order = arg; 847 } 848 } 849 } 850