1 // SPDX-License-Identifier: GPL-2.0 2 /* 3 * Slab allocator functions that are independent of the allocator strategy 4 * 5 * (C) 2012 Christoph Lameter <cl@gentwo.org> 6 */ 7 #include <linux/slab.h> 8 9 #include <linux/mm.h> 10 #include <linux/poison.h> 11 #include <linux/interrupt.h> 12 #include <linux/memory.h> 13 #include <linux/cache.h> 14 #include <linux/compiler.h> 15 #include <linux/kfence.h> 16 #include <linux/module.h> 17 #include <linux/cpu.h> 18 #include <linux/uaccess.h> 19 #include <linux/seq_file.h> 20 #include <linux/dma-mapping.h> 21 #include <linux/swiotlb.h> 22 #include <linux/proc_fs.h> 23 #include <linux/debugfs.h> 24 #include <linux/kmemleak.h> 25 #include <linux/kasan.h> 26 #include <asm/cacheflush.h> 27 #include <asm/tlbflush.h> 28 #include <asm/page.h> 29 #include <linux/memcontrol.h> 30 #include <linux/stackdepot.h> 31 #include <trace/events/rcu.h> 32 33 #include "../kernel/rcu/rcu.h" 34 #include "internal.h" 35 #include "slab.h" 36 37 #define CREATE_TRACE_POINTS 38 #include <trace/events/kmem.h> 39 40 enum slab_state slab_state; 41 LIST_HEAD(slab_caches); 42 DEFINE_MUTEX(slab_mutex); 43 struct kmem_cache *kmem_cache; 44 45 /* 46 * Set of flags that will prevent slab merging. 47 * Any flag that adds per-object metadata should be included, 48 * since slab merging can update s->inuse that affects the metadata layout. 49 */ 50 #define SLAB_NEVER_MERGE (SLAB_DEBUG_FLAGS | SLAB_TYPESAFE_BY_RCU | \ 51 SLAB_NOLEAKTRACE | SLAB_FAILSLAB | SLAB_NO_MERGE | \ 52 SLAB_OBJ_EXT_IN_OBJ) 53 54 #define SLAB_MERGE_SAME (SLAB_RECLAIM_ACCOUNT | SLAB_CACHE_DMA | \ 55 SLAB_CACHE_DMA32 | SLAB_ACCOUNT | SLAB_MAY_ACCOUNT) 56 57 /* 58 * Merge control. If this is set then no merging of slab caches will occur. 59 */ 60 static bool slab_nomerge = !IS_ENABLED(CONFIG_SLAB_MERGE_DEFAULT); 61 62 static int __init setup_slab_nomerge(char *str) 63 { 64 slab_nomerge = true; 65 return 1; 66 } 67 68 static int __init setup_slab_merge(char *str) 69 { 70 slab_nomerge = false; 71 return 1; 72 } 73 74 __setup_param("slub_nomerge", slub_nomerge, setup_slab_nomerge, 0); 75 __setup_param("slub_merge", slub_merge, setup_slab_merge, 0); 76 77 __setup("slab_nomerge", setup_slab_nomerge); 78 __setup("slab_merge", setup_slab_merge); 79 80 /* 81 * Determine the size of a slab object 82 */ 83 unsigned int kmem_cache_size(struct kmem_cache *s) 84 { 85 return s->object_size; 86 } 87 EXPORT_SYMBOL(kmem_cache_size); 88 89 #ifdef CONFIG_DEBUG_VM 90 91 static bool kmem_cache_is_duplicate_name(const char *name) 92 { 93 struct kmem_cache *s; 94 95 list_for_each_entry(s, &slab_caches, list) { 96 if (!strcmp(s->name, name)) 97 return true; 98 } 99 100 return false; 101 } 102 103 static int kmem_cache_sanity_check(const char *name, unsigned int size) 104 { 105 if (!name || in_interrupt() || size > KMALLOC_MAX_SIZE) { 106 pr_err("kmem_cache_create(%s) integrity check failed\n", name); 107 return -EINVAL; 108 } 109 110 /* Duplicate names will confuse slabtop, et al */ 111 WARN(kmem_cache_is_duplicate_name(name), 112 "kmem_cache of name '%s' already exists\n", name); 113 114 WARN_ON(strchr(name, ' ')); /* It confuses parsers */ 115 return 0; 116 } 117 #else 118 static inline int kmem_cache_sanity_check(const char *name, unsigned int size) 119 { 120 return 0; 121 } 122 #endif 123 124 /* 125 * Figure out what the alignment of the objects will be given a set of 126 * flags, a user specified alignment and the size of the objects. 127 */ 128 static unsigned int calculate_alignment(slab_flags_t flags, 129 unsigned int align, unsigned int size) 130 { 131 /* 132 * If the user wants hardware cache aligned objects then follow that 133 * suggestion if the object is sufficiently large. 134 * 135 * The hardware cache alignment cannot override the specified 136 * alignment though. If that is greater then use it. 137 */ 138 if (flags & SLAB_HWCACHE_ALIGN) { 139 unsigned int ralign; 140 141 ralign = cache_line_size(); 142 while (size <= ralign / 2) 143 ralign /= 2; 144 align = max(align, ralign); 145 } 146 147 align = max(align, arch_slab_minalign()); 148 149 return ALIGN(align, sizeof(void *)); 150 } 151 152 /* 153 * Find a mergeable slab cache 154 */ 155 int slab_unmergeable(struct kmem_cache *s) 156 { 157 if (slab_nomerge || (s->flags & SLAB_NEVER_MERGE)) 158 return 1; 159 160 if (s->ctor) 161 return 1; 162 163 #ifdef CONFIG_HARDENED_USERCOPY 164 if (s->usersize) 165 return 1; 166 #endif 167 168 /* 169 * We may have set a slab to be unmergeable during bootstrap. 170 */ 171 if (s->refcount < 0) 172 return 1; 173 174 return 0; 175 } 176 177 bool slab_args_unmergeable(struct kmem_cache_args *args, slab_flags_t flags) 178 { 179 if (slab_nomerge) 180 return true; 181 182 if (args->ctor) 183 return true; 184 185 if (IS_ENABLED(CONFIG_HARDENED_USERCOPY) && args->usersize) 186 return true; 187 188 if (flags & SLAB_NEVER_MERGE) 189 return true; 190 191 return false; 192 } 193 194 static struct kmem_cache *find_mergeable(unsigned int size, slab_flags_t flags, 195 const char *name, struct kmem_cache_args *args) 196 { 197 struct kmem_cache *s; 198 unsigned int align; 199 200 flags = kmem_cache_flags(flags, name); 201 if (slab_args_unmergeable(args, flags)) 202 return NULL; 203 204 size = ALIGN(size, sizeof(void *)); 205 align = calculate_alignment(flags, args->align, size); 206 size = ALIGN(size, align); 207 208 list_for_each_entry_reverse(s, &slab_caches, list) { 209 if (slab_unmergeable(s)) 210 continue; 211 212 if (size > s->size) 213 continue; 214 215 if ((flags & SLAB_MERGE_SAME) != (s->flags & SLAB_MERGE_SAME)) 216 continue; 217 /* 218 * Check if alignment is compatible. 219 * Courtesy of Adrian Drzewiecki 220 */ 221 if ((s->size & ~(align - 1)) != s->size) 222 continue; 223 224 if (s->size - size >= sizeof(void *)) 225 continue; 226 227 return s; 228 } 229 return NULL; 230 } 231 232 static struct kmem_cache *create_cache(const char *name, 233 unsigned int object_size, 234 struct kmem_cache_args *args, 235 slab_flags_t flags) 236 { 237 struct kmem_cache *s; 238 int err; 239 240 /* If a custom freelist pointer is requested make sure it's sane. */ 241 err = -EINVAL; 242 if (args->use_freeptr_offset && 243 (args->freeptr_offset >= object_size || 244 (!(flags & SLAB_TYPESAFE_BY_RCU) && !args->ctor) || 245 !IS_ALIGNED(args->freeptr_offset, __alignof__(freeptr_t)))) 246 goto out; 247 248 err = -ENOMEM; 249 s = kmem_cache_zalloc(kmem_cache, GFP_KERNEL); 250 if (!s) 251 goto out; 252 err = do_kmem_cache_create(s, name, object_size, args, flags); 253 if (err) 254 goto out_free_cache; 255 256 s->refcount = 1; 257 list_add(&s->list, &slab_caches); 258 return s; 259 260 out_free_cache: 261 kmem_cache_free(kmem_cache, s); 262 out: 263 return ERR_PTR(err); 264 } 265 266 static struct kmem_cache * 267 __kmem_cache_alias(const char *name, unsigned int size, slab_flags_t flags, 268 struct kmem_cache_args *args) 269 { 270 struct kmem_cache *s; 271 272 s = find_mergeable(size, flags, name, args); 273 if (s) { 274 if (sysfs_slab_alias(s, name)) 275 pr_err("SLUB: Unable to add cache alias %s to sysfs\n", 276 name); 277 278 s->refcount++; 279 280 /* 281 * Adjust the object sizes so that we clear 282 * the complete object on kzalloc. 283 */ 284 s->object_size = max(s->object_size, size); 285 s->inuse = max(s->inuse, ALIGN(size, sizeof(void *))); 286 } 287 288 return s; 289 } 290 291 /** 292 * __kmem_cache_create_args - Create a kmem cache. 293 * @name: A string which is used in /proc/slabinfo to identify this cache. 294 * @object_size: The size of objects to be created in this cache. 295 * @args: Additional arguments for the cache creation (see 296 * &struct kmem_cache_args). 297 * @flags: See the descriptions of individual flags. The common ones are listed 298 * in the description below. 299 * 300 * Not to be called directly, use the kmem_cache_create() wrapper with the same 301 * parameters. 302 * 303 * Commonly used @flags: 304 * 305 * &SLAB_ACCOUNT - Account allocations to memcg. 306 * 307 * &SLAB_HWCACHE_ALIGN - Align objects on cache line boundaries. 308 * 309 * &SLAB_RECLAIM_ACCOUNT - Objects are reclaimable. 310 * 311 * &SLAB_TYPESAFE_BY_RCU - Slab page (not individual objects) freeing delayed 312 * by a grace period - see the full description before using. 313 * 314 * Context: Cannot be called within a interrupt, but can be interrupted. 315 * 316 * Return: a pointer to the cache on success, NULL on failure. 317 */ 318 struct kmem_cache *__kmem_cache_create_args(const char *name, 319 unsigned int object_size, 320 struct kmem_cache_args *args, 321 slab_flags_t flags) 322 { 323 struct kmem_cache *s = NULL; 324 const char *cache_name; 325 int err; 326 327 #ifdef CONFIG_SLUB_DEBUG 328 /* 329 * If no slab_debug was enabled globally, the static key is not yet 330 * enabled by setup_slub_debug(). Enable it if the cache is being 331 * created with any of the debugging flags passed explicitly. 332 * It's also possible that this is the first cache created with 333 * SLAB_STORE_USER and we should init stack_depot for it. 334 */ 335 if (flags & SLAB_DEBUG_FLAGS) 336 static_branch_enable(&slub_debug_enabled); 337 if (flags & SLAB_STORE_USER) 338 stack_depot_init(); 339 #else 340 flags &= ~SLAB_DEBUG_FLAGS; 341 #endif 342 343 /* 344 * Caches with specific capacity are special enough. It's simpler to 345 * make them unmergeable. 346 */ 347 if (args->sheaf_capacity) 348 flags |= SLAB_NO_MERGE; 349 350 mutex_lock(&slab_mutex); 351 352 err = kmem_cache_sanity_check(name, object_size); 353 if (err) { 354 goto out_unlock; 355 } 356 357 if (flags & ~SLAB_FLAGS_PERMITTED) { 358 err = -EINVAL; 359 goto out_unlock; 360 } 361 362 /* 363 * For now we assume any cache can be used with __GFP_ACCOUNT and thus 364 * may need to store objcg pointers for objects 365 */ 366 if (!mem_cgroup_kmem_disabled()) 367 flags |= SLAB_MAY_ACCOUNT; 368 369 /* Fail closed on bad usersize of useroffset values. */ 370 if (!IS_ENABLED(CONFIG_HARDENED_USERCOPY) || 371 WARN_ON(!args->usersize && args->useroffset) || 372 WARN_ON(object_size < args->usersize || 373 object_size - args->usersize < args->useroffset)) 374 args->usersize = args->useroffset = 0; 375 376 s = __kmem_cache_alias(name, object_size, flags, args); 377 if (s) 378 goto out_unlock; 379 380 cache_name = kstrdup_const(name, GFP_KERNEL); 381 if (!cache_name) { 382 err = -ENOMEM; 383 goto out_unlock; 384 } 385 386 args->align = calculate_alignment(flags, args->align, object_size); 387 s = create_cache(cache_name, object_size, args, flags); 388 if (IS_ERR(s)) { 389 err = PTR_ERR(s); 390 kfree_const(cache_name); 391 } 392 393 out_unlock: 394 mutex_unlock(&slab_mutex); 395 396 if (err) { 397 if (flags & SLAB_PANIC) 398 panic("%s: Failed to create slab '%s'. Error %d\n", 399 __func__, name, err); 400 else { 401 pr_warn("%s(%s) failed with error %d\n", 402 __func__, name, err); 403 dump_stack(); 404 } 405 return NULL; 406 } 407 return s; 408 } 409 EXPORT_SYMBOL(__kmem_cache_create_args); 410 411 static struct kmem_cache *kmem_buckets_cache __ro_after_init; 412 413 /** 414 * kmem_buckets_create - Create a set of caches that handle dynamic sized 415 * allocations via kmem_buckets_alloc() 416 * @name: A prefix string which is used in /proc/slabinfo to identify this 417 * cache. The individual caches with have their sizes as the suffix. 418 * @flags: SLAB flags (see kmem_cache_create() for details). 419 * @useroffset: Starting offset within an allocation that may be copied 420 * to/from userspace. 421 * @usersize: How many bytes, starting at @useroffset, may be copied 422 * to/from userspace. 423 * @ctor: A constructor for the objects, run when new allocations are made. 424 * 425 * Cannot be called within an interrupt, but can be interrupted. 426 * 427 * Return: a pointer to the cache on success, NULL on failure. When 428 * CONFIG_SLAB_BUCKETS is not enabled, ZERO_SIZE_PTR is returned, and 429 * subsequent calls to kmem_buckets_alloc() will fall back to kmalloc(). 430 * (i.e. callers only need to check for NULL on failure.) 431 */ 432 kmem_buckets *kmem_buckets_create(const char *name, slab_flags_t flags, 433 unsigned int useroffset, 434 unsigned int usersize, 435 void (*ctor)(void *)) 436 { 437 unsigned long mask = 0; 438 unsigned int idx; 439 kmem_buckets *b; 440 441 BUILD_BUG_ON(ARRAY_SIZE(kmalloc_caches[KMALLOC_NORMAL]) > BITS_PER_LONG); 442 443 /* 444 * When the separate buckets API is not built in, just return 445 * a non-NULL value for the kmem_buckets pointer, which will be 446 * unused when performing allocations. 447 */ 448 if (!IS_ENABLED(CONFIG_SLAB_BUCKETS)) 449 return ZERO_SIZE_PTR; 450 451 if (WARN_ON(!kmem_buckets_cache)) 452 return NULL; 453 454 b = kmem_cache_alloc(kmem_buckets_cache, GFP_KERNEL|__GFP_ZERO); 455 if (WARN_ON(!b)) 456 return NULL; 457 458 flags |= SLAB_NO_MERGE; 459 460 for (idx = 0; idx < ARRAY_SIZE(kmalloc_caches[KMALLOC_NORMAL]); idx++) { 461 char *short_size, *cache_name; 462 unsigned int cache_useroffset, cache_usersize; 463 unsigned int size, aligned_idx; 464 465 if (!kmalloc_caches[KMALLOC_NORMAL][idx]) 466 continue; 467 468 size = kmalloc_caches[KMALLOC_NORMAL][idx]->object_size; 469 if (!size) 470 continue; 471 472 short_size = strchr(kmalloc_caches[KMALLOC_NORMAL][idx]->name, '-'); 473 if (WARN_ON(!short_size)) 474 goto fail; 475 476 if (useroffset >= size) { 477 cache_useroffset = 0; 478 cache_usersize = 0; 479 } else { 480 cache_useroffset = useroffset; 481 cache_usersize = min(size - cache_useroffset, usersize); 482 } 483 484 aligned_idx = __kmalloc_index(size, false); 485 if (!(*b)[aligned_idx]) { 486 cache_name = kasprintf(GFP_KERNEL, "%s-%s", name, short_size + 1); 487 if (WARN_ON(!cache_name)) 488 goto fail; 489 (*b)[aligned_idx] = kmem_cache_create_usercopy(cache_name, size, 490 0, flags, cache_useroffset, 491 cache_usersize, ctor); 492 kfree(cache_name); 493 if (WARN_ON(!(*b)[aligned_idx])) 494 goto fail; 495 set_bit(aligned_idx, &mask); 496 } 497 if (idx != aligned_idx) 498 (*b)[idx] = (*b)[aligned_idx]; 499 } 500 501 return b; 502 503 fail: 504 for_each_set_bit(idx, &mask, ARRAY_SIZE(kmalloc_caches[KMALLOC_NORMAL])) 505 kmem_cache_destroy((*b)[idx]); 506 kmem_cache_free(kmem_buckets_cache, b); 507 508 return NULL; 509 } 510 EXPORT_SYMBOL(kmem_buckets_create); 511 512 /* 513 * For a given kmem_cache, kmem_cache_destroy() should only be called 514 * once or there will be a use-after-free problem. The actual deletion 515 * and release of the kobject does not need slab_mutex or cpu_hotplug_lock 516 * protection. So they are now done without holding those locks. 517 */ 518 static void kmem_cache_release(struct kmem_cache *s) 519 { 520 kfence_shutdown_cache(s); 521 if (__is_defined(SLAB_SUPPORTS_SYSFS) && slab_state >= FULL) 522 sysfs_slab_release(s); 523 else 524 slab_kmem_cache_release(s); 525 } 526 527 void slab_kmem_cache_release(struct kmem_cache *s) 528 { 529 __kmem_cache_release(s); 530 kfree_const(s->name); 531 kmem_cache_free(kmem_cache, s); 532 } 533 534 void kmem_cache_destroy(struct kmem_cache *s) 535 { 536 int err; 537 538 if (unlikely(!s) || !kasan_check_byte(s)) 539 return; 540 541 /* in-flight kfree_rcu()'s may include objects from our cache */ 542 kvfree_rcu_barrier_on_cache(s); 543 544 if (IS_ENABLED(CONFIG_SLUB_RCU_DEBUG) && 545 (s->flags & SLAB_TYPESAFE_BY_RCU)) { 546 /* 547 * Under CONFIG_SLUB_RCU_DEBUG, when objects in a 548 * SLAB_TYPESAFE_BY_RCU slab are freed, SLUB will internally 549 * defer their freeing with call_rcu(). 550 * Wait for such call_rcu() invocations here before actually 551 * destroying the cache. 552 * 553 * It doesn't matter that we haven't looked at the slab refcount 554 * yet - slabs with SLAB_TYPESAFE_BY_RCU can't be merged, so 555 * the refcount should be 1 here. 556 */ 557 rcu_barrier(); 558 } 559 560 /* Wait for deferred work from kmalloc/kfree_nolock() */ 561 deferred_work_barrier(); 562 563 cpus_read_lock(); 564 mutex_lock(&slab_mutex); 565 566 s->refcount--; 567 if (s->refcount) { 568 mutex_unlock(&slab_mutex); 569 cpus_read_unlock(); 570 return; 571 } 572 573 /* free asan quarantined objects */ 574 kasan_cache_shutdown(s); 575 576 err = __kmem_cache_shutdown(s); 577 if (!slab_in_kunit_test()) 578 WARN(err, "%s %s: Slab cache still has objects when called from %pS", 579 __func__, s->name, (void *)_RET_IP_); 580 581 list_del(&s->list); 582 583 mutex_unlock(&slab_mutex); 584 cpus_read_unlock(); 585 586 if (slab_state >= FULL) 587 sysfs_slab_unlink(s); 588 debugfs_slab_release(s); 589 590 if (err) 591 return; 592 593 if (s->flags & SLAB_TYPESAFE_BY_RCU) 594 rcu_barrier(); 595 596 kmem_cache_release(s); 597 } 598 EXPORT_SYMBOL(kmem_cache_destroy); 599 600 /** 601 * kmem_cache_shrink - Shrink a cache. 602 * @cachep: The cache to shrink. 603 * 604 * Releases as many slabs as possible for a cache. 605 * To help debugging, a zero exit status indicates all slabs were released. 606 * 607 * Return: %0 if all slabs were released, non-zero otherwise 608 */ 609 int kmem_cache_shrink(struct kmem_cache *cachep) 610 { 611 kasan_cache_shrink(cachep); 612 613 return __kmem_cache_shrink(cachep); 614 } 615 EXPORT_SYMBOL(kmem_cache_shrink); 616 617 bool slab_is_available(void) 618 { 619 return slab_state >= UP; 620 } 621 622 #ifdef CONFIG_PRINTK 623 static void kmem_obj_info(struct kmem_obj_info *kpp, void *object, struct slab *slab) 624 { 625 if (__kfence_obj_info(kpp, object, slab)) 626 return; 627 __kmem_obj_info(kpp, object, slab); 628 } 629 630 /** 631 * kmem_dump_obj - Print available slab provenance information 632 * @object: slab object for which to find provenance information. 633 * 634 * This function uses pr_cont(), so that the caller is expected to have 635 * printed out whatever preamble is appropriate. The provenance information 636 * depends on the type of object and on how much debugging is enabled. 637 * For a slab-cache object, the fact that it is a slab object is printed, 638 * and, if available, the slab name, return address, and stack trace from 639 * the allocation and last free path of that object. 640 * 641 * Return: %true if the pointer is to a not-yet-freed object from 642 * kmalloc() or kmem_cache_alloc(), either %true or %false if the pointer 643 * is to an already-freed object, and %false otherwise. 644 */ 645 bool kmem_dump_obj(void *object) 646 { 647 char *cp = IS_ENABLED(CONFIG_MMU) ? "" : "/vmalloc"; 648 int i; 649 struct slab *slab; 650 unsigned long ptroffset; 651 struct kmem_obj_info kp = { }; 652 653 /* Some arches consider ZERO_SIZE_PTR to be a valid address. */ 654 if (object < (void *)PAGE_SIZE || !virt_addr_valid(object)) 655 return false; 656 slab = virt_to_slab(object); 657 if (!slab) 658 return false; 659 660 kmem_obj_info(&kp, object, slab); 661 if (kp.kp_slab_cache) 662 pr_cont(" slab%s %s", cp, kp.kp_slab_cache->name); 663 else 664 pr_cont(" slab%s", cp); 665 if (is_kfence_address(object)) 666 pr_cont(" (kfence)"); 667 if (kp.kp_objp) 668 pr_cont(" start %px", kp.kp_objp); 669 if (kp.kp_data_offset) 670 pr_cont(" data offset %lu", kp.kp_data_offset); 671 if (kp.kp_objp) { 672 ptroffset = ((char *)object - (char *)kp.kp_objp) - kp.kp_data_offset; 673 pr_cont(" pointer offset %lu", ptroffset); 674 } 675 if (kp.kp_slab_cache && kp.kp_slab_cache->object_size) 676 pr_cont(" size %u", kp.kp_slab_cache->object_size); 677 if (kp.kp_ret) 678 pr_cont(" allocated at %pS\n", kp.kp_ret); 679 else 680 pr_cont("\n"); 681 for (i = 0; i < ARRAY_SIZE(kp.kp_stack); i++) { 682 if (!kp.kp_stack[i]) 683 break; 684 pr_info(" %pS\n", kp.kp_stack[i]); 685 } 686 687 if (kp.kp_free_stack[0]) 688 pr_cont(" Free path:\n"); 689 690 for (i = 0; i < ARRAY_SIZE(kp.kp_free_stack); i++) { 691 if (!kp.kp_free_stack[i]) 692 break; 693 pr_info(" %pS\n", kp.kp_free_stack[i]); 694 } 695 696 return true; 697 } 698 EXPORT_SYMBOL_GPL(kmem_dump_obj); 699 #endif 700 701 /* Create a cache during boot when no slab services are available yet */ 702 void __init create_boot_cache(struct kmem_cache *s, const char *name, 703 unsigned int size, slab_flags_t flags, 704 unsigned int useroffset, unsigned int usersize) 705 { 706 int err; 707 unsigned int align = ARCH_KMALLOC_MINALIGN; 708 struct kmem_cache_args kmem_args = {}; 709 710 /* 711 * kmalloc caches guarantee alignment of at least the largest 712 * power-of-two divisor of the size. For power-of-two sizes, 713 * it is the size itself. 714 */ 715 if (flags & SLAB_KMALLOC) 716 align = max(align, 1U << (ffs(size) - 1)); 717 kmem_args.align = calculate_alignment(flags, align, size); 718 719 #ifdef CONFIG_HARDENED_USERCOPY 720 kmem_args.useroffset = useroffset; 721 kmem_args.usersize = usersize; 722 #endif 723 724 err = do_kmem_cache_create(s, name, size, &kmem_args, flags); 725 726 if (err) 727 panic("Creation of kmalloc slab %s size=%u failed. Reason %d\n", 728 name, size, err); 729 730 s->refcount = -1; /* Exempt from merging for now */ 731 } 732 733 static struct kmem_cache *__init create_kmalloc_cache(const char *name, 734 unsigned int size, 735 slab_flags_t flags) 736 { 737 struct kmem_cache *s = kmem_cache_zalloc(kmem_cache, GFP_NOWAIT); 738 739 if (!s) 740 panic("Out of memory when creating slab %s\n", name); 741 742 create_boot_cache(s, name, size, flags | SLAB_KMALLOC, 0, size); 743 list_add(&s->list, &slab_caches); 744 s->refcount = 1; 745 return s; 746 } 747 748 kmem_buckets kmalloc_caches[NR_KMALLOC_TYPES] __ro_after_init = 749 { /* initialization for https://llvm.org/pr42570 */ }; 750 EXPORT_SYMBOL(kmalloc_caches); 751 752 #ifdef CONFIG_KMALLOC_PARTITION_RANDOM 753 unsigned long random_kmalloc_seed __ro_after_init; 754 EXPORT_SYMBOL(random_kmalloc_seed); 755 #endif 756 757 /* 758 * Conversion table for small slabs sizes / 8 to the index in the 759 * kmalloc array. This is necessary for slabs < 192 since we have non power 760 * of two cache sizes there. The size of larger slabs can be determined using 761 * fls. 762 */ 763 u8 kmalloc_size_index[24] __ro_after_init = { 764 3, /* 8 */ 765 4, /* 16 */ 766 5, /* 24 */ 767 5, /* 32 */ 768 6, /* 40 */ 769 6, /* 48 */ 770 6, /* 56 */ 771 6, /* 64 */ 772 1, /* 72 */ 773 1, /* 80 */ 774 1, /* 88 */ 775 1, /* 96 */ 776 7, /* 104 */ 777 7, /* 112 */ 778 7, /* 120 */ 779 7, /* 128 */ 780 2, /* 136 */ 781 2, /* 144 */ 782 2, /* 152 */ 783 2, /* 160 */ 784 2, /* 168 */ 785 2, /* 176 */ 786 2, /* 184 */ 787 2 /* 192 */ 788 }; 789 790 size_t kmalloc_size_roundup(size_t size) 791 { 792 if (size && size <= KMALLOC_MAX_CACHE_SIZE) { 793 struct kmem_cache *s; 794 795 /* 796 * The flags don't matter since size_index is common to all. 797 * Neither does the caller for just getting ->object_size. 798 */ 799 s = kmalloc_slab(size, NULL, GFP_KERNEL, __kmalloc_token(0), 800 SLAB_ALLOC_DEFAULT); 801 return s->object_size; 802 } 803 804 /* Above the smaller buckets, size is a multiple of page size. */ 805 if (size && size <= KMALLOC_MAX_SIZE) 806 return PAGE_SIZE << get_order(size); 807 808 /* 809 * Return 'size' for 0 - kmalloc() returns ZERO_SIZE_PTR 810 * and very large size - kmalloc() may fail. 811 */ 812 return size; 813 814 } 815 EXPORT_SYMBOL(kmalloc_size_roundup); 816 817 #ifdef CONFIG_ZONE_DMA 818 #define KMALLOC_DMA_NAME(sz) .name[KMALLOC_DMA] = "dma-kmalloc-" #sz, 819 #else 820 #define KMALLOC_DMA_NAME(sz) 821 #endif 822 823 #ifdef CONFIG_MEMCG 824 #define KMALLOC_CGROUP_NAME(sz) .name[KMALLOC_CGROUP] = "kmalloc-cg-" #sz, 825 #else 826 #define KMALLOC_CGROUP_NAME(sz) 827 #endif 828 829 #ifndef CONFIG_SLUB_TINY 830 #define KMALLOC_RCL_NAME(sz) .name[KMALLOC_RECLAIM] = "kmalloc-rcl-" #sz, 831 #else 832 #define KMALLOC_RCL_NAME(sz) 833 #endif 834 835 #ifdef CONFIG_KMALLOC_PARTITION_CACHES 836 #define __KMALLOC_PARTITION_CONCAT(a, b) a ## b 837 #define KMALLOC_PARTITION_NAME(N, sz) __KMALLOC_PARTITION_CONCAT(KMA_PART_, N)(sz) 838 #define KMA_PART_1(sz) .name[KMALLOC_PARTITION_START + 1] = "kmalloc-part-01-" #sz, 839 #define KMA_PART_2(sz) KMA_PART_1(sz) .name[KMALLOC_PARTITION_START + 2] = "kmalloc-part-02-" #sz, 840 #define KMA_PART_3(sz) KMA_PART_2(sz) .name[KMALLOC_PARTITION_START + 3] = "kmalloc-part-03-" #sz, 841 #define KMA_PART_4(sz) KMA_PART_3(sz) .name[KMALLOC_PARTITION_START + 4] = "kmalloc-part-04-" #sz, 842 #define KMA_PART_5(sz) KMA_PART_4(sz) .name[KMALLOC_PARTITION_START + 5] = "kmalloc-part-05-" #sz, 843 #define KMA_PART_6(sz) KMA_PART_5(sz) .name[KMALLOC_PARTITION_START + 6] = "kmalloc-part-06-" #sz, 844 #define KMA_PART_7(sz) KMA_PART_6(sz) .name[KMALLOC_PARTITION_START + 7] = "kmalloc-part-07-" #sz, 845 #define KMA_PART_8(sz) KMA_PART_7(sz) .name[KMALLOC_PARTITION_START + 8] = "kmalloc-part-08-" #sz, 846 #define KMA_PART_9(sz) KMA_PART_8(sz) .name[KMALLOC_PARTITION_START + 9] = "kmalloc-part-09-" #sz, 847 #define KMA_PART_10(sz) KMA_PART_9(sz) .name[KMALLOC_PARTITION_START + 10] = "kmalloc-part-10-" #sz, 848 #define KMA_PART_11(sz) KMA_PART_10(sz) .name[KMALLOC_PARTITION_START + 11] = "kmalloc-part-11-" #sz, 849 #define KMA_PART_12(sz) KMA_PART_11(sz) .name[KMALLOC_PARTITION_START + 12] = "kmalloc-part-12-" #sz, 850 #define KMA_PART_13(sz) KMA_PART_12(sz) .name[KMALLOC_PARTITION_START + 13] = "kmalloc-part-13-" #sz, 851 #define KMA_PART_14(sz) KMA_PART_13(sz) .name[KMALLOC_PARTITION_START + 14] = "kmalloc-part-14-" #sz, 852 #define KMA_PART_15(sz) KMA_PART_14(sz) .name[KMALLOC_PARTITION_START + 15] = "kmalloc-part-15-" #sz, 853 #else // CONFIG_KMALLOC_PARTITION_CACHES 854 #define KMALLOC_PARTITION_NAME(N, sz) 855 #endif 856 857 #ifdef CONFIG_SLAB_OBJ_EXT 858 #define KMALLOC_NO_OBJ_EXT_NAME(sz) .name[KMALLOC_NO_OBJ_EXT] = "kmalloc-no-objext-" #sz, 859 #else 860 #define KMALLOC_NO_OBJ_EXT_NAME(sz) 861 #endif 862 863 #define INIT_KMALLOC_INFO(__size, __short_size) \ 864 { \ 865 .name[KMALLOC_NORMAL] = "kmalloc-" #__short_size, \ 866 KMALLOC_RCL_NAME(__short_size) \ 867 KMALLOC_CGROUP_NAME(__short_size) \ 868 KMALLOC_DMA_NAME(__short_size) \ 869 KMALLOC_PARTITION_NAME(KMALLOC_PARTITION_CACHES_NR, __short_size) \ 870 KMALLOC_NO_OBJ_EXT_NAME(__short_size) \ 871 .size = __size, \ 872 } 873 874 /* 875 * kmalloc_info[] is to make slab_debug=,kmalloc-xx option work at boot time. 876 * kmalloc_index() supports up to 2^21=2MB, so the final entry of the table is 877 * kmalloc-2M. 878 */ 879 const struct kmalloc_info_struct kmalloc_info[] __initconst = { 880 INIT_KMALLOC_INFO(0, 0), 881 INIT_KMALLOC_INFO(96, 96), 882 INIT_KMALLOC_INFO(192, 192), 883 INIT_KMALLOC_INFO(8, 8), 884 INIT_KMALLOC_INFO(16, 16), 885 INIT_KMALLOC_INFO(32, 32), 886 INIT_KMALLOC_INFO(64, 64), 887 INIT_KMALLOC_INFO(128, 128), 888 INIT_KMALLOC_INFO(256, 256), 889 INIT_KMALLOC_INFO(512, 512), 890 INIT_KMALLOC_INFO(1024, 1k), 891 INIT_KMALLOC_INFO(2048, 2k), 892 INIT_KMALLOC_INFO(4096, 4k), 893 INIT_KMALLOC_INFO(8192, 8k), 894 INIT_KMALLOC_INFO(16384, 16k), 895 INIT_KMALLOC_INFO(32768, 32k), 896 INIT_KMALLOC_INFO(65536, 64k), 897 INIT_KMALLOC_INFO(131072, 128k), 898 INIT_KMALLOC_INFO(262144, 256k), 899 INIT_KMALLOC_INFO(524288, 512k), 900 INIT_KMALLOC_INFO(1048576, 1M), 901 INIT_KMALLOC_INFO(2097152, 2M) 902 }; 903 904 /* 905 * Patch up the size_index table if we have strange large alignment 906 * requirements for the kmalloc array. This is only the case for 907 * MIPS it seems. The standard arches will not generate any code here. 908 * 909 * Largest permitted alignment is 256 bytes due to the way we 910 * handle the index determination for the smaller caches. 911 * 912 * Make sure that nothing crazy happens if someone starts tinkering 913 * around with ARCH_KMALLOC_MINALIGN 914 */ 915 void __init setup_kmalloc_cache_index_table(void) 916 { 917 unsigned int i; 918 919 BUILD_BUG_ON(KMALLOC_MIN_SIZE > 256 || 920 !is_power_of_2(KMALLOC_MIN_SIZE)); 921 922 for (i = 8; i < KMALLOC_MIN_SIZE; i += 8) { 923 unsigned int elem = size_index_elem(i); 924 925 if (elem >= ARRAY_SIZE(kmalloc_size_index)) 926 break; 927 kmalloc_size_index[elem] = KMALLOC_SHIFT_LOW; 928 } 929 930 if (KMALLOC_MIN_SIZE >= 64) { 931 /* 932 * The 96 byte sized cache is not used if the alignment 933 * is 64 byte. 934 */ 935 for (i = 64 + 8; i <= 96; i += 8) 936 kmalloc_size_index[size_index_elem(i)] = 7; 937 938 } 939 940 if (KMALLOC_MIN_SIZE >= 128) { 941 /* 942 * The 192 byte sized cache is not used if the alignment 943 * is 128 byte. Redirect kmalloc to use the 256 byte cache 944 * instead. 945 */ 946 for (i = 128 + 8; i <= 192; i += 8) 947 kmalloc_size_index[size_index_elem(i)] = 8; 948 } 949 } 950 951 static unsigned int __kmalloc_minalign(void) 952 { 953 unsigned int minalign = dma_get_cache_alignment(); 954 955 if (IS_ENABLED(CONFIG_DMA_BOUNCE_UNALIGNED_KMALLOC) && 956 is_swiotlb_allocated()) 957 minalign = ARCH_KMALLOC_MINALIGN; 958 959 return max(minalign, arch_slab_minalign()); 960 } 961 962 static void __init 963 new_kmalloc_cache(int idx, enum kmalloc_cache_type type) 964 { 965 slab_flags_t flags = 0; 966 unsigned int minalign = __kmalloc_minalign(); 967 unsigned int aligned_size = kmalloc_info[idx].size; 968 int aligned_idx = idx; 969 970 if ((KMALLOC_RECLAIM != KMALLOC_NORMAL) && (type == KMALLOC_RECLAIM)) { 971 flags |= SLAB_RECLAIM_ACCOUNT; 972 } else if (IS_ENABLED(CONFIG_MEMCG) && (type == KMALLOC_CGROUP)) { 973 if (mem_cgroup_kmem_disabled()) { 974 kmalloc_caches[type][idx] = kmalloc_caches[KMALLOC_NORMAL][idx]; 975 return; 976 } 977 flags |= SLAB_ACCOUNT; 978 } else if (IS_ENABLED(CONFIG_SLAB_OBJ_EXT) && type == KMALLOC_NO_OBJ_EXT) { 979 if (!need_kmalloc_no_objext()) { 980 kmalloc_caches[type][idx] = kmalloc_caches[KMALLOC_NORMAL][idx]; 981 return; 982 } 983 flags |= SLAB_NO_OBJ_EXT | SLAB_NO_MERGE; 984 } else if (IS_ENABLED(CONFIG_ZONE_DMA) && (type == KMALLOC_DMA)) { 985 flags |= SLAB_CACHE_DMA; 986 } 987 988 #ifdef CONFIG_KMALLOC_PARTITION_CACHES 989 if (type >= KMALLOC_PARTITION_START && type <= KMALLOC_PARTITION_END) 990 flags |= SLAB_NO_MERGE; 991 #endif 992 993 /* 994 * If memcg_kmem is enabled and this is a KMALLOC_NORMAL cache and not 995 * aliased with any other type, make sure it's never merged with any other 996 * cache. 997 * 998 * In other cases the kmalloc cache may end up being used for a 999 * __GFP_ACCOUNT allocation so mark it as such. The exception is a 1000 * KMALLOC_NO_OBJ_EXT cache. 1001 */ 1002 if (!mem_cgroup_kmem_disabled()) { 1003 if (type == KMALLOC_NORMAL && KMALLOC_RECLAIM != KMALLOC_NORMAL) 1004 flags |= SLAB_NO_MERGE; 1005 else if (!(flags & SLAB_NO_OBJ_EXT)) 1006 flags |= SLAB_MAY_ACCOUNT; 1007 } 1008 1009 if (minalign > ARCH_KMALLOC_MINALIGN) { 1010 aligned_size = ALIGN(aligned_size, minalign); 1011 aligned_idx = __kmalloc_index(aligned_size, false); 1012 } 1013 1014 if (!kmalloc_caches[type][aligned_idx]) 1015 kmalloc_caches[type][aligned_idx] = create_kmalloc_cache( 1016 kmalloc_info[aligned_idx].name[type], 1017 aligned_size, flags); 1018 if (idx != aligned_idx) 1019 kmalloc_caches[type][idx] = kmalloc_caches[type][aligned_idx]; 1020 } 1021 1022 /* 1023 * Create the kmalloc array. Some of the regular kmalloc arrays 1024 * may already have been created because they were needed to 1025 * enable allocations for slab creation. 1026 */ 1027 void __init create_kmalloc_caches(void) 1028 { 1029 int i; 1030 enum kmalloc_cache_type type; 1031 1032 /* 1033 * Including KMALLOC_CGROUP if CONFIG_MEMCG defined 1034 */ 1035 for (type = KMALLOC_NORMAL; type < NR_KMALLOC_TYPES; type++) { 1036 /* Caches that are NOT of the two-to-the-power-of size. */ 1037 if (KMALLOC_MIN_SIZE <= 32) 1038 new_kmalloc_cache(1, type); 1039 if (KMALLOC_MIN_SIZE <= 64) 1040 new_kmalloc_cache(2, type); 1041 1042 /* Caches that are of the two-to-the-power-of size. */ 1043 for (i = KMALLOC_SHIFT_LOW; i <= KMALLOC_SHIFT_HIGH; i++) 1044 new_kmalloc_cache(i, type); 1045 } 1046 #ifdef CONFIG_KMALLOC_PARTITION_RANDOM 1047 random_kmalloc_seed = get_random_u64(); 1048 #endif 1049 1050 /* Kmalloc array is now usable */ 1051 slab_state = UP; 1052 1053 if (IS_ENABLED(CONFIG_SLAB_BUCKETS)) 1054 kmem_buckets_cache = kmem_cache_create("kmalloc_buckets", 1055 sizeof(kmem_buckets), 1056 0, SLAB_NO_MERGE, NULL); 1057 } 1058 1059 gfp_t kmalloc_fix_flags(gfp_t flags) 1060 { 1061 gfp_t invalid_mask = flags & GFP_SLAB_BUG_MASK; 1062 1063 flags &= ~GFP_SLAB_BUG_MASK; 1064 pr_warn("Unexpected gfp: %#x (%pGg). Fixing up to gfp: %#x (%pGg). Fix your code!\n", 1065 invalid_mask, &invalid_mask, flags, &flags); 1066 dump_stack(); 1067 1068 return flags; 1069 } 1070 1071 #ifdef CONFIG_SLAB_FREELIST_RANDOM 1072 /* Randomize a generic freelist */ 1073 static void freelist_randomize(unsigned int *list, 1074 unsigned int count) 1075 { 1076 unsigned int rand; 1077 unsigned int i; 1078 1079 for (i = 0; i < count; i++) 1080 list[i] = i; 1081 1082 /* Fisher-Yates shuffle */ 1083 for (i = count - 1; i > 0; i--) { 1084 rand = get_random_u32_below(i + 1); 1085 swap(list[i], list[rand]); 1086 } 1087 } 1088 1089 /* Create a random sequence per cache */ 1090 int cache_random_seq_create(struct kmem_cache *cachep, unsigned int count, 1091 gfp_t gfp) 1092 { 1093 1094 if (count < 2 || cachep->random_seq) 1095 return 0; 1096 1097 cachep->random_seq = kcalloc(count, sizeof(unsigned int), gfp); 1098 if (!cachep->random_seq) 1099 return -ENOMEM; 1100 1101 freelist_randomize(cachep->random_seq, count); 1102 return 0; 1103 } 1104 1105 /* Destroy the per-cache random freelist sequence */ 1106 void cache_random_seq_destroy(struct kmem_cache *cachep) 1107 { 1108 kfree(cachep->random_seq); 1109 cachep->random_seq = NULL; 1110 } 1111 #endif /* CONFIG_SLAB_FREELIST_RANDOM */ 1112 1113 #ifdef CONFIG_SLUB_DEBUG 1114 #define SLABINFO_RIGHTS (0400) 1115 1116 static void print_slabinfo_header(struct seq_file *m) 1117 { 1118 /* 1119 * Output format version, so at least we can change it 1120 * without _too_ many complaints. 1121 */ 1122 seq_puts(m, "slabinfo - version: 2.1\n"); 1123 seq_puts(m, "# name <active_objs> <num_objs> <objsize> <objperslab> <pagesperslab>"); 1124 seq_puts(m, " : tunables <limit> <batchcount> <sharedfactor>"); 1125 seq_puts(m, " : slabdata <active_slabs> <num_slabs> <sharedavail>"); 1126 seq_putc(m, '\n'); 1127 } 1128 1129 static void *slab_start(struct seq_file *m, loff_t *pos) 1130 { 1131 mutex_lock(&slab_mutex); 1132 return seq_list_start(&slab_caches, *pos); 1133 } 1134 1135 static void *slab_next(struct seq_file *m, void *p, loff_t *pos) 1136 { 1137 return seq_list_next(p, &slab_caches, pos); 1138 } 1139 1140 static void slab_stop(struct seq_file *m, void *p) 1141 { 1142 mutex_unlock(&slab_mutex); 1143 } 1144 1145 static void cache_show(struct kmem_cache *s, struct seq_file *m) 1146 { 1147 struct slabinfo sinfo; 1148 1149 memset(&sinfo, 0, sizeof(sinfo)); 1150 get_slabinfo(s, &sinfo); 1151 1152 seq_printf(m, "%-17s %6lu %6lu %6u %4u %4d", 1153 s->name, sinfo.active_objs, sinfo.num_objs, s->size, 1154 sinfo.objects_per_slab, (1 << sinfo.cache_order)); 1155 1156 seq_printf(m, " : tunables %4u %4u %4u", 1157 sinfo.limit, sinfo.batchcount, sinfo.shared); 1158 seq_printf(m, " : slabdata %6lu %6lu %6lu", 1159 sinfo.active_slabs, sinfo.num_slabs, sinfo.shared_avail); 1160 seq_putc(m, '\n'); 1161 } 1162 1163 static int slab_show(struct seq_file *m, void *p) 1164 { 1165 struct kmem_cache *s = list_entry(p, struct kmem_cache, list); 1166 1167 if (p == slab_caches.next) 1168 print_slabinfo_header(m); 1169 cache_show(s, m); 1170 return 0; 1171 } 1172 1173 void dump_unreclaimable_slab(void) 1174 { 1175 struct kmem_cache *s; 1176 struct slabinfo sinfo; 1177 1178 /* 1179 * Here acquiring slab_mutex is risky since we don't prefer to get 1180 * sleep in oom path. But, without mutex hold, it may introduce a 1181 * risk of crash. 1182 * Use mutex_trylock to protect the list traverse, dump nothing 1183 * without acquiring the mutex. 1184 */ 1185 if (!mutex_trylock(&slab_mutex)) { 1186 pr_warn("excessive unreclaimable slab but cannot dump stats\n"); 1187 return; 1188 } 1189 1190 pr_info("Unreclaimable slab info:\n"); 1191 pr_info("Name Used Total\n"); 1192 1193 list_for_each_entry(s, &slab_caches, list) { 1194 if (s->flags & SLAB_RECLAIM_ACCOUNT) 1195 continue; 1196 1197 get_slabinfo(s, &sinfo); 1198 1199 if (sinfo.num_objs > 0) 1200 pr_info("%-17s %10luKB %10luKB\n", s->name, 1201 (sinfo.active_objs * s->size) / 1024, 1202 (sinfo.num_objs * s->size) / 1024); 1203 } 1204 mutex_unlock(&slab_mutex); 1205 } 1206 1207 /* 1208 * slabinfo_op - iterator that generates /proc/slabinfo 1209 * 1210 * Output layout: 1211 * cache-name 1212 * num-active-objs 1213 * total-objs 1214 * object size 1215 * num-active-slabs 1216 * total-slabs 1217 * num-pages-per-slab 1218 * + further values on SMP and with statistics enabled 1219 */ 1220 static const struct seq_operations slabinfo_op = { 1221 .start = slab_start, 1222 .next = slab_next, 1223 .stop = slab_stop, 1224 .show = slab_show, 1225 }; 1226 1227 static int slabinfo_open(struct inode *inode, struct file *file) 1228 { 1229 return seq_open(file, &slabinfo_op); 1230 } 1231 1232 static const struct proc_ops slabinfo_proc_ops = { 1233 .proc_flags = PROC_ENTRY_PERMANENT, 1234 .proc_open = slabinfo_open, 1235 .proc_read = seq_read, 1236 .proc_lseek = seq_lseek, 1237 .proc_release = seq_release, 1238 }; 1239 1240 static int __init slab_proc_init(void) 1241 { 1242 proc_create("slabinfo", SLABINFO_RIGHTS, NULL, &slabinfo_proc_ops); 1243 return 0; 1244 } 1245 module_init(slab_proc_init); 1246 1247 #endif /* CONFIG_SLUB_DEBUG */ 1248 1249 /** 1250 * kfree_sensitive - Clear sensitive information in memory before freeing 1251 * @p: object to free memory of 1252 * 1253 * The memory of the object @p points to is zeroed before freed. 1254 * If @p is %NULL, kfree_sensitive() does nothing. 1255 * 1256 * Note: this function zeroes the whole allocated buffer which can be a good 1257 * deal bigger than the requested buffer size passed to kmalloc(). So be 1258 * careful when using this function in performance sensitive code. 1259 */ 1260 void kfree_sensitive(const void *p) 1261 { 1262 size_t ks; 1263 void *mem = (void *)p; 1264 1265 ks = ksize(mem); 1266 if (ks) { 1267 kasan_unpoison_range(mem, ks); 1268 memzero_explicit(mem, ks); 1269 } 1270 kfree(mem); 1271 } 1272 EXPORT_SYMBOL(kfree_sensitive); 1273 1274 #ifdef CONFIG_BPF_SYSCALL 1275 #include <linux/btf.h> 1276 1277 __bpf_kfunc_start_defs(); 1278 1279 __bpf_kfunc struct kmem_cache *bpf_get_kmem_cache(u64 addr) 1280 { 1281 struct slab *slab; 1282 1283 if (!virt_addr_valid((void *)(long)addr)) 1284 return NULL; 1285 1286 slab = virt_to_slab((void *)(long)addr); 1287 return slab ? slab->slab_cache : NULL; 1288 } 1289 1290 __bpf_kfunc_end_defs(); 1291 #endif /* CONFIG_BPF_SYSCALL */ 1292 1293 /* Tracepoints definitions. */ 1294 EXPORT_TRACEPOINT_SYMBOL(kmalloc); 1295 EXPORT_TRACEPOINT_SYMBOL(kmem_cache_alloc); 1296 EXPORT_TRACEPOINT_SYMBOL(kfree); 1297 EXPORT_TRACEPOINT_SYMBOL(kmem_cache_free); 1298 1299 void kfree_call_rcu_nolock(struct kvfree_rcu_head *head, void *ptr) 1300 { 1301 struct slab *slab; 1302 1303 if (!IS_ENABLED(CONFIG_KVFREE_RCU_BATCHED)) 1304 goto fallback; 1305 1306 if (unlikely(is_vmalloc_addr(ptr))) 1307 goto fallback; 1308 1309 slab = virt_to_slab(ptr); 1310 if (unlikely(!slab)) 1311 goto fallback; 1312 1313 if (unlikely(IS_ENABLED(CONFIG_NUMA) && slab_nid(slab) != numa_mem_id())) 1314 goto fallback; 1315 1316 if (unlikely(!__kfree_rcu_sheaf(slab->slab_cache, ptr, SLAB_FREE_NOLOCK))) 1317 goto fallback; 1318 1319 return; 1320 1321 fallback: 1322 defer_kfree_rcu(head); 1323 } 1324 EXPORT_SYMBOL_GPL(kfree_call_rcu_nolock); 1325 1326 #ifndef CONFIG_KVFREE_RCU_BATCHED 1327 1328 void kvfree_call_rcu(struct kvfree_rcu_head *head, void *ptr) 1329 { 1330 if (head) { 1331 kasan_record_aux_stack(ptr); 1332 call_rcu(&head->head, kvfree_rcu_cb); 1333 return; 1334 } 1335 1336 // kvfree_rcu(one_arg) call. 1337 might_sleep(); 1338 synchronize_rcu(); 1339 kvfree(ptr); 1340 } 1341 EXPORT_SYMBOL_GPL(kvfree_call_rcu); 1342 1343 void kvfree_rcu_barrier(void) 1344 { 1345 deferred_work_barrier(); 1346 rcu_barrier(); 1347 } 1348 1349 void kvfree_rcu_barrier_on_cache(struct kmem_cache *s) 1350 { 1351 deferred_work_barrier(); 1352 rcu_barrier(); 1353 } 1354 1355 void __init kvfree_rcu_init(void) 1356 { 1357 } 1358 1359 #else /* CONFIG_KVFREE_RCU_BATCHED */ 1360 1361 /* 1362 * This rcu parameter is runtime-read-only. It reflects 1363 * a minimum allowed number of objects which can be cached 1364 * per-CPU. Object size is equal to one page. This value 1365 * can be changed at boot time. 1366 */ 1367 static int rcu_min_cached_objs = 5; 1368 module_param(rcu_min_cached_objs, int, 0444); 1369 1370 // A page shrinker can ask for pages to be freed to make them 1371 // available for other parts of the system. This usually happens 1372 // under low memory conditions, and in that case we should also 1373 // defer page-cache filling for a short time period. 1374 // 1375 // The default value is 5 seconds, which is long enough to reduce 1376 // interference with the shrinker while it asks other systems to 1377 // drain their caches. 1378 static int rcu_delay_page_cache_fill_msec = 5000; 1379 module_param(rcu_delay_page_cache_fill_msec, int, 0444); 1380 1381 static struct workqueue_struct *rcu_reclaim_wq; 1382 1383 /* Maximum number of jiffies to wait before draining a batch. */ 1384 #define KFREE_DRAIN_JIFFIES (5 * HZ) 1385 #define KFREE_N_BATCHES 2 1386 #define FREE_N_CHANNELS 2 1387 1388 /** 1389 * struct kvfree_rcu_bulk_data - single block to store kvfree_rcu() pointers 1390 * @list: List node. All blocks are linked between each other 1391 * @gp_snap: Snapshot of RCU state for objects placed to this bulk 1392 * @nr_records: Number of active pointers in the array 1393 * @records: Array of the kvfree_rcu() pointers 1394 */ 1395 struct kvfree_rcu_bulk_data { 1396 struct list_head list; 1397 struct rcu_gp_seq gp_snap; 1398 unsigned long nr_records; 1399 void *records[] __counted_by(nr_records); 1400 }; 1401 1402 /* 1403 * This macro defines how many entries the "records" array 1404 * will contain. It is based on the fact that the size of 1405 * kvfree_rcu_bulk_data structure becomes exactly one page. 1406 */ 1407 #define KVFREE_BULK_MAX_ENTR \ 1408 ((PAGE_SIZE - sizeof(struct kvfree_rcu_bulk_data)) / sizeof(void *)) 1409 1410 /** 1411 * struct kfree_rcu_cpu_work - single batch of kfree_rcu() requests 1412 * @rcu_work: Let queue_rcu_work() invoke workqueue handler after grace period 1413 * @head_free: List of kfree_rcu() objects waiting for a grace period 1414 * @head_free_gp_snap: Grace-period snapshot to check for attempted premature frees. 1415 * @bulk_head_free: Bulk-List of kvfree_rcu() objects waiting for a grace period 1416 * @krcp: Pointer to @kfree_rcu_cpu structure 1417 */ 1418 1419 struct kfree_rcu_cpu_work { 1420 struct rcu_work rcu_work; 1421 struct kvfree_rcu_head *head_free; 1422 struct rcu_gp_seq head_free_gp_snap; 1423 struct list_head bulk_head_free[FREE_N_CHANNELS]; 1424 struct kfree_rcu_cpu *krcp; 1425 }; 1426 1427 /** 1428 * struct kfree_rcu_cpu - batch up kfree_rcu() requests for RCU grace period 1429 * @head: List of kfree_rcu() objects not yet waiting for a grace period 1430 * @head_gp_snap: Snapshot of RCU state for objects placed to "@head" 1431 * @bulk_head: Bulk-List of kvfree_rcu() objects not yet waiting for a grace period 1432 * @krw_arr: Array of batches of kfree_rcu() objects waiting for a grace period 1433 * @lock: Synchronize access to this structure 1434 * @monitor_work: Promote @head to @head_free after KFREE_DRAIN_JIFFIES 1435 * @initialized: The @rcu_work fields have been initialized 1436 * @head_count: Number of objects in rcu_head singular list 1437 * @bulk_count: Number of objects in bulk-list 1438 * @bkvcache: 1439 * A simple cache list that contains objects for reuse purpose. 1440 * In order to save some per-cpu space the list is singular. 1441 * Even though it is lockless an access has to be protected by the 1442 * per-cpu lock. 1443 * @page_cache_work: A work to refill the cache when it is empty 1444 * @backoff_page_cache_fill: Delay cache refills 1445 * @work_in_progress: Indicates that page_cache_work is running 1446 * @hrtimer: A hrtimer for scheduling a page_cache_work 1447 * @nr_bkv_objs: number of allocated objects at @bkvcache. 1448 * 1449 * This is a per-CPU structure. The reason that it is not included in 1450 * the rcu_data structure is to permit this code to be extracted from 1451 * the RCU files. Such extraction could allow further optimization of 1452 * the interactions with the slab allocators. 1453 */ 1454 struct kfree_rcu_cpu { 1455 // Objects queued on a linked list 1456 // through their rcu_head structures. 1457 struct kvfree_rcu_head *head; 1458 unsigned long head_gp_snap; 1459 atomic_t head_count; 1460 1461 // Objects queued on a bulk-list. 1462 struct list_head bulk_head[FREE_N_CHANNELS]; 1463 atomic_t bulk_count[FREE_N_CHANNELS]; 1464 1465 struct kfree_rcu_cpu_work krw_arr[KFREE_N_BATCHES]; 1466 raw_spinlock_t lock; 1467 struct delayed_work monitor_work; 1468 bool initialized; 1469 1470 struct delayed_work page_cache_work; 1471 atomic_t backoff_page_cache_fill; 1472 atomic_t work_in_progress; 1473 struct hrtimer hrtimer; 1474 1475 struct llist_head bkvcache; 1476 int nr_bkv_objs; 1477 }; 1478 1479 static DEFINE_PER_CPU(struct kfree_rcu_cpu, krc) = { 1480 .lock = __RAW_SPIN_LOCK_UNLOCKED(krc.lock), 1481 }; 1482 1483 static __always_inline void 1484 debug_rcu_bhead_unqueue(struct kvfree_rcu_bulk_data *bhead) 1485 { 1486 #ifdef CONFIG_DEBUG_OBJECTS_RCU_HEAD 1487 int i; 1488 1489 for (i = 0; i < bhead->nr_records; i++) 1490 debug_rcu_head_unqueue((struct rcu_head *)(bhead->records[i])); 1491 #endif 1492 } 1493 1494 static inline struct kfree_rcu_cpu * 1495 krc_this_cpu_lock(unsigned long *flags) 1496 { 1497 struct kfree_rcu_cpu *krcp; 1498 1499 local_irq_save(*flags); // For safely calling this_cpu_ptr(). 1500 krcp = this_cpu_ptr(&krc); 1501 raw_spin_lock(&krcp->lock); 1502 1503 return krcp; 1504 } 1505 1506 static inline void 1507 krc_this_cpu_unlock(struct kfree_rcu_cpu *krcp, unsigned long flags) 1508 { 1509 raw_spin_unlock_irqrestore(&krcp->lock, flags); 1510 } 1511 1512 static inline struct kvfree_rcu_bulk_data * 1513 get_cached_bnode(struct kfree_rcu_cpu *krcp) 1514 { 1515 if (!krcp->nr_bkv_objs) 1516 return NULL; 1517 1518 WRITE_ONCE(krcp->nr_bkv_objs, krcp->nr_bkv_objs - 1); 1519 return (struct kvfree_rcu_bulk_data *) 1520 llist_del_first(&krcp->bkvcache); 1521 } 1522 1523 static inline bool 1524 put_cached_bnode(struct kfree_rcu_cpu *krcp, 1525 struct kvfree_rcu_bulk_data *bnode) 1526 { 1527 // Check the limit. 1528 if (krcp->nr_bkv_objs >= rcu_min_cached_objs) 1529 return false; 1530 1531 llist_add((struct llist_node *) bnode, &krcp->bkvcache); 1532 WRITE_ONCE(krcp->nr_bkv_objs, krcp->nr_bkv_objs + 1); 1533 return true; 1534 } 1535 1536 static int 1537 drain_page_cache(struct kfree_rcu_cpu *krcp) 1538 { 1539 unsigned long flags; 1540 struct llist_node *page_list, *pos, *n; 1541 int freed = 0; 1542 1543 if (!rcu_min_cached_objs) 1544 return 0; 1545 1546 raw_spin_lock_irqsave(&krcp->lock, flags); 1547 page_list = llist_del_all(&krcp->bkvcache); 1548 WRITE_ONCE(krcp->nr_bkv_objs, 0); 1549 raw_spin_unlock_irqrestore(&krcp->lock, flags); 1550 1551 llist_for_each_safe(pos, n, page_list) { 1552 free_page((unsigned long)pos); 1553 freed++; 1554 } 1555 1556 return freed; 1557 } 1558 1559 static void 1560 kvfree_rcu_bulk(struct kfree_rcu_cpu *krcp, 1561 struct kvfree_rcu_bulk_data *bnode, int idx) 1562 { 1563 unsigned long flags; 1564 int i; 1565 1566 if (!WARN_ON_ONCE(!poll_state_synchronize_rcu_full(&bnode->gp_snap))) { 1567 debug_rcu_bhead_unqueue(bnode); 1568 rcu_lock_acquire(&rcu_callback_map); 1569 if (idx == 0) { // kmalloc() / kfree(). 1570 trace_rcu_invoke_kfree_bulk_callback( 1571 "slab", bnode->nr_records, 1572 bnode->records); 1573 1574 kfree_bulk(bnode->nr_records, bnode->records); 1575 } else { // vmalloc() / vfree(). 1576 for (i = 0; i < bnode->nr_records; i++) { 1577 trace_rcu_invoke_kvfree_callback( 1578 "slab", bnode->records[i], 0); 1579 1580 vfree(bnode->records[i]); 1581 } 1582 } 1583 rcu_lock_release(&rcu_callback_map); 1584 } 1585 1586 raw_spin_lock_irqsave(&krcp->lock, flags); 1587 if (put_cached_bnode(krcp, bnode)) 1588 bnode = NULL; 1589 raw_spin_unlock_irqrestore(&krcp->lock, flags); 1590 1591 if (bnode) 1592 free_page((unsigned long) bnode); 1593 1594 cond_resched_tasks_rcu_qs(); 1595 } 1596 1597 static void 1598 kvfree_rcu_list(struct kvfree_rcu_head *head) 1599 { 1600 struct kvfree_rcu_head *next; 1601 1602 for (; head; head = next) { 1603 void *ptr = kvmalloc_obj_start_addr(head); 1604 unsigned long offset = (void *) head - ptr; 1605 1606 next = head->next; 1607 debug_rcu_head_unqueue((struct rcu_head *)ptr); 1608 rcu_lock_acquire(&rcu_callback_map); 1609 trace_rcu_invoke_kvfree_callback("slab", head, offset); 1610 1611 kvfree(ptr); 1612 1613 rcu_lock_release(&rcu_callback_map); 1614 cond_resched_tasks_rcu_qs(); 1615 } 1616 } 1617 1618 /* 1619 * This function is invoked in workqueue context after a grace period. 1620 * It frees all the objects queued on ->bulk_head_free or ->head_free. 1621 */ 1622 static void kfree_rcu_work(struct work_struct *work) 1623 { 1624 unsigned long flags; 1625 struct kvfree_rcu_bulk_data *bnode, *n; 1626 struct list_head bulk_head[FREE_N_CHANNELS]; 1627 struct kvfree_rcu_head *head; 1628 struct kfree_rcu_cpu *krcp; 1629 struct kfree_rcu_cpu_work *krwp; 1630 struct rcu_gp_seq head_gp_snap; 1631 int i; 1632 1633 krwp = container_of(to_rcu_work(work), 1634 struct kfree_rcu_cpu_work, rcu_work); 1635 krcp = krwp->krcp; 1636 1637 raw_spin_lock_irqsave(&krcp->lock, flags); 1638 // Channels 1 and 2. 1639 for (i = 0; i < FREE_N_CHANNELS; i++) 1640 list_replace_init(&krwp->bulk_head_free[i], &bulk_head[i]); 1641 1642 // Channel 3. 1643 head = krwp->head_free; 1644 krwp->head_free = NULL; 1645 head_gp_snap = krwp->head_free_gp_snap; 1646 raw_spin_unlock_irqrestore(&krcp->lock, flags); 1647 1648 // Handle the first two channels. 1649 for (i = 0; i < FREE_N_CHANNELS; i++) { 1650 // Start from the tail page, so a GP is likely passed for it. 1651 list_for_each_entry_safe(bnode, n, &bulk_head[i], list) 1652 kvfree_rcu_bulk(krcp, bnode, i); 1653 } 1654 1655 /* 1656 * This is used when the "bulk" path can not be used for the 1657 * double-argument of kvfree_rcu(). This happens when the 1658 * page-cache is empty, which means that objects are instead 1659 * queued on a linked list through their rcu_head structures. 1660 * This list is named "Channel 3". 1661 */ 1662 if (head && !WARN_ON_ONCE(!poll_state_synchronize_rcu_full(&head_gp_snap))) 1663 kvfree_rcu_list(head); 1664 } 1665 1666 static bool kfree_rcu_sheaf(void *obj) 1667 { 1668 struct kmem_cache *s; 1669 struct slab *slab; 1670 1671 if (is_vmalloc_addr(obj)) 1672 return false; 1673 1674 slab = virt_to_slab(obj); 1675 if (unlikely(!slab)) 1676 return false; 1677 1678 s = slab->slab_cache; 1679 if (likely(!IS_ENABLED(CONFIG_NUMA) || slab_nid(slab) == numa_mem_id())) 1680 return __kfree_rcu_sheaf(s, obj, SLAB_FREE_DEFAULT); 1681 1682 return false; 1683 } 1684 1685 static bool 1686 need_offload_krc(struct kfree_rcu_cpu *krcp) 1687 { 1688 int i; 1689 1690 for (i = 0; i < FREE_N_CHANNELS; i++) 1691 if (!list_empty(&krcp->bulk_head[i])) 1692 return true; 1693 1694 return !!READ_ONCE(krcp->head); 1695 } 1696 1697 static bool 1698 need_wait_for_krwp_work(struct kfree_rcu_cpu_work *krwp) 1699 { 1700 int i; 1701 1702 for (i = 0; i < FREE_N_CHANNELS; i++) 1703 if (!list_empty(&krwp->bulk_head_free[i])) 1704 return true; 1705 1706 return !!krwp->head_free; 1707 } 1708 1709 static int krc_count(struct kfree_rcu_cpu *krcp) 1710 { 1711 int sum = atomic_read(&krcp->head_count); 1712 int i; 1713 1714 for (i = 0; i < FREE_N_CHANNELS; i++) 1715 sum += atomic_read(&krcp->bulk_count[i]); 1716 1717 return sum; 1718 } 1719 1720 static void 1721 __schedule_delayed_monitor_work(struct kfree_rcu_cpu *krcp) 1722 { 1723 long delay, delay_left; 1724 1725 delay = krc_count(krcp) >= KVFREE_BULK_MAX_ENTR ? 1:KFREE_DRAIN_JIFFIES; 1726 if (delayed_work_pending(&krcp->monitor_work)) { 1727 delay_left = krcp->monitor_work.timer.expires - jiffies; 1728 if (delay < delay_left) 1729 mod_delayed_work(rcu_reclaim_wq, &krcp->monitor_work, delay); 1730 return; 1731 } 1732 queue_delayed_work(rcu_reclaim_wq, &krcp->monitor_work, delay); 1733 } 1734 1735 static void 1736 schedule_delayed_monitor_work(struct kfree_rcu_cpu *krcp) 1737 { 1738 unsigned long flags; 1739 1740 raw_spin_lock_irqsave(&krcp->lock, flags); 1741 __schedule_delayed_monitor_work(krcp); 1742 raw_spin_unlock_irqrestore(&krcp->lock, flags); 1743 } 1744 1745 static void 1746 kvfree_rcu_drain_ready(struct kfree_rcu_cpu *krcp) 1747 { 1748 struct list_head bulk_ready[FREE_N_CHANNELS]; 1749 struct kvfree_rcu_bulk_data *bnode, *n; 1750 struct kvfree_rcu_head *head_ready = NULL; 1751 unsigned long flags; 1752 int i; 1753 1754 raw_spin_lock_irqsave(&krcp->lock, flags); 1755 for (i = 0; i < FREE_N_CHANNELS; i++) { 1756 INIT_LIST_HEAD(&bulk_ready[i]); 1757 1758 list_for_each_entry_safe_reverse(bnode, n, &krcp->bulk_head[i], list) { 1759 if (!poll_state_synchronize_rcu_full(&bnode->gp_snap)) 1760 break; 1761 1762 atomic_sub(bnode->nr_records, &krcp->bulk_count[i]); 1763 list_move(&bnode->list, &bulk_ready[i]); 1764 } 1765 } 1766 1767 if (krcp->head && poll_state_synchronize_rcu(krcp->head_gp_snap)) { 1768 head_ready = krcp->head; 1769 atomic_set(&krcp->head_count, 0); 1770 WRITE_ONCE(krcp->head, NULL); 1771 } 1772 raw_spin_unlock_irqrestore(&krcp->lock, flags); 1773 1774 for (i = 0; i < FREE_N_CHANNELS; i++) { 1775 list_for_each_entry_safe(bnode, n, &bulk_ready[i], list) 1776 kvfree_rcu_bulk(krcp, bnode, i); 1777 } 1778 1779 if (head_ready) 1780 kvfree_rcu_list(head_ready); 1781 } 1782 1783 /* 1784 * Return: %true if a work is queued, %false otherwise. 1785 */ 1786 static bool 1787 kvfree_rcu_queue_batch(struct kfree_rcu_cpu *krcp) 1788 { 1789 unsigned long flags; 1790 bool queued = false; 1791 int i, j; 1792 1793 raw_spin_lock_irqsave(&krcp->lock, flags); 1794 1795 // Attempt to start a new batch. 1796 for (i = 0; i < KFREE_N_BATCHES; i++) { 1797 struct kfree_rcu_cpu_work *krwp = &(krcp->krw_arr[i]); 1798 1799 // Try to detach bulk_head or head and attach it, only when 1800 // all channels are free. Any channel is not free means at krwp 1801 // there is on-going rcu work to handle krwp's free business. 1802 if (need_wait_for_krwp_work(krwp)) 1803 continue; 1804 1805 // kvfree_rcu_drain_ready() might handle this krcp, if so give up. 1806 if (need_offload_krc(krcp)) { 1807 // Channel 1 corresponds to the SLAB-pointer bulk path. 1808 // Channel 2 corresponds to vmalloc-pointer bulk path. 1809 for (j = 0; j < FREE_N_CHANNELS; j++) { 1810 if (list_empty(&krwp->bulk_head_free[j])) { 1811 atomic_set(&krcp->bulk_count[j], 0); 1812 list_replace_init(&krcp->bulk_head[j], 1813 &krwp->bulk_head_free[j]); 1814 } 1815 } 1816 1817 // Channel 3 corresponds to both SLAB and vmalloc 1818 // objects queued on the linked list. 1819 if (!krwp->head_free) { 1820 krwp->head_free = krcp->head; 1821 get_state_synchronize_rcu_full(&krwp->head_free_gp_snap); 1822 atomic_set(&krcp->head_count, 0); 1823 WRITE_ONCE(krcp->head, NULL); 1824 } 1825 1826 // One work is per one batch, so there are three 1827 // "free channels", the batch can handle. Break 1828 // the loop since it is done with this CPU thus 1829 // queuing an RCU work is _always_ success here. 1830 queued = queue_rcu_work(rcu_reclaim_wq, &krwp->rcu_work); 1831 WARN_ON_ONCE(!queued); 1832 break; 1833 } 1834 } 1835 1836 raw_spin_unlock_irqrestore(&krcp->lock, flags); 1837 return queued; 1838 } 1839 1840 /* 1841 * This function is invoked after the KFREE_DRAIN_JIFFIES timeout. 1842 */ 1843 static void kfree_rcu_monitor(struct work_struct *work) 1844 { 1845 struct kfree_rcu_cpu *krcp = container_of(work, 1846 struct kfree_rcu_cpu, monitor_work.work); 1847 1848 // Drain ready for reclaim. 1849 kvfree_rcu_drain_ready(krcp); 1850 1851 // Queue a batch for a rest. 1852 kvfree_rcu_queue_batch(krcp); 1853 1854 // If there is nothing to detach, it means that our job is 1855 // successfully done here. In case of having at least one 1856 // of the channels that is still busy we should rearm the 1857 // work to repeat an attempt. Because previous batches are 1858 // still in progress. 1859 if (need_offload_krc(krcp)) 1860 schedule_delayed_monitor_work(krcp); 1861 } 1862 1863 static void fill_page_cache_func(struct work_struct *work) 1864 { 1865 struct kvfree_rcu_bulk_data *bnode; 1866 struct kfree_rcu_cpu *krcp = 1867 container_of(work, struct kfree_rcu_cpu, 1868 page_cache_work.work); 1869 unsigned long flags; 1870 int nr_pages; 1871 bool pushed; 1872 int i; 1873 1874 nr_pages = atomic_read(&krcp->backoff_page_cache_fill) ? 1875 1 : rcu_min_cached_objs; 1876 1877 for (i = READ_ONCE(krcp->nr_bkv_objs); i < nr_pages; i++) { 1878 bnode = (struct kvfree_rcu_bulk_data *) 1879 __get_free_page(GFP_KERNEL | __GFP_NORETRY | __GFP_NOMEMALLOC | __GFP_NOWARN); 1880 1881 if (!bnode) 1882 break; 1883 1884 raw_spin_lock_irqsave(&krcp->lock, flags); 1885 pushed = put_cached_bnode(krcp, bnode); 1886 raw_spin_unlock_irqrestore(&krcp->lock, flags); 1887 1888 if (!pushed) { 1889 free_page((unsigned long) bnode); 1890 break; 1891 } 1892 } 1893 1894 atomic_set(&krcp->work_in_progress, 0); 1895 atomic_set(&krcp->backoff_page_cache_fill, 0); 1896 } 1897 1898 // Record ptr in a page managed by krcp, with the pre-krc_this_cpu_lock() 1899 // state specified by flags. If can_alloc is true, the caller must 1900 // be schedulable and not be holding any locks or mutexes that might be 1901 // acquired by the memory allocator or anything that it might invoke. 1902 // Returns true if ptr was successfully recorded, else the caller must 1903 // use a fallback. 1904 static inline bool 1905 add_ptr_to_bulk_krc_lock(struct kfree_rcu_cpu **krcp, 1906 unsigned long *flags, void *ptr, bool can_alloc) 1907 { 1908 struct kvfree_rcu_bulk_data *bnode; 1909 int idx; 1910 1911 *krcp = krc_this_cpu_lock(flags); 1912 if (unlikely(!(*krcp)->initialized)) 1913 return false; 1914 1915 idx = !!is_vmalloc_addr(ptr); 1916 bnode = list_first_entry_or_null(&(*krcp)->bulk_head[idx], 1917 struct kvfree_rcu_bulk_data, list); 1918 1919 /* Check if a new block is required. */ 1920 if (!bnode || bnode->nr_records == KVFREE_BULK_MAX_ENTR) { 1921 bnode = get_cached_bnode(*krcp); 1922 if (!bnode && can_alloc) { 1923 krc_this_cpu_unlock(*krcp, *flags); 1924 1925 // __GFP_NORETRY - allows a light-weight direct reclaim 1926 // what is OK from minimizing of fallback hitting point of 1927 // view. Apart of that it forbids any OOM invoking what is 1928 // also beneficial since we are about to release memory soon. 1929 // 1930 // __GFP_NOMEMALLOC - prevents from consuming of all the 1931 // memory reserves. Please note we have a fallback path. 1932 // 1933 // __GFP_NOWARN - it is supposed that an allocation can 1934 // be failed under low memory or high memory pressure 1935 // scenarios. 1936 bnode = (struct kvfree_rcu_bulk_data *) 1937 __get_free_page(GFP_KERNEL | __GFP_NORETRY | __GFP_NOMEMALLOC | __GFP_NOWARN); 1938 raw_spin_lock_irqsave(&(*krcp)->lock, *flags); 1939 } 1940 1941 if (!bnode) 1942 return false; 1943 1944 // Initialize the new block and attach it. 1945 bnode->nr_records = 0; 1946 list_add(&bnode->list, &(*krcp)->bulk_head[idx]); 1947 } 1948 1949 // Finally insert and update the GP for this page. 1950 bnode->nr_records++; 1951 bnode->records[bnode->nr_records - 1] = ptr; 1952 get_state_synchronize_rcu_full(&bnode->gp_snap); 1953 atomic_inc(&(*krcp)->bulk_count[idx]); 1954 1955 return true; 1956 } 1957 1958 static enum hrtimer_restart 1959 schedule_page_work_fn(struct hrtimer *t) 1960 { 1961 struct kfree_rcu_cpu *krcp = 1962 container_of(t, struct kfree_rcu_cpu, hrtimer); 1963 1964 queue_delayed_work(system_highpri_wq, &krcp->page_cache_work, 0); 1965 return HRTIMER_NORESTART; 1966 } 1967 1968 static void 1969 run_page_cache_worker(struct kfree_rcu_cpu *krcp) 1970 { 1971 // If cache disabled, bail out. 1972 if (!rcu_min_cached_objs) 1973 return; 1974 1975 if (rcu_scheduler_active == RCU_SCHEDULER_RUNNING && 1976 !atomic_xchg(&krcp->work_in_progress, 1)) { 1977 if (atomic_read(&krcp->backoff_page_cache_fill)) { 1978 queue_delayed_work(rcu_reclaim_wq, 1979 &krcp->page_cache_work, 1980 msecs_to_jiffies(rcu_delay_page_cache_fill_msec)); 1981 } else { 1982 hrtimer_setup(&krcp->hrtimer, schedule_page_work_fn, CLOCK_MONOTONIC, 1983 HRTIMER_MODE_REL); 1984 hrtimer_start(&krcp->hrtimer, 0, HRTIMER_MODE_REL); 1985 } 1986 } 1987 } 1988 1989 void __init kfree_rcu_scheduler_running(void) 1990 { 1991 int cpu; 1992 1993 for_each_possible_cpu(cpu) { 1994 struct kfree_rcu_cpu *krcp = per_cpu_ptr(&krc, cpu); 1995 1996 if (need_offload_krc(krcp)) 1997 schedule_delayed_monitor_work(krcp); 1998 } 1999 } 2000 2001 /* 2002 * Queue a request for lazy invocation of the appropriate free routine 2003 * after a grace period. Please note that three paths are maintained, 2004 * two for the common case using arrays of pointers and a third one that 2005 * is used only when the main paths cannot be used, for example, due to 2006 * memory pressure. 2007 * 2008 * Each kvfree_call_rcu() request is added to a batch. The batch will be drained 2009 * every KFREE_DRAIN_JIFFIES number of jiffies. All the objects in the batch will 2010 * be free'd in workqueue context. This allows us to: batch requests together to 2011 * reduce the number of grace periods during heavy kfree_rcu()/kvfree_rcu() load. 2012 */ 2013 void kvfree_call_rcu(struct kvfree_rcu_head *head, void *ptr) 2014 { 2015 unsigned long flags; 2016 struct kfree_rcu_cpu *krcp; 2017 bool success; 2018 2019 /* 2020 * Please note there is a limitation for the head-less 2021 * variant, that is why there is a clear rule for such 2022 * objects: it can be used from might_sleep() context 2023 * only. For other places please embed an rcu_head to 2024 * your data. 2025 */ 2026 if (!head) 2027 might_sleep(); 2028 2029 /* 2030 * kvfree_rcu() is called by set_cpus_allowed_force() with 2031 * task_struct::pi_lock acquired. On PREEMPT_RT the local_trylock() 2032 * usage below will acquire the waitlock which must be avoided. 2033 * Therefore avoid it on PREEMPT_RT. 2034 */ 2035 if (!IS_ENABLED(CONFIG_PREEMPT_RT) && kfree_rcu_sheaf(ptr)) 2036 return; 2037 2038 // Queue the object but don't yet schedule the batch. 2039 if (debug_rcu_head_queue(ptr)) { 2040 // Probable double kfree_rcu(), just leak. 2041 WARN_ONCE(1, "%s(): Double-freed call. rcu_head %p\n", 2042 __func__, head); 2043 2044 // Mark as success and leave. 2045 return; 2046 } 2047 2048 kasan_record_aux_stack(ptr); 2049 success = add_ptr_to_bulk_krc_lock(&krcp, &flags, ptr, !head); 2050 if (!success) { 2051 run_page_cache_worker(krcp); 2052 2053 if (head == NULL) 2054 // Inline if kvfree_rcu(one_arg) call. 2055 goto unlock_return; 2056 2057 head->next = krcp->head; 2058 WRITE_ONCE(krcp->head, head); 2059 atomic_inc(&krcp->head_count); 2060 2061 // Take a snapshot for this krcp. 2062 krcp->head_gp_snap = get_state_synchronize_rcu(); 2063 success = true; 2064 } 2065 2066 /* 2067 * The kvfree_rcu() caller considers the pointer freed at this point 2068 * and likely removes any references to it. Since the actual slab 2069 * freeing (and kmemleak_free()) is deferred, tell kmemleak to ignore 2070 * this object (no scanning or false positives reporting). 2071 */ 2072 kmemleak_ignore(ptr); 2073 2074 // Set timer to drain after KFREE_DRAIN_JIFFIES. 2075 if (rcu_scheduler_active == RCU_SCHEDULER_RUNNING) 2076 __schedule_delayed_monitor_work(krcp); 2077 2078 unlock_return: 2079 krc_this_cpu_unlock(krcp, flags); 2080 2081 /* 2082 * Inline kvfree() after synchronize_rcu(). We can do 2083 * it from might_sleep() context only, so the current 2084 * CPU can pass the QS state. 2085 */ 2086 if (!success) { 2087 debug_rcu_head_unqueue((struct rcu_head *) ptr); 2088 synchronize_rcu(); 2089 kvfree(ptr); 2090 } 2091 } 2092 EXPORT_SYMBOL_GPL(kvfree_call_rcu); 2093 2094 static inline void __kvfree_rcu_barrier(void) 2095 { 2096 struct kfree_rcu_cpu_work *krwp; 2097 struct kfree_rcu_cpu *krcp; 2098 bool queued; 2099 int i, cpu; 2100 2101 /* 2102 * Firstly we detach objects and queue them over an RCU-batch 2103 * for all CPUs. Finally queued works are flushed for each CPU. 2104 * 2105 * Please note. If there are outstanding batches for a particular 2106 * CPU, those have to be finished first following by queuing a new. 2107 */ 2108 for_each_possible_cpu(cpu) { 2109 krcp = per_cpu_ptr(&krc, cpu); 2110 2111 /* 2112 * Check if this CPU has any objects which have been queued for a 2113 * new GP completion. If not(means nothing to detach), we are done 2114 * with it. If any batch is pending/running for this "krcp", below 2115 * per-cpu flush_rcu_work() waits its completion(see last step). 2116 */ 2117 if (!need_offload_krc(krcp)) 2118 continue; 2119 2120 while (1) { 2121 /* 2122 * If we are not able to queue a new RCU work it means: 2123 * - batches for this CPU are still in flight which should 2124 * be flushed first and then repeat; 2125 * - no objects to detach, because of concurrency. 2126 */ 2127 queued = kvfree_rcu_queue_batch(krcp); 2128 2129 /* 2130 * Bail out, if there is no need to offload this "krcp" 2131 * anymore. As noted earlier it can run concurrently. 2132 */ 2133 if (queued || !need_offload_krc(krcp)) 2134 break; 2135 2136 /* There are ongoing batches. */ 2137 for (i = 0; i < KFREE_N_BATCHES; i++) { 2138 krwp = &(krcp->krw_arr[i]); 2139 flush_rcu_work(&krwp->rcu_work); 2140 } 2141 } 2142 } 2143 2144 /* 2145 * Now we guarantee that all objects are flushed. 2146 */ 2147 for_each_possible_cpu(cpu) { 2148 krcp = per_cpu_ptr(&krc, cpu); 2149 2150 /* 2151 * A monitor work can drain ready to reclaim objects 2152 * directly. Wait its completion if running or pending. 2153 */ 2154 cancel_delayed_work_sync(&krcp->monitor_work); 2155 2156 for (i = 0; i < KFREE_N_BATCHES; i++) { 2157 krwp = &(krcp->krw_arr[i]); 2158 flush_rcu_work(&krwp->rcu_work); 2159 } 2160 } 2161 } 2162 2163 /** 2164 * kvfree_rcu_barrier - Wait until all in-flight kvfree_rcu() complete. 2165 * 2166 * Note that a single argument of kvfree_rcu() call has a slow path that 2167 * triggers synchronize_rcu() following by freeing a pointer. It is done 2168 * before the return from the function. Therefore for any single-argument 2169 * call that will result in a kfree() to a cache that is to be destroyed 2170 * during module exit, it is developer's responsibility to ensure that all 2171 * such calls have returned before the call to kmem_cache_destroy(). 2172 */ 2173 void kvfree_rcu_barrier(void) 2174 { 2175 flush_all_rcu_sheaves(); 2176 __kvfree_rcu_barrier(); 2177 } 2178 2179 /** 2180 * kvfree_rcu_barrier_on_cache - Wait for in-flight kvfree_rcu() calls on a 2181 * specific slab cache. 2182 * @s: slab cache to wait for 2183 * 2184 * See the description of kvfree_rcu_barrier() for details. 2185 */ 2186 void kvfree_rcu_barrier_on_cache(struct kmem_cache *s) 2187 { 2188 /* kfree_rcu_nolock() might have deferred frees even without sheaves */ 2189 deferred_work_barrier(); 2190 2191 if (cache_has_sheaves(s)) { 2192 cpus_read_lock(); 2193 flush_rcu_sheaves_on_cache(s); 2194 cpus_read_unlock(); 2195 } 2196 2197 rcu_barrier(); 2198 __kvfree_rcu_barrier(); 2199 } 2200 2201 static unsigned long 2202 kfree_rcu_shrink_count(struct shrinker *shrink, struct shrink_control *sc) 2203 { 2204 int cpu; 2205 unsigned long count = 0; 2206 2207 /* Snapshot count of all CPUs */ 2208 for_each_possible_cpu(cpu) { 2209 struct kfree_rcu_cpu *krcp = per_cpu_ptr(&krc, cpu); 2210 2211 count += krc_count(krcp); 2212 count += READ_ONCE(krcp->nr_bkv_objs); 2213 atomic_set(&krcp->backoff_page_cache_fill, 1); 2214 } 2215 2216 return count == 0 ? SHRINK_EMPTY : count; 2217 } 2218 2219 static unsigned long 2220 kfree_rcu_shrink_scan(struct shrinker *shrink, struct shrink_control *sc) 2221 { 2222 int cpu, freed = 0; 2223 2224 for_each_possible_cpu(cpu) { 2225 int count; 2226 struct kfree_rcu_cpu *krcp = per_cpu_ptr(&krc, cpu); 2227 2228 count = krc_count(krcp); 2229 count += drain_page_cache(krcp); 2230 kfree_rcu_monitor(&krcp->monitor_work.work); 2231 2232 sc->nr_to_scan -= count; 2233 freed += count; 2234 2235 if (sc->nr_to_scan <= 0) 2236 break; 2237 } 2238 2239 return freed == 0 ? SHRINK_STOP : freed; 2240 } 2241 2242 void __init kvfree_rcu_init(void) 2243 { 2244 int cpu; 2245 int i, j; 2246 struct shrinker *kfree_rcu_shrinker; 2247 2248 rcu_reclaim_wq = alloc_workqueue("kvfree_rcu_reclaim", 2249 WQ_UNBOUND | WQ_MEM_RECLAIM, 0); 2250 WARN_ON(!rcu_reclaim_wq); 2251 2252 /* Clamp it to [0:100] seconds interval. */ 2253 if (rcu_delay_page_cache_fill_msec < 0 || 2254 rcu_delay_page_cache_fill_msec > 100 * MSEC_PER_SEC) { 2255 2256 rcu_delay_page_cache_fill_msec = 2257 clamp(rcu_delay_page_cache_fill_msec, 0, 2258 (int) (100 * MSEC_PER_SEC)); 2259 2260 pr_info("Adjusting rcutree.rcu_delay_page_cache_fill_msec to %d ms.\n", 2261 rcu_delay_page_cache_fill_msec); 2262 } 2263 2264 for_each_possible_cpu(cpu) { 2265 struct kfree_rcu_cpu *krcp = per_cpu_ptr(&krc, cpu); 2266 2267 for (i = 0; i < KFREE_N_BATCHES; i++) { 2268 INIT_RCU_WORK(&krcp->krw_arr[i].rcu_work, kfree_rcu_work); 2269 krcp->krw_arr[i].krcp = krcp; 2270 2271 for (j = 0; j < FREE_N_CHANNELS; j++) 2272 INIT_LIST_HEAD(&krcp->krw_arr[i].bulk_head_free[j]); 2273 } 2274 2275 for (i = 0; i < FREE_N_CHANNELS; i++) 2276 INIT_LIST_HEAD(&krcp->bulk_head[i]); 2277 2278 INIT_DELAYED_WORK(&krcp->monitor_work, kfree_rcu_monitor); 2279 INIT_DELAYED_WORK(&krcp->page_cache_work, fill_page_cache_func); 2280 krcp->initialized = true; 2281 } 2282 2283 kfree_rcu_shrinker = shrinker_alloc(0, "slab-kvfree-rcu"); 2284 if (!kfree_rcu_shrinker) { 2285 pr_err("Failed to allocate kfree_rcu() shrinker!\n"); 2286 return; 2287 } 2288 2289 kfree_rcu_shrinker->count_objects = kfree_rcu_shrink_count; 2290 kfree_rcu_shrinker->scan_objects = kfree_rcu_shrink_scan; 2291 2292 shrinker_register(kfree_rcu_shrinker); 2293 } 2294 2295 #endif /* CONFIG_KVFREE_RCU_BATCHED */ 2296