1 // SPDX-License-Identifier: GPL-2.0-only 2 /* 3 * linux/kernel/softirq.c 4 * 5 * Copyright (C) 1992 Linus Torvalds 6 * 7 * Rewritten. Old one was good in 2.2, but in 2.3 it was immoral. --ANK (990903) 8 */ 9 10 #define pr_fmt(fmt) KBUILD_MODNAME ": " fmt 11 12 #define INSTANTIATE_EXPORTED_INTERRUPT_DISABLE 13 #include <linux/export.h> 14 #include <linux/kernel_stat.h> 15 #include <linux/interrupt.h> 16 #include <linux/init.h> 17 #include <linux/local_lock.h> 18 #include <linux/mm.h> 19 #include <linux/notifier.h> 20 #include <linux/percpu.h> 21 #include <linux/cpu.h> 22 #include <linux/freezer.h> 23 #include <linux/kthread.h> 24 #include <linux/rcupdate.h> 25 #include <linux/ftrace.h> 26 #include <linux/smp.h> 27 #include <linux/smpboot.h> 28 #include <linux/tick.h> 29 #include <linux/irq.h> 30 #include <linux/wait_bit.h> 31 #include <linux/workqueue.h> 32 33 #include <asm/softirq_stack.h> 34 35 #define CREATE_TRACE_POINTS 36 #include <trace/events/irq.h> 37 38 /* 39 - No shared variables, all the data are CPU local. 40 - If a softirq needs serialization, let it serialize itself 41 by its own spinlocks. 42 - Even if softirq is serialized, only local cpu is marked for 43 execution. Hence, we get something sort of weak cpu binding. 44 Though it is still not clear, will it result in better locality 45 or will not. 46 47 Examples: 48 - NET RX softirq. It is multithreaded and does not require 49 any global serialization. 50 - NET TX softirq. It kicks software netdevice queues, hence 51 it is logically serialized per device, but this serialization 52 is invisible to common code. 53 - Tasklets: serialized wrt itself. 54 */ 55 56 #ifndef __ARCH_IRQ_STAT 57 DEFINE_PER_CPU_ALIGNED(irq_cpustat_t, irq_stat); 58 EXPORT_PER_CPU_SYMBOL(irq_stat); 59 #endif 60 61 static struct softirq_action softirq_vec[NR_SOFTIRQS] __cacheline_aligned_in_smp; 62 63 DEFINE_PER_CPU(struct task_struct *, ksoftirqd); 64 65 const char * const softirq_to_name[NR_SOFTIRQS] = { 66 "HI", "TIMER", "NET_TX", "NET_RX", "BLOCK", "IRQ_POLL", 67 "TASKLET", "SCHED", "HRTIMER", "RCU" 68 }; 69 70 /* 71 * we cannot loop indefinitely here to avoid userspace starvation, 72 * but we also don't want to introduce a worst case 1/HZ latency 73 * to the pending events, so lets the scheduler to balance 74 * the softirq load for us. 75 */ 76 static void wakeup_softirqd(void) 77 { 78 /* Interrupts are disabled: no need to stop preemption */ 79 struct task_struct *tsk = __this_cpu_read(ksoftirqd); 80 81 if (tsk) 82 wake_up_process(tsk); 83 } 84 85 #ifdef CONFIG_TRACE_IRQFLAGS 86 DEFINE_PER_CPU(int, hardirqs_enabled); 87 DEFINE_PER_CPU(int, hardirq_context); 88 EXPORT_PER_CPU_SYMBOL_GPL(hardirqs_enabled); 89 EXPORT_PER_CPU_SYMBOL_GPL(hardirq_context); 90 #endif 91 92 DEFINE_PER_CPU(unsigned long, local_interrupt_disable_state); 93 94 void _local_interrupt_save_state(unsigned long flags) 95 { 96 __local_interrupt_save_state(flags); 97 } 98 EXPORT_SYMBOL(_local_interrupt_save_state); 99 100 void _local_interrupt_enable(void) 101 { 102 __local_interrupt_enable(); 103 } 104 EXPORT_SYMBOL(_local_interrupt_enable); 105 106 #ifndef CONFIG_HAS_SEPARATE_PREEMPT_RESCHED_BITS 107 /* 108 * Any 32bit architecture that still cares about performance should 109 * probably ensure this is near preempt_count. 110 */ 111 DEFINE_PER_CPU(unsigned int, nmi_nesting); 112 #endif 113 114 /* 115 * SOFTIRQ_OFFSET usage: 116 * 117 * On !RT kernels 'count' is the preempt counter, on RT kernels this applies 118 * to a per CPU counter and to task::softirqs_disabled_cnt. 119 * 120 * - count is changed by SOFTIRQ_OFFSET on entering or leaving softirq 121 * processing. 122 * 123 * - count is changed by SOFTIRQ_DISABLE_OFFSET (= 2 * SOFTIRQ_OFFSET) 124 * on local_bh_disable or local_bh_enable. 125 * 126 * This lets us distinguish between whether we are currently processing 127 * softirq and whether we just have bh disabled. 128 */ 129 #ifdef CONFIG_PREEMPT_RT 130 131 /* 132 * RT accounts for BH disabled sections in task::softirqs_disabled_cnt and 133 * also in per CPU softirq_ctrl::cnt. This is necessary to allow tasks in a 134 * softirq disabled section to be preempted. 135 * 136 * The per task counter is used for softirq_count(), in_softirq() and 137 * in_serving_softirqs() because these counts are only valid when the task 138 * holding softirq_ctrl::lock is running. 139 * 140 * The per CPU counter prevents pointless wakeups of ksoftirqd in case that 141 * the task which is in a softirq disabled section is preempted or blocks. 142 */ 143 struct softirq_ctrl { 144 local_lock_t lock; 145 int cnt; 146 }; 147 148 static DEFINE_PER_CPU(struct softirq_ctrl, softirq_ctrl) = { 149 .lock = INIT_LOCAL_LOCK(softirq_ctrl.lock), 150 }; 151 152 #ifdef CONFIG_DEBUG_LOCK_ALLOC 153 static struct lock_class_key bh_lock_key; 154 struct lockdep_map bh_lock_map = { 155 .name = "local_bh", 156 .key = &bh_lock_key, 157 .wait_type_outer = LD_WAIT_FREE, 158 .wait_type_inner = LD_WAIT_CONFIG, /* PREEMPT_RT makes BH preemptible. */ 159 .lock_type = LD_LOCK_PERCPU, 160 }; 161 EXPORT_SYMBOL_GPL(bh_lock_map); 162 #endif 163 164 /** 165 * local_bh_blocked() - Check for idle whether BH processing is blocked 166 * 167 * Returns false if the per CPU softirq::cnt is 0 otherwise true. 168 * 169 * This is invoked from the idle task to guard against false positive 170 * softirq pending warnings, which would happen when the task which holds 171 * softirq_ctrl::lock was the only running task on the CPU and blocks on 172 * some other lock. 173 */ 174 bool local_bh_blocked(void) 175 { 176 return __this_cpu_read(softirq_ctrl.cnt) != 0; 177 } 178 179 void __local_bh_disable_ip(unsigned long ip, unsigned int cnt) 180 { 181 unsigned long flags; 182 int newcnt; 183 184 WARN_ON_ONCE(in_hardirq()); 185 186 lock_map_acquire_read(&bh_lock_map); 187 188 /* First entry of a task into a BH disabled section? */ 189 if (!current->softirq_disable_cnt) { 190 if (preemptible()) { 191 if (IS_ENABLED(CONFIG_PREEMPT_RT_NEEDS_BH_LOCK)) 192 local_lock(&softirq_ctrl.lock); 193 else 194 migrate_disable(); 195 196 /* Required to meet the RCU bottomhalf requirements. */ 197 rcu_read_lock(); 198 } else { 199 DEBUG_LOCKS_WARN_ON(this_cpu_read(softirq_ctrl.cnt)); 200 } 201 } 202 203 /* 204 * Track the per CPU softirq disabled state. On RT this is per CPU 205 * state to allow preemption of bottom half disabled sections. 206 */ 207 if (IS_ENABLED(CONFIG_PREEMPT_RT_NEEDS_BH_LOCK)) { 208 newcnt = this_cpu_add_return(softirq_ctrl.cnt, cnt); 209 /* 210 * Reflect the result in the task state to prevent recursion on the 211 * local lock and to make softirq_count() & al work. 212 */ 213 current->softirq_disable_cnt = newcnt; 214 215 if (IS_ENABLED(CONFIG_TRACE_IRQFLAGS) && newcnt == cnt) { 216 raw_local_irq_save(flags); 217 lockdep_softirqs_off(ip); 218 raw_local_irq_restore(flags); 219 } 220 } else { 221 bool sirq_dis = false; 222 223 if (!current->softirq_disable_cnt) 224 sirq_dis = true; 225 226 this_cpu_add(softirq_ctrl.cnt, cnt); 227 current->softirq_disable_cnt += cnt; 228 WARN_ON_ONCE(current->softirq_disable_cnt < 0); 229 230 if (IS_ENABLED(CONFIG_TRACE_IRQFLAGS) && sirq_dis) { 231 raw_local_irq_save(flags); 232 lockdep_softirqs_off(ip); 233 raw_local_irq_restore(flags); 234 } 235 } 236 } 237 EXPORT_SYMBOL(__local_bh_disable_ip); 238 239 static void __local_bh_enable(unsigned int cnt, bool unlock) 240 { 241 unsigned long flags; 242 bool sirq_en = false; 243 int newcnt; 244 245 if (IS_ENABLED(CONFIG_PREEMPT_RT_NEEDS_BH_LOCK)) { 246 DEBUG_LOCKS_WARN_ON(current->softirq_disable_cnt != 247 this_cpu_read(softirq_ctrl.cnt)); 248 if (softirq_count() == cnt) 249 sirq_en = true; 250 } else { 251 if (current->softirq_disable_cnt == cnt) 252 sirq_en = true; 253 } 254 255 if (IS_ENABLED(CONFIG_TRACE_IRQFLAGS) && sirq_en) { 256 raw_local_irq_save(flags); 257 lockdep_softirqs_on(_RET_IP_); 258 raw_local_irq_restore(flags); 259 } 260 261 if (IS_ENABLED(CONFIG_PREEMPT_RT_NEEDS_BH_LOCK)) { 262 newcnt = this_cpu_sub_return(softirq_ctrl.cnt, cnt); 263 current->softirq_disable_cnt = newcnt; 264 265 if (!newcnt && unlock) { 266 rcu_read_unlock(); 267 local_unlock(&softirq_ctrl.lock); 268 } 269 } else { 270 current->softirq_disable_cnt -= cnt; 271 this_cpu_sub(softirq_ctrl.cnt, cnt); 272 if (unlock && !current->softirq_disable_cnt) { 273 migrate_enable(); 274 rcu_read_unlock(); 275 } else { 276 WARN_ON_ONCE(current->softirq_disable_cnt < 0); 277 } 278 } 279 } 280 281 void __local_bh_enable_ip(unsigned long ip, unsigned int cnt) 282 { 283 bool preempt_on = preemptible(); 284 unsigned long flags; 285 u32 pending; 286 int curcnt; 287 288 WARN_ON_ONCE(in_hardirq()); 289 lockdep_assert_irqs_enabled(); 290 291 lock_map_release(&bh_lock_map); 292 293 local_irq_save(flags); 294 if (IS_ENABLED(CONFIG_PREEMPT_RT_NEEDS_BH_LOCK)) 295 curcnt = this_cpu_read(softirq_ctrl.cnt); 296 else 297 curcnt = current->softirq_disable_cnt; 298 299 /* 300 * If this is not reenabling soft interrupts, no point in trying to 301 * run pending ones. 302 */ 303 if (curcnt != cnt) 304 goto out; 305 306 pending = local_softirq_pending(); 307 if (!pending) 308 goto out; 309 310 /* 311 * If this was called from non preemptible context, wake up the 312 * softirq daemon. 313 */ 314 if (!preempt_on) { 315 wakeup_softirqd(); 316 goto out; 317 } 318 319 /* 320 * Adjust softirq count to SOFTIRQ_OFFSET which makes 321 * in_serving_softirq() become true. 322 */ 323 cnt = SOFTIRQ_OFFSET; 324 __local_bh_enable(cnt, false); 325 __do_softirq(); 326 327 out: 328 __local_bh_enable(cnt, preempt_on); 329 local_irq_restore(flags); 330 } 331 EXPORT_SYMBOL(__local_bh_enable_ip); 332 333 /* 334 * Invoked from ksoftirqd_run() outside of the interrupt disabled section 335 * to acquire the per CPU local lock for reentrancy protection. 336 */ 337 static inline void ksoftirqd_run_begin(void) 338 { 339 __local_bh_disable_ip(_RET_IP_, SOFTIRQ_OFFSET); 340 local_irq_disable(); 341 } 342 343 /* Counterpart to ksoftirqd_run_begin() */ 344 static inline void ksoftirqd_run_end(void) 345 { 346 /* pairs with the lock_map_acquire_read() in ksoftirqd_run_begin() */ 347 lock_map_release(&bh_lock_map); 348 __local_bh_enable(SOFTIRQ_OFFSET, true); 349 WARN_ON_ONCE(in_interrupt()); 350 local_irq_enable(); 351 } 352 353 static inline void softirq_handle_begin(void) { } 354 static inline void softirq_handle_end(void) { } 355 356 static inline bool should_wake_ksoftirqd(void) 357 { 358 return !this_cpu_read(softirq_ctrl.cnt); 359 } 360 361 static inline void invoke_softirq(void) 362 { 363 if (should_wake_ksoftirqd()) 364 wakeup_softirqd(); 365 } 366 367 #define SCHED_SOFTIRQ_MASK BIT(SCHED_SOFTIRQ) 368 369 /* 370 * flush_smp_call_function_queue() can raise a soft interrupt in a function 371 * call. On RT kernels this is undesired and the only known functionalities 372 * are in the block layer which is disabled on RT, and in the scheduler for 373 * idle load balancing. If soft interrupts get raised which haven't been 374 * raised before the flush, warn if it is not a SCHED_SOFTIRQ so it can be 375 * investigated. 376 */ 377 void do_softirq_post_smp_call_flush(unsigned int was_pending) 378 { 379 unsigned int is_pending = local_softirq_pending(); 380 381 if (unlikely(was_pending != is_pending)) { 382 WARN_ON_ONCE(was_pending != (is_pending & ~SCHED_SOFTIRQ_MASK)); 383 invoke_softirq(); 384 } 385 } 386 387 #else /* CONFIG_PREEMPT_RT */ 388 389 /* 390 * This one is for softirq.c-internal use, where hardirqs are disabled 391 * legitimately: 392 */ 393 #ifdef CONFIG_TRACE_IRQFLAGS 394 void __local_bh_disable_ip(unsigned long ip, unsigned int cnt) 395 { 396 unsigned long flags; 397 398 WARN_ON_ONCE(in_hardirq()); 399 400 raw_local_irq_save(flags); 401 /* 402 * The preempt tracer hooks into preempt_count_add and will break 403 * lockdep because it calls back into lockdep after SOFTIRQ_OFFSET 404 * is set and before current->softirq_enabled is cleared. 405 * We must manually increment preempt_count here and manually 406 * call the trace_preempt_off later. 407 */ 408 __preempt_count_add(cnt); 409 /* 410 * Were softirqs turned off above: 411 */ 412 if (softirq_count() == (cnt & SOFTIRQ_MASK)) 413 lockdep_softirqs_off(ip); 414 raw_local_irq_restore(flags); 415 416 if (preempt_count() == cnt) { 417 #ifdef CONFIG_DEBUG_PREEMPT 418 current->preempt_disable_ip = get_lock_parent_ip(); 419 #endif 420 trace_preempt_off(CALLER_ADDR0, get_lock_parent_ip()); 421 } 422 } 423 EXPORT_SYMBOL(__local_bh_disable_ip); 424 #endif /* CONFIG_TRACE_IRQFLAGS */ 425 426 static void __local_bh_enable(unsigned int cnt) 427 { 428 lockdep_assert_irqs_disabled(); 429 430 if (preempt_count() == cnt) 431 trace_preempt_on(CALLER_ADDR0, get_lock_parent_ip()); 432 433 if (softirq_count() == (cnt & SOFTIRQ_MASK)) 434 lockdep_softirqs_on(_RET_IP_); 435 436 __preempt_count_sub(cnt); 437 } 438 439 /* 440 * Special-case - softirqs can safely be enabled by __do_softirq(), 441 * without processing still-pending softirqs: 442 */ 443 void _local_bh_enable(void) 444 { 445 WARN_ON_ONCE(in_hardirq()); 446 __local_bh_enable(SOFTIRQ_DISABLE_OFFSET); 447 } 448 EXPORT_SYMBOL(_local_bh_enable); 449 450 void __local_bh_enable_ip(unsigned long ip, unsigned int cnt) 451 { 452 WARN_ON_ONCE(in_hardirq()); 453 lockdep_assert_irqs_enabled(); 454 #ifdef CONFIG_TRACE_IRQFLAGS 455 local_irq_disable(); 456 #endif 457 /* 458 * Are softirqs going to be turned on now: 459 */ 460 if (softirq_count() == SOFTIRQ_DISABLE_OFFSET) 461 lockdep_softirqs_on(ip); 462 /* 463 * Keep preemption disabled until we are done with 464 * softirq processing: 465 */ 466 __preempt_count_sub(cnt - 1); 467 468 if (unlikely(!in_interrupt() && local_softirq_pending())) { 469 /* 470 * Run softirq if any pending. And do it in its own stack 471 * as we may be calling this deep in a task call stack already. 472 */ 473 do_softirq(); 474 } 475 476 preempt_count_dec(); 477 #ifdef CONFIG_TRACE_IRQFLAGS 478 local_irq_enable(); 479 #endif 480 preempt_check_resched(); 481 } 482 EXPORT_SYMBOL(__local_bh_enable_ip); 483 484 static inline void softirq_handle_begin(void) 485 { 486 __local_bh_disable_ip(_RET_IP_, SOFTIRQ_OFFSET); 487 } 488 489 static inline void softirq_handle_end(void) 490 { 491 __local_bh_enable(SOFTIRQ_OFFSET); 492 WARN_ON_ONCE(in_interrupt()); 493 } 494 495 static inline void ksoftirqd_run_begin(void) 496 { 497 local_irq_disable(); 498 } 499 500 static inline void ksoftirqd_run_end(void) 501 { 502 local_irq_enable(); 503 } 504 505 static inline bool should_wake_ksoftirqd(void) 506 { 507 return true; 508 } 509 510 static inline void invoke_softirq(void) 511 { 512 if (!force_irqthreads() || !__this_cpu_read(ksoftirqd)) { 513 #ifdef CONFIG_HAVE_IRQ_EXIT_ON_IRQ_STACK 514 /* 515 * We can safely execute softirq on the current stack if 516 * it is the irq stack, because it should be near empty 517 * at this stage. 518 */ 519 __do_softirq(); 520 #else 521 /* 522 * Otherwise, irq_exit() is called on the task stack that can 523 * be potentially deep already. So call softirq in its own stack 524 * to prevent from any overrun. 525 */ 526 do_softirq_own_stack(); 527 #endif 528 } else { 529 wakeup_softirqd(); 530 } 531 } 532 533 asmlinkage __visible void do_softirq(void) 534 { 535 __u32 pending; 536 unsigned long flags; 537 538 if (in_interrupt()) 539 return; 540 541 local_irq_save(flags); 542 543 pending = local_softirq_pending(); 544 545 if (pending) 546 do_softirq_own_stack(); 547 548 local_irq_restore(flags); 549 } 550 551 #endif /* !CONFIG_PREEMPT_RT */ 552 553 /* 554 * We restart softirq processing for at most MAX_SOFTIRQ_RESTART times, 555 * but break the loop if need_resched() is set or after 2 ms. 556 * The MAX_SOFTIRQ_TIME provides a nice upper bound in most cases, but in 557 * certain cases, such as stop_machine(), jiffies may cease to 558 * increment and so we need the MAX_SOFTIRQ_RESTART limit as 559 * well to make sure we eventually return from this method. 560 * 561 * These limits have been established via experimentation. 562 * The two things to balance is latency against fairness - 563 * we want to handle softirqs as soon as possible, but they 564 * should not be able to lock up the box. 565 */ 566 #define MAX_SOFTIRQ_TIME msecs_to_jiffies(2) 567 #define MAX_SOFTIRQ_RESTART 10 568 569 #ifdef CONFIG_TRACE_IRQFLAGS 570 /* 571 * When we run softirqs from irq_exit() and thus on the hardirq stack we need 572 * to keep the lockdep irq context tracking as tight as possible in order to 573 * not miss-qualify lock contexts and miss possible deadlocks. 574 */ 575 576 static inline bool lockdep_softirq_start(void) 577 { 578 bool in_hardirq = false; 579 580 if (lockdep_hardirq_context()) { 581 in_hardirq = true; 582 lockdep_hardirq_exit(); 583 } 584 585 lockdep_softirq_enter(); 586 587 return in_hardirq; 588 } 589 590 static inline void lockdep_softirq_end(bool in_hardirq) 591 { 592 lockdep_softirq_exit(); 593 594 if (in_hardirq) 595 lockdep_hardirq_enter(); 596 } 597 #else 598 static inline bool lockdep_softirq_start(void) { return false; } 599 static inline void lockdep_softirq_end(bool in_hardirq) { } 600 #endif 601 602 static void handle_softirqs(bool ksirqd) 603 { 604 unsigned long end = jiffies + MAX_SOFTIRQ_TIME; 605 unsigned long old_flags = current->flags; 606 int max_restart = MAX_SOFTIRQ_RESTART; 607 struct softirq_action *h; 608 bool in_hardirq; 609 __u32 pending; 610 int softirq_bit; 611 612 /* 613 * Mask out PF_MEMALLOC as the current task context is borrowed for the 614 * softirq. A softirq handled, such as network RX, might set PF_MEMALLOC 615 * again if the socket is related to swapping. 616 */ 617 current->flags &= ~PF_MEMALLOC; 618 619 pending = local_softirq_pending(); 620 621 softirq_handle_begin(); 622 in_hardirq = lockdep_softirq_start(); 623 account_softirq_enter(current); 624 625 restart: 626 /* Reset the pending bitmask before enabling irqs */ 627 set_softirq_pending(0); 628 629 local_irq_enable(); 630 631 h = softirq_vec; 632 633 while ((softirq_bit = ffs(pending))) { 634 unsigned int vec_nr; 635 int prev_count; 636 637 h += softirq_bit - 1; 638 639 vec_nr = h - softirq_vec; 640 prev_count = preempt_count(); 641 642 kstat_incr_softirqs_this_cpu(vec_nr); 643 644 trace_softirq_entry(vec_nr); 645 h->action(); 646 trace_softirq_exit(vec_nr); 647 if (unlikely(prev_count != preempt_count())) { 648 pr_err("huh, entered softirq %u %s %p with preempt_count %08x, exited with %08x?\n", 649 vec_nr, softirq_to_name[vec_nr], h->action, 650 prev_count, preempt_count()); 651 preempt_count_set(prev_count); 652 } 653 h++; 654 pending >>= softirq_bit; 655 } 656 657 if (!IS_ENABLED(CONFIG_PREEMPT_RT) && ksirqd) 658 rcu_softirq_qs(); 659 660 local_irq_disable(); 661 662 pending = local_softirq_pending(); 663 if (pending) { 664 if (time_before(jiffies, end) && !need_resched() && 665 --max_restart) 666 goto restart; 667 668 wakeup_softirqd(); 669 } 670 671 account_softirq_exit(current); 672 lockdep_softirq_end(in_hardirq); 673 softirq_handle_end(); 674 current_restore_flags(old_flags, PF_MEMALLOC); 675 } 676 677 asmlinkage __visible void __softirq_entry __do_softirq(void) 678 { 679 handle_softirqs(false); 680 } 681 682 /** 683 * irq_enter_rcu - Enter an interrupt context with RCU watching 684 */ 685 void irq_enter_rcu(void) 686 { 687 __irq_enter_raw(); 688 689 /* 690 * If this is a nested interrupt that hits the exit_to_user_mode_loop 691 * where it has enabled interrupts but before it has hit schedule() we 692 * could have hrtimers in an undefined state. Fix it up here. 693 */ 694 hrtimer_rearm_deferred(); 695 696 if (tick_nohz_full_cpu(smp_processor_id()) || 697 (is_idle_task(current) && (irq_count() == HARDIRQ_OFFSET))) 698 tick_irq_enter(); 699 700 account_hardirq_enter(current); 701 } 702 703 /** 704 * irq_enter - Enter an interrupt context including RCU update 705 */ 706 void irq_enter(void) 707 { 708 ct_irq_enter(); 709 irq_enter_rcu(); 710 } 711 712 static inline void tick_irq_exit(void) 713 { 714 #ifdef CONFIG_NO_HZ_COMMON 715 int cpu = smp_processor_id(); 716 717 /* Make sure that timer wheel updates are propagated */ 718 if ((sched_core_idle_cpu(cpu) && !need_resched()) || tick_nohz_full_cpu(cpu)) { 719 if (!in_hardirq()) 720 tick_nohz_irq_exit(); 721 } 722 #endif 723 } 724 725 #ifdef CONFIG_IRQ_FORCED_THREADING 726 DEFINE_PER_CPU(struct task_struct *, ktimerd); 727 DEFINE_PER_CPU(unsigned long, pending_timer_softirq); 728 729 static void wake_timersd(void) 730 { 731 struct task_struct *tsk = __this_cpu_read(ktimerd); 732 733 if (tsk) 734 wake_up_process(tsk); 735 } 736 737 #else 738 739 static inline void wake_timersd(void) { } 740 741 #endif 742 743 static inline void __irq_exit_rcu(void) 744 { 745 #ifndef __ARCH_IRQ_EXIT_IRQS_DISABLED 746 local_irq_disable(); 747 #else 748 lockdep_assert_irqs_disabled(); 749 #endif 750 account_hardirq_exit(current); 751 preempt_count_sub(HARDIRQ_OFFSET); 752 if (!in_interrupt() && local_softirq_pending()) { 753 /* 754 * If we left hrtimers unarmed, make sure to arm them now, 755 * before enabling interrupts to run softirq. 756 */ 757 hrtimer_rearm_deferred(); 758 invoke_softirq(); 759 } 760 761 if (IS_ENABLED(CONFIG_IRQ_FORCED_THREADING) && force_irqthreads() && 762 local_timers_pending_force_th() && !(in_nmi() | in_hardirq())) 763 wake_timersd(); 764 765 tick_irq_exit(); 766 } 767 768 /** 769 * irq_exit_rcu() - Exit an interrupt context without updating RCU 770 * 771 * Also processes softirqs if needed and possible. 772 */ 773 void irq_exit_rcu(void) 774 { 775 __irq_exit_rcu(); 776 /* must be last! */ 777 lockdep_hardirq_exit(); 778 } 779 780 /** 781 * irq_exit - Exit an interrupt context, update RCU and lockdep 782 * 783 * Also processes softirqs if needed and possible. 784 */ 785 void irq_exit(void) 786 { 787 __irq_exit_rcu(); 788 ct_irq_exit(); 789 /* must be last! */ 790 lockdep_hardirq_exit(); 791 } 792 793 /* 794 * This function must run with irqs disabled! 795 */ 796 inline void raise_softirq_irqoff(unsigned int nr) 797 { 798 __raise_softirq_irqoff(nr); 799 800 /* 801 * If we're in an interrupt or softirq, we're done 802 * (this also catches softirq-disabled code). We will 803 * actually run the softirq once we return from 804 * the irq or softirq. 805 * 806 * Otherwise we wake up ksoftirqd to make sure we 807 * schedule the softirq soon. 808 */ 809 if (!in_interrupt() && should_wake_ksoftirqd()) 810 wakeup_softirqd(); 811 } 812 813 void raise_softirq(unsigned int nr) 814 { 815 unsigned long flags; 816 817 local_irq_save(flags); 818 raise_softirq_irqoff(nr); 819 local_irq_restore(flags); 820 } 821 822 void __raise_softirq_irqoff(unsigned int nr) 823 { 824 lockdep_assert_irqs_disabled(); 825 trace_softirq_raise(nr); 826 or_softirq_pending(1UL << nr); 827 } 828 829 void open_softirq(int nr, void (*action)(void)) 830 { 831 softirq_vec[nr].action = action; 832 } 833 834 /* 835 * Tasklets 836 */ 837 struct tasklet_head { 838 struct tasklet_struct *head; 839 struct tasklet_struct **tail; 840 }; 841 842 static DEFINE_PER_CPU(struct tasklet_head, tasklet_vec); 843 static DEFINE_PER_CPU(struct tasklet_head, tasklet_hi_vec); 844 845 static void __tasklet_schedule_common(struct tasklet_struct *t, 846 struct tasklet_head __percpu *headp, 847 unsigned int softirq_nr) 848 { 849 struct tasklet_head *head; 850 unsigned long flags; 851 852 local_irq_save(flags); 853 head = this_cpu_ptr(headp); 854 t->next = NULL; 855 *head->tail = t; 856 head->tail = &(t->next); 857 raise_softirq_irqoff(softirq_nr); 858 local_irq_restore(flags); 859 } 860 861 void __tasklet_schedule(struct tasklet_struct *t) 862 { 863 __tasklet_schedule_common(t, &tasklet_vec, 864 TASKLET_SOFTIRQ); 865 } 866 EXPORT_SYMBOL(__tasklet_schedule); 867 868 void __tasklet_hi_schedule(struct tasklet_struct *t) 869 { 870 __tasklet_schedule_common(t, &tasklet_hi_vec, 871 HI_SOFTIRQ); 872 } 873 EXPORT_SYMBOL(__tasklet_hi_schedule); 874 875 static bool tasklet_clear_sched(struct tasklet_struct *t) 876 { 877 if (test_and_clear_wake_up_bit(TASKLET_STATE_SCHED, &t->state)) 878 return true; 879 880 WARN_ONCE(1, "tasklet SCHED state not set: %s %pS\n", 881 t->use_callback ? "callback" : "func", 882 t->use_callback ? (void *)t->callback : (void *)t->func); 883 884 return false; 885 } 886 887 #ifdef CONFIG_PREEMPT_RT 888 struct tasklet_sync_callback { 889 spinlock_t cb_lock; 890 atomic_t cb_waiters; 891 }; 892 893 static DEFINE_PER_CPU(struct tasklet_sync_callback, tasklet_sync_callback) = { 894 .cb_lock = __SPIN_LOCK_UNLOCKED(tasklet_sync_callback.cb_lock), 895 .cb_waiters = ATOMIC_INIT(0), 896 }; 897 898 static void tasklet_lock_callback(void) 899 { 900 spin_lock(this_cpu_ptr(&tasklet_sync_callback.cb_lock)); 901 } 902 903 static void tasklet_unlock_callback(void) 904 { 905 spin_unlock(this_cpu_ptr(&tasklet_sync_callback.cb_lock)); 906 } 907 908 static void tasklet_callback_cancel_wait_running(void) 909 { 910 struct tasklet_sync_callback *sync_cb = this_cpu_ptr(&tasklet_sync_callback); 911 912 atomic_inc(&sync_cb->cb_waiters); 913 spin_lock(&sync_cb->cb_lock); 914 atomic_dec(&sync_cb->cb_waiters); 915 spin_unlock(&sync_cb->cb_lock); 916 } 917 918 static void tasklet_callback_sync_wait_running(void) 919 { 920 struct tasklet_sync_callback *sync_cb = this_cpu_ptr(&tasklet_sync_callback); 921 922 if (atomic_read(&sync_cb->cb_waiters)) { 923 spin_unlock(&sync_cb->cb_lock); 924 spin_lock(&sync_cb->cb_lock); 925 } 926 } 927 928 #else /* !CONFIG_PREEMPT_RT: */ 929 930 static void tasklet_lock_callback(void) { } 931 static void tasklet_unlock_callback(void) { } 932 static void tasklet_callback_sync_wait_running(void) { } 933 934 #ifdef CONFIG_SMP 935 static void tasklet_callback_cancel_wait_running(void) { } 936 #endif 937 #endif /* !CONFIG_PREEMPT_RT */ 938 939 static void tasklet_action_common(struct tasklet_head *tl_head, 940 unsigned int softirq_nr) 941 { 942 struct tasklet_struct *list; 943 944 local_irq_disable(); 945 list = tl_head->head; 946 tl_head->head = NULL; 947 tl_head->tail = &tl_head->head; 948 local_irq_enable(); 949 950 tasklet_lock_callback(); 951 while (list) { 952 struct tasklet_struct *t = list; 953 954 list = list->next; 955 956 if (tasklet_trylock(t)) { 957 if (!atomic_read(&t->count)) { 958 if (tasklet_clear_sched(t)) { 959 if (t->use_callback) { 960 trace_tasklet_entry(t, t->callback); 961 t->callback(t); 962 trace_tasklet_exit(t, t->callback); 963 } else { 964 trace_tasklet_entry(t, t->func); 965 t->func(t->data); 966 trace_tasklet_exit(t, t->func); 967 } 968 } 969 tasklet_unlock(t); 970 tasklet_callback_sync_wait_running(); 971 continue; 972 } 973 tasklet_unlock(t); 974 } 975 976 local_irq_disable(); 977 t->next = NULL; 978 *tl_head->tail = t; 979 tl_head->tail = &t->next; 980 __raise_softirq_irqoff(softirq_nr); 981 local_irq_enable(); 982 } 983 tasklet_unlock_callback(); 984 } 985 986 static __latent_entropy void tasklet_action(void) 987 { 988 workqueue_softirq_action(false); 989 tasklet_action_common(this_cpu_ptr(&tasklet_vec), TASKLET_SOFTIRQ); 990 } 991 992 static __latent_entropy void tasklet_hi_action(void) 993 { 994 workqueue_softirq_action(true); 995 tasklet_action_common(this_cpu_ptr(&tasklet_hi_vec), HI_SOFTIRQ); 996 } 997 998 void tasklet_setup(struct tasklet_struct *t, 999 void (*callback)(struct tasklet_struct *)) 1000 { 1001 t->next = NULL; 1002 t->state = 0; 1003 atomic_set(&t->count, 0); 1004 t->callback = callback; 1005 t->use_callback = true; 1006 t->data = 0; 1007 } 1008 EXPORT_SYMBOL(tasklet_setup); 1009 1010 void tasklet_init(struct tasklet_struct *t, 1011 void (*func)(unsigned long), unsigned long data) 1012 { 1013 t->next = NULL; 1014 t->state = 0; 1015 atomic_set(&t->count, 0); 1016 t->func = func; 1017 t->use_callback = false; 1018 t->data = data; 1019 } 1020 EXPORT_SYMBOL(tasklet_init); 1021 1022 #if defined(CONFIG_SMP) || defined(CONFIG_PREEMPT_RT) 1023 /* 1024 * Do not use in new code. Waiting for tasklets from atomic contexts is 1025 * error prone and should be avoided. 1026 */ 1027 void tasklet_unlock_spin_wait(struct tasklet_struct *t) 1028 { 1029 while (test_bit(TASKLET_STATE_RUN, &(t)->state)) { 1030 if (IS_ENABLED(CONFIG_PREEMPT_RT)) { 1031 /* 1032 * Prevent a live lock when current preempted soft 1033 * interrupt processing or prevents ksoftirqd from 1034 * running. 1035 */ 1036 tasklet_callback_cancel_wait_running(); 1037 } else { 1038 cpu_relax(); 1039 } 1040 } 1041 } 1042 EXPORT_SYMBOL(tasklet_unlock_spin_wait); 1043 #endif 1044 1045 void tasklet_kill(struct tasklet_struct *t) 1046 { 1047 if (in_interrupt()) 1048 pr_notice("Attempt to kill tasklet from interrupt\n"); 1049 1050 wait_on_bit_lock(&t->state, TASKLET_STATE_SCHED, TASK_UNINTERRUPTIBLE); 1051 1052 tasklet_unlock_wait(t); 1053 tasklet_clear_sched(t); 1054 } 1055 EXPORT_SYMBOL(tasklet_kill); 1056 1057 #if defined(CONFIG_SMP) || defined(CONFIG_PREEMPT_RT) 1058 void tasklet_unlock(struct tasklet_struct *t) 1059 { 1060 clear_and_wake_up_bit(TASKLET_STATE_RUN, &t->state); 1061 } 1062 EXPORT_SYMBOL_GPL(tasklet_unlock); 1063 1064 void tasklet_unlock_wait(struct tasklet_struct *t) 1065 { 1066 wait_on_bit(&t->state, TASKLET_STATE_RUN, TASK_UNINTERRUPTIBLE); 1067 } 1068 EXPORT_SYMBOL_GPL(tasklet_unlock_wait); 1069 #endif 1070 1071 void __init softirq_init(void) 1072 { 1073 int cpu; 1074 1075 for_each_possible_cpu(cpu) { 1076 per_cpu(tasklet_vec, cpu).tail = 1077 &per_cpu(tasklet_vec, cpu).head; 1078 per_cpu(tasklet_hi_vec, cpu).tail = 1079 &per_cpu(tasklet_hi_vec, cpu).head; 1080 } 1081 1082 open_softirq(TASKLET_SOFTIRQ, tasklet_action); 1083 open_softirq(HI_SOFTIRQ, tasklet_hi_action); 1084 } 1085 1086 static int ksoftirqd_should_run(unsigned int cpu) 1087 { 1088 return local_softirq_pending(); 1089 } 1090 1091 static void run_ksoftirqd(unsigned int cpu) 1092 { 1093 ksoftirqd_run_begin(); 1094 if (local_softirq_pending()) { 1095 /* 1096 * We can safely run softirq on inline stack, as we are not deep 1097 * in the task stack here. 1098 */ 1099 handle_softirqs(true); 1100 ksoftirqd_run_end(); 1101 cond_resched(); 1102 return; 1103 } 1104 ksoftirqd_run_end(); 1105 } 1106 1107 #ifdef CONFIG_HOTPLUG_CPU 1108 static int takeover_tasklets(unsigned int cpu) 1109 { 1110 workqueue_softirq_dead(cpu); 1111 1112 /* CPU is dead, so no lock needed. */ 1113 local_irq_disable(); 1114 1115 /* Find end, append list for that CPU. */ 1116 if (&per_cpu(tasklet_vec, cpu).head != per_cpu(tasklet_vec, cpu).tail) { 1117 *__this_cpu_read(tasklet_vec.tail) = per_cpu(tasklet_vec, cpu).head; 1118 __this_cpu_write(tasklet_vec.tail, per_cpu(tasklet_vec, cpu).tail); 1119 per_cpu(tasklet_vec, cpu).head = NULL; 1120 per_cpu(tasklet_vec, cpu).tail = &per_cpu(tasklet_vec, cpu).head; 1121 } 1122 raise_softirq_irqoff(TASKLET_SOFTIRQ); 1123 1124 if (&per_cpu(tasklet_hi_vec, cpu).head != per_cpu(tasklet_hi_vec, cpu).tail) { 1125 *__this_cpu_read(tasklet_hi_vec.tail) = per_cpu(tasklet_hi_vec, cpu).head; 1126 __this_cpu_write(tasklet_hi_vec.tail, per_cpu(tasklet_hi_vec, cpu).tail); 1127 per_cpu(tasklet_hi_vec, cpu).head = NULL; 1128 per_cpu(tasklet_hi_vec, cpu).tail = &per_cpu(tasklet_hi_vec, cpu).head; 1129 } 1130 raise_softirq_irqoff(HI_SOFTIRQ); 1131 1132 local_irq_enable(); 1133 return 0; 1134 } 1135 #else 1136 #define takeover_tasklets NULL 1137 #endif /* CONFIG_HOTPLUG_CPU */ 1138 1139 static struct smp_hotplug_thread softirq_threads = { 1140 .store = &ksoftirqd, 1141 .thread_should_run = ksoftirqd_should_run, 1142 .thread_fn = run_ksoftirqd, 1143 .thread_comm = "ksoftirqd/%u", 1144 }; 1145 1146 #ifdef CONFIG_IRQ_FORCED_THREADING 1147 static void ktimerd_setup(unsigned int cpu) 1148 { 1149 /* Above SCHED_NORMAL to handle timers before regular tasks. */ 1150 sched_set_fifo_low(current); 1151 } 1152 1153 static int ktimerd_should_run(unsigned int cpu) 1154 { 1155 return local_timers_pending_force_th(); 1156 } 1157 1158 void raise_ktimers_thread(unsigned int nr) 1159 { 1160 trace_softirq_raise(nr); 1161 __this_cpu_or(pending_timer_softirq, BIT(nr)); 1162 } 1163 1164 static void run_ktimerd(unsigned int cpu) 1165 { 1166 unsigned int timer_si; 1167 1168 ksoftirqd_run_begin(); 1169 1170 timer_si = local_timers_pending_force_th(); 1171 __this_cpu_write(pending_timer_softirq, 0); 1172 or_softirq_pending(timer_si); 1173 1174 __do_softirq(); 1175 1176 ksoftirqd_run_end(); 1177 } 1178 1179 static struct smp_hotplug_thread timer_thread = { 1180 .store = &ktimerd, 1181 .setup = ktimerd_setup, 1182 .thread_should_run = ktimerd_should_run, 1183 .thread_fn = run_ktimerd, 1184 .thread_comm = "ktimers/%u", 1185 }; 1186 #endif 1187 1188 static __init int spawn_ksoftirqd(void) 1189 { 1190 cpuhp_setup_state_nocalls(CPUHP_SOFTIRQ_DEAD, "softirq:dead", NULL, 1191 takeover_tasklets); 1192 BUG_ON(smpboot_register_percpu_thread(&softirq_threads)); 1193 #ifdef CONFIG_IRQ_FORCED_THREADING 1194 if (force_irqthreads()) 1195 BUG_ON(smpboot_register_percpu_thread(&timer_thread)); 1196 #endif 1197 return 0; 1198 } 1199 early_initcall(spawn_ksoftirqd); 1200 1201 /* 1202 * [ These __weak aliases are kept in a separate compilation unit, so that 1203 * GCC does not inline them incorrectly. ] 1204 */ 1205 1206 int __init __weak early_irq_init(void) 1207 { 1208 return 0; 1209 } 1210 1211 int __init __weak arch_probe_nr_irqs(void) 1212 { 1213 return NR_IRQS_LEGACY; 1214 } 1215 1216 int __init __weak arch_early_irq_init(void) 1217 { 1218 return 0; 1219 } 1220 1221 unsigned int __weak arch_dynirq_lower_bound(unsigned int from) 1222 { 1223 return from; 1224 } 1225