xref: /linux/kernel/sched/fair.c (revision 65efcccddc83d6a19e8a2e2a6117811e39e589bf)
1 // SPDX-License-Identifier: GPL-2.0
2 /*
3  * Completely Fair Scheduling (CFS) Class (SCHED_NORMAL/SCHED_BATCH)
4  *
5  *  Copyright (C) 2007 Red Hat, Inc., Ingo Molnar <mingo@redhat.com>
6  *
7  *  Interactivity improvements by Mike Galbraith
8  *  (C) 2007 Mike Galbraith <efault@gmx.de>
9  *
10  *  Various enhancements by Dmitry Adamushko.
11  *  (C) 2007 Dmitry Adamushko <dmitry.adamushko@gmail.com>
12  *
13  *  Group scheduling enhancements by Srivatsa Vaddagiri
14  *  Copyright IBM Corporation, 2007
15  *  Author: Srivatsa Vaddagiri <vatsa@linux.vnet.ibm.com>
16  *
17  *  Scaled math optimizations by Thomas Gleixner
18  *  Copyright (C) 2007, Linutronix GmbH, Thomas Gleixner <tglx@kernel.org>
19  *
20  *  Adaptive scheduling granularity, math enhancements by Peter Zijlstra
21  *  Copyright (C) 2007 Red Hat, Inc., Peter Zijlstra
22  */
23 #include <linux/energy_model.h>
24 #include <linux/mmap_lock.h>
25 #include <linux/hugetlb_inline.h>
26 #include <linux/jiffies.h>
27 #include <linux/mm_api.h>
28 #include <linux/highmem.h>
29 #include <linux/hrtimer.h>
30 #include <linux/hrtimer_bases.h>
31 #include <linux/spinlock_api.h>
32 #include <linux/cpumask_api.h>
33 #include <linux/lockdep_api.h>
34 #include <linux/softirq.h>
35 #include <linux/refcount_api.h>
36 #include <linux/topology.h>
37 #include <linux/sched/clock.h>
38 #include <linux/sched/cond_resched.h>
39 #include <linux/sched/cputime.h>
40 #include <linux/sched/isolation.h>
41 #include <linux/sched/nohz.h>
42 #include <linux/sched/prio.h>
43 #include <linux/static_call.h>
44 
45 #include <linux/cpuidle.h>
46 #include <linux/interrupt.h>
47 #include <linux/memory-tiers.h>
48 #include <linux/mempolicy.h>
49 #include <linux/mutex_api.h>
50 #include <linux/profile.h>
51 #include <linux/psi.h>
52 #include <linux/ratelimit.h>
53 #include <linux/task_work.h>
54 #include <linux/rbtree_augmented.h>
55 
56 #include <asm/switch_to.h>
57 
58 #include <uapi/linux/sched/types.h>
59 
60 #include "sched.h"
61 #include "stats.h"
62 #include "autogroup.h"
63 
64 /*
65  * The initial- and re-scaling of tunables is configurable
66  *
67  * Options are:
68  *
69  *   SCHED_TUNABLESCALING_NONE - unscaled, always *1
70  *   SCHED_TUNABLESCALING_LOG - scaled logarithmically, *1+ilog(ncpus)
71  *   SCHED_TUNABLESCALING_LINEAR - scaled linear, *ncpus
72  *
73  * (default SCHED_TUNABLESCALING_LOG = *(1+ilog(ncpus))
74  */
75 unsigned int sysctl_sched_tunable_scaling = SCHED_TUNABLESCALING_LOG;
76 
77 /*
78  * Default base time slice (request size r_i) for SCHED_NORMAL/SCHED_BATCH:
79  *
80  * Under EEVDF this is the request size used to compute the virtual
81  * deadline; see update_deadline().
82  *
83  * (default: 0.70 msec * (1 + ilog(ncpus)), units: nanoseconds)
84  */
85 unsigned int sysctl_sched_base_slice			= 700000ULL;
86 static unsigned int normalized_sysctl_sched_base_slice	= 700000ULL;
87 
88 __read_mostly unsigned int sysctl_sched_migration_cost	= 500000UL;
89 
90 static int __init setup_sched_thermal_decay_shift(char *str)
91 {
92 	pr_warn("Ignoring the deprecated sched_thermal_decay_shift= option\n");
93 	return 1;
94 }
95 __setup("sched_thermal_decay_shift=", setup_sched_thermal_decay_shift);
96 
97 /*
98  * For asym packing, by default the lower numbered CPU has higher priority.
99  */
100 int __weak arch_asym_cpu_priority(int cpu)
101 {
102 	return -cpu;
103 }
104 
105 /*
106  * The margin used when comparing utilization with CPU capacity.
107  *
108  * (default: ~20%)
109  */
110 #define fits_capacity(cap, max)	((cap) * 1280 < (max) * 1024)
111 
112 /*
113  * The margin used when comparing CPU capacities.
114  * is 'cap1' noticeably greater than 'cap2'
115  *
116  * (default: ~5%)
117  */
118 #define capacity_greater(cap1, cap2) ((cap1) * 1024 > (cap2) * 1078)
119 
120 #ifdef CONFIG_CFS_BANDWIDTH
121 /*
122  * Amount of runtime to allocate from global (tg) to local (per-cfs_rq) pool
123  * each time a cfs_rq requests quota.
124  *
125  * Note: in the case that the slice exceeds the runtime remaining (either due
126  * to consumption or the quota being specified to be smaller than the slice)
127  * we will always only issue the remaining available time.
128  *
129  * (default: 5 msec, units: microseconds)
130  */
131 static unsigned int sysctl_sched_cfs_bandwidth_slice		= 5000UL;
132 #endif
133 
134 #ifdef CONFIG_NUMA_BALANCING
135 /* Restrict the NUMA promotion throughput (MB/s) for each target node. */
136 static unsigned int sysctl_numa_balancing_promote_rate_limit = 65536;
137 #endif
138 
139 #ifdef CONFIG_SYSCTL
140 static const struct ctl_table sched_fair_sysctls[] = {
141 #ifdef CONFIG_CFS_BANDWIDTH
142 	{
143 		.procname       = "sched_cfs_bandwidth_slice_us",
144 		.data           = &sysctl_sched_cfs_bandwidth_slice,
145 		.maxlen         = sizeof(unsigned int),
146 		.mode           = 0644,
147 		.proc_handler   = proc_dointvec_minmax,
148 		.extra1         = SYSCTL_ONE,
149 	},
150 #endif
151 #ifdef CONFIG_NUMA_BALANCING
152 	{
153 		.procname	= "numa_balancing_promote_rate_limit_MBps",
154 		.data		= &sysctl_numa_balancing_promote_rate_limit,
155 		.maxlen		= sizeof(unsigned int),
156 		.mode		= 0644,
157 		.proc_handler	= proc_dointvec_minmax,
158 		.extra1		= SYSCTL_ZERO,
159 	},
160 #endif /* CONFIG_NUMA_BALANCING */
161 };
162 
163 static int __init sched_fair_sysctl_init(void)
164 {
165 	register_sysctl_init("kernel", sched_fair_sysctls);
166 	return 0;
167 }
168 late_initcall(sched_fair_sysctl_init);
169 #endif /* CONFIG_SYSCTL */
170 
171 static inline void update_load_add(struct load_weight *lw, unsigned long inc)
172 {
173 	lw->weight += inc;
174 	lw->inv_weight = 0;
175 }
176 
177 static inline void update_load_sub(struct load_weight *lw, unsigned long dec)
178 {
179 	lw->weight -= dec;
180 	lw->inv_weight = 0;
181 }
182 
183 static inline void update_load_set(struct load_weight *lw, unsigned long w)
184 {
185 	lw->weight = w;
186 	lw->inv_weight = 0;
187 }
188 
189 /*
190  * Increase the granularity value when there are more CPUs,
191  * because with more CPUs the 'effective latency' as visible
192  * to users decreases. But the relationship is not linear,
193  * so pick a second-best guess by going with the log2 of the
194  * number of CPUs.
195  *
196  * This idea comes from the SD scheduler of Con Kolivas:
197  */
198 static unsigned int get_update_sysctl_factor(void)
199 {
200 	unsigned int cpus = min_t(unsigned int, num_online_cpus(), 8);
201 	unsigned int factor;
202 
203 	switch (sysctl_sched_tunable_scaling) {
204 	case SCHED_TUNABLESCALING_NONE:
205 		factor = 1;
206 		break;
207 	case SCHED_TUNABLESCALING_LINEAR:
208 		factor = cpus;
209 		break;
210 	case SCHED_TUNABLESCALING_LOG:
211 	default:
212 		factor = 1 + ilog2(cpus);
213 		break;
214 	}
215 
216 	return factor;
217 }
218 
219 static void update_sysctl(void)
220 {
221 	unsigned int factor = get_update_sysctl_factor();
222 
223 #define SET_SYSCTL(name) \
224 	(sysctl_##name = (factor) * normalized_sysctl_##name)
225 	SET_SYSCTL(sched_base_slice);
226 #undef SET_SYSCTL
227 }
228 
229 void __init sched_init_granularity(void)
230 {
231 	update_sysctl();
232 }
233 
234 #ifndef CONFIG_64BIT
235 #define WMULT_CONST	(~0U)
236 #define WMULT_SHIFT	32
237 
238 static void __update_inv_weight(struct load_weight *lw)
239 {
240 	unsigned long w;
241 
242 	if (likely(lw->inv_weight))
243 		return;
244 
245 	w = scale_load_down(lw->weight);
246 
247 	if (BITS_PER_LONG > 32 && unlikely(w >= WMULT_CONST))
248 		lw->inv_weight = 1;
249 	else if (unlikely(!w))
250 		lw->inv_weight = WMULT_CONST;
251 	else
252 		lw->inv_weight = WMULT_CONST / w;
253 }
254 
255 /*
256  * delta_exec * weight / lw.weight
257  *   OR
258  * (delta_exec * (weight * lw->inv_weight)) >> WMULT_SHIFT
259  *
260  * Either weight := NICE_0_LOAD and lw \e sched_prio_to_wmult[], in which case
261  * we're guaranteed shift stays positive because inv_weight is guaranteed to
262  * fit 32 bits, and NICE_0_LOAD gives another 10 bits; therefore shift >= 22.
263  *
264  * Or, weight =< lw.weight (because lw.weight is the runqueue weight), thus
265  * weight/lw.weight <= 1, and therefore our shift will also be positive.
266  */
267 static u64 __calc_delta(u64 delta_exec, unsigned long weight, struct load_weight *lw)
268 {
269 	u64 fact = scale_load_down(weight);
270 	u32 fact_hi = (u32)(fact >> 32);
271 	int shift = WMULT_SHIFT;
272 	int fs;
273 
274 	__update_inv_weight(lw);
275 
276 	if (unlikely(fact_hi)) {
277 		fs = fls(fact_hi);
278 		shift -= fs;
279 		fact >>= fs;
280 	}
281 
282 	fact = mul_u32_u32(fact, lw->inv_weight);
283 
284 	fact_hi = (u32)(fact >> 32);
285 	if (fact_hi) {
286 		fs = fls(fact_hi);
287 		shift -= fs;
288 		fact >>= fs;
289 	}
290 
291 	return mul_u64_u32_shr(delta_exec, fact, shift);
292 }
293 #else
294 static u64 __calc_delta(u64 delta_exec, unsigned long weight, struct load_weight *lw)
295 {
296 	return (delta_exec * weight) / lw->weight;
297 }
298 #endif
299 
300 /*
301  * delta /= w
302  */
303 static inline u64 calc_delta_fair(u64 delta, struct sched_entity *se)
304 {
305 	if (se->h_load.weight != NICE_0_LOAD)
306 		delta = __calc_delta(delta, NICE_0_LOAD, &se->h_load);
307 
308 	return delta;
309 }
310 
311 const struct sched_class fair_sched_class;
312 
313 /**************************************************************
314  * CFS operations on generic schedulable entities:
315  */
316 
317 #ifdef CONFIG_FAIR_GROUP_SCHED
318 
319 /* Walk up scheduling entities hierarchy */
320 #define for_each_sched_entity(se) \
321 		for (; se; se = se->parent)
322 
323 static inline bool list_add_leaf_cfs_rq(struct cfs_rq *cfs_rq)
324 {
325 	struct rq *rq = rq_of(cfs_rq);
326 	int cpu = cpu_of(rq);
327 
328 	if (cfs_rq->on_list)
329 		return rq->tmp_alone_branch == &rq->leaf_cfs_rq_list;
330 
331 	cfs_rq->on_list = 1;
332 
333 	/*
334 	 * Ensure we either appear before our parent (if already
335 	 * enqueued) or force our parent to appear after us when it is
336 	 * enqueued. The fact that we always enqueue bottom-up
337 	 * reduces this to two cases and a special case for the root
338 	 * cfs_rq. Furthermore, it also means that we will always reset
339 	 * tmp_alone_branch either when the branch is connected
340 	 * to a tree or when we reach the top of the tree
341 	 */
342 	if (cfs_rq->tg->parent &&
343 	    tg_cfs_rq(cfs_rq->tg->parent, cpu)->on_list) {
344 		/*
345 		 * If parent is already on the list, we add the child
346 		 * just before. Thanks to circular linked property of
347 		 * the list, this means to put the child at the tail
348 		 * of the list that starts by parent.
349 		 */
350 		list_add_tail_rcu(&cfs_rq->leaf_cfs_rq_list,
351 			&(tg_cfs_rq(cfs_rq->tg->parent, cpu)->leaf_cfs_rq_list));
352 		/*
353 		 * The branch is now connected to its tree so we can
354 		 * reset tmp_alone_branch to the beginning of the
355 		 * list.
356 		 */
357 		rq->tmp_alone_branch = &rq->leaf_cfs_rq_list;
358 		return true;
359 	}
360 
361 	if (!cfs_rq->tg->parent) {
362 		/*
363 		 * cfs rq without parent should be put
364 		 * at the tail of the list.
365 		 */
366 		list_add_tail_rcu(&cfs_rq->leaf_cfs_rq_list,
367 			&rq->leaf_cfs_rq_list);
368 		/*
369 		 * We have reach the top of a tree so we can reset
370 		 * tmp_alone_branch to the beginning of the list.
371 		 */
372 		rq->tmp_alone_branch = &rq->leaf_cfs_rq_list;
373 		return true;
374 	}
375 
376 	/*
377 	 * The parent has not already been added so we want to
378 	 * make sure that it will be put after us.
379 	 * tmp_alone_branch points to the begin of the branch
380 	 * where we will add parent.
381 	 */
382 	list_add_rcu(&cfs_rq->leaf_cfs_rq_list, rq->tmp_alone_branch);
383 	/*
384 	 * update tmp_alone_branch to points to the new begin
385 	 * of the branch
386 	 */
387 	rq->tmp_alone_branch = &cfs_rq->leaf_cfs_rq_list;
388 	return false;
389 }
390 
391 static inline void list_del_leaf_cfs_rq(struct cfs_rq *cfs_rq)
392 {
393 	if (cfs_rq->on_list) {
394 		struct rq *rq = rq_of(cfs_rq);
395 
396 		/*
397 		 * With cfs_rq being unthrottled/throttled during an enqueue,
398 		 * it can happen the tmp_alone_branch points to the leaf that
399 		 * we finally want to delete. In this case, tmp_alone_branch moves
400 		 * to the prev element but it will point to rq->leaf_cfs_rq_list
401 		 * at the end of the enqueue.
402 		 */
403 		if (rq->tmp_alone_branch == &cfs_rq->leaf_cfs_rq_list)
404 			rq->tmp_alone_branch = cfs_rq->leaf_cfs_rq_list.prev;
405 
406 		list_del_rcu(&cfs_rq->leaf_cfs_rq_list);
407 		cfs_rq->on_list = 0;
408 	}
409 }
410 
411 static inline void assert_list_leaf_cfs_rq(struct rq *rq)
412 {
413 	WARN_ON_ONCE(rq->tmp_alone_branch != &rq->leaf_cfs_rq_list);
414 }
415 
416 /* Iterate through all leaf cfs_rq's on a runqueue */
417 #define for_each_leaf_cfs_rq_safe(rq, cfs_rq, pos)			\
418 	list_for_each_entry_safe(cfs_rq, pos, &rq->leaf_cfs_rq_list,	\
419 				 leaf_cfs_rq_list)
420 
421 /* Do the two (enqueued) entities belong to the same group ? */
422 static inline struct cfs_rq *
423 is_same_group(struct sched_entity *se, struct sched_entity *pse)
424 {
425 	if (se->cfs_rq == pse->cfs_rq)
426 		return se->cfs_rq;
427 
428 	return NULL;
429 }
430 
431 static inline struct sched_entity *parent_entity(const struct sched_entity *se)
432 {
433 	return se->parent;
434 }
435 
436 static int tg_is_idle(struct task_group *tg)
437 {
438 	return tg->idle > 0;
439 }
440 
441 static int cfs_rq_is_idle(struct cfs_rq *cfs_rq)
442 {
443 	return cfs_rq->idle > 0;
444 }
445 
446 static int se_is_idle(struct sched_entity *se)
447 {
448 	if (entity_is_task(se))
449 		return task_has_idle_policy(task_of(se));
450 	return cfs_rq_is_idle(group_cfs_rq(se));
451 }
452 
453 #else /* !CONFIG_FAIR_GROUP_SCHED: */
454 
455 #define for_each_sched_entity(se) \
456 		for (; se; se = NULL)
457 
458 static inline bool list_add_leaf_cfs_rq(struct cfs_rq *cfs_rq)
459 {
460 	return true;
461 }
462 
463 static inline void list_del_leaf_cfs_rq(struct cfs_rq *cfs_rq)
464 {
465 }
466 
467 static inline void assert_list_leaf_cfs_rq(struct rq *rq)
468 {
469 }
470 
471 #define for_each_leaf_cfs_rq_safe(rq, cfs_rq, pos)	\
472 		for (cfs_rq = &rq->cfs, pos = NULL; cfs_rq; cfs_rq = pos)
473 
474 static inline struct sched_entity *parent_entity(struct sched_entity *se)
475 {
476 	return NULL;
477 }
478 
479 static inline int tg_is_idle(struct task_group *tg)
480 {
481 	return 0;
482 }
483 
484 static int cfs_rq_is_idle(struct cfs_rq *cfs_rq)
485 {
486 	return 0;
487 }
488 
489 static int se_is_idle(struct sched_entity *se)
490 {
491 	return task_has_idle_policy(task_of(se));
492 }
493 
494 #endif /* !CONFIG_FAIR_GROUP_SCHED */
495 
496 static __always_inline
497 bool account_cfs_rq_runtime(struct cfs_rq *cfs_rq, u64 delta_exec);
498 
499 /**************************************************************
500  * Scheduling class tree data structure manipulation methods:
501  */
502 
503 extern void __BUILD_BUG_vruntime_cmp(void);
504 
505 /* Use __builtin_strcmp() because of __HAVE_ARCH_STRCMP: */
506 
507 #define vruntime_cmp(A, CMP_STR, B) ({				\
508 	int __res = 0;						\
509 								\
510 	if (!__builtin_strcmp(CMP_STR, "<")) {			\
511 		__res = ((s64)((A)-(B)) < 0);			\
512 	} else if (!__builtin_strcmp(CMP_STR, "<=")) {		\
513 		__res = ((s64)((A)-(B)) <= 0);			\
514 	} else if (!__builtin_strcmp(CMP_STR, ">")) {		\
515 		__res = ((s64)((A)-(B)) > 0);			\
516 	} else if (!__builtin_strcmp(CMP_STR, ">=")) {		\
517 		__res = ((s64)((A)-(B)) >= 0);			\
518 	} else {						\
519 		/* Unknown operator throws linker error: */	\
520 		__BUILD_BUG_vruntime_cmp();			\
521 	}							\
522 								\
523 	__res;							\
524 })
525 
526 extern void __BUILD_BUG_vruntime_op(void);
527 
528 #define vruntime_op(A, OP_STR, B) ({				\
529 	s64 __res = 0;						\
530 								\
531 	if (!__builtin_strcmp(OP_STR, "-")) {			\
532 		__res = (s64)((A)-(B));				\
533 	} else {						\
534 		/* Unknown operator throws linker error: */	\
535 		__BUILD_BUG_vruntime_op();			\
536 	}							\
537 								\
538 	__res;						\
539 })
540 
541 
542 static inline __maybe_unused u64 max_vruntime(u64 max_vruntime, u64 vruntime)
543 {
544 	if (vruntime_cmp(vruntime, ">", max_vruntime))
545 		max_vruntime = vruntime;
546 
547 	return max_vruntime;
548 }
549 
550 static inline __maybe_unused u64 min_vruntime(u64 min_vruntime, u64 vruntime)
551 {
552 	if (vruntime_cmp(vruntime, "<", min_vruntime))
553 		min_vruntime = vruntime;
554 
555 	return min_vruntime;
556 }
557 
558 static inline bool entity_before(const struct sched_entity *a,
559 				 const struct sched_entity *b)
560 {
561 	/*
562 	 * Tiebreak on vruntime seems unnecessary since it can
563 	 * hardly happen.
564 	 */
565 	return vruntime_cmp(a->deadline, "<", b->deadline);
566 }
567 
568 /*
569  * Per avg_vruntime() below, cfs_rq::zero_vruntime is only slightly stale
570  * and this value should be no more than two lag bounds. Which puts it in the
571  * general order of:
572  *
573  *	(slice + TICK_NSEC) << NICE_0_LOAD_SHIFT
574  *
575  * which is around 44 bits in size (on 64bit); that is 20 for
576  * NICE_0_LOAD_SHIFT, another 20 for NSEC_PER_MSEC and then a handful for
577  * however many msec the actual slice+tick ends up begin.
578  *
579  * (disregarding the actual divide-by-weight part makes for the worst case
580  * weight of 2, which nicely cancels vs the fuzz in zero_vruntime not actually
581  * being the zero-lag point).
582  */
583 static inline s64 entity_key(struct cfs_rq *cfs_rq, struct sched_entity *se)
584 {
585 	return vruntime_op(se->vruntime, "-", cfs_rq->zero_vruntime);
586 }
587 
588 #define __node_2_se(node) \
589 	rb_entry((node), struct sched_entity, run_node)
590 
591 /*
592  * Compute virtual time from the per-task service numbers:
593  *
594  * Fair schedulers conserve lag:
595  *
596  *   \Sum lag_i = 0
597  *
598  * Where lag_i is given by:
599  *
600  *   lag_i = S - s_i = w_i * (V - v_i)
601  *
602  * Where S is the ideal service time and V is it's virtual time counterpart.
603  * Therefore:
604  *
605  *   \Sum lag_i = 0
606  *   \Sum w_i * (V - v_i) = 0
607  *   \Sum (w_i * V - w_i * v_i) = 0
608  *
609  * From which we can solve an expression for V in v_i (which we have in
610  * se->vruntime):
611  *
612  *       \Sum v_i * w_i   \Sum v_i * w_i
613  *   V = -------------- = --------------
614  *          \Sum w_i            W
615  *
616  * Specifically, this is the weighted average of all entity virtual runtimes.
617  *
618  * [[ NOTE: this is only equal to the ideal scheduler under the condition
619  *          that join/leave operations happen at lag_i = 0, otherwise the
620  *          virtual time has non-contiguous motion equivalent to:
621  *
622  *	      V +-= lag_i / W
623  *
624  *	    Also see the comment in place_entity() that deals with this. ]]
625  *
626  * However, since v_i is u64, and the multiplication could easily overflow
627  * transform it into a relative form that uses smaller quantities:
628  *
629  * Substitute: v_i == (v_i - v0) + v0
630  *
631  *     \Sum ((v_i - v0) + v0) * w_i   \Sum (v_i - v0) * w_i
632  * V = ---------------------------- = --------------------- + v0
633  *                  W                            W
634  *
635  * Which we track using:
636  *
637  *                    v0 := cfs_rq->zero_vruntime
638  * \Sum (v_i - v0) * w_i := cfs_rq->sum_w_vruntime
639  *              \Sum w_i := cfs_rq->sum_weight
640  *
641  * Since zero_vruntime closely tracks the per-task service, these
642  * deltas: (v_i - v0), will be in the order of the maximal (virtual) lag
643  * induced in the system due to quantisation.
644  */
645 static inline unsigned long avg_vruntime_weight(struct cfs_rq *cfs_rq, unsigned long w)
646 {
647 #ifdef CONFIG_64BIT
648 	if (cfs_rq->sum_shift)
649 		w = max(2UL, w >> cfs_rq->sum_shift);
650 #endif
651 	return w;
652 }
653 
654 static inline void
655 __sum_w_vruntime_add(struct cfs_rq *cfs_rq, struct sched_entity *se)
656 {
657 	unsigned long weight = avg_vruntime_weight(cfs_rq, se->h_load.weight);
658 	s64 w_vruntime, key = entity_key(cfs_rq, se);
659 
660 	w_vruntime = key * weight;
661 	WARN_ON_ONCE((w_vruntime >> 63) != (w_vruntime >> 62));
662 
663 	cfs_rq->sum_w_vruntime += w_vruntime;
664 	cfs_rq->sum_weight += weight;
665 }
666 
667 static void
668 sum_w_vruntime_add_paranoid(struct cfs_rq *cfs_rq, struct sched_entity *se)
669 {
670 	unsigned long weight;
671 	s64 key, tmp;
672 
673 again:
674 	weight = avg_vruntime_weight(cfs_rq, se->h_load.weight);
675 	key = entity_key(cfs_rq, se);
676 
677 	if (check_mul_overflow(key, weight, &key))
678 		goto overflow;
679 
680 	if (check_add_overflow(cfs_rq->sum_w_vruntime, key, &tmp))
681 		goto overflow;
682 
683 	cfs_rq->sum_w_vruntime = tmp;
684 	cfs_rq->sum_weight += weight;
685 	return;
686 
687 overflow:
688 	/*
689 	 * There's gotta be a limit -- if we're still failing at this point
690 	 * there's really nothing much to be done about things.
691 	 */
692 	BUG_ON(cfs_rq->sum_shift >= 10);
693 	cfs_rq->sum_shift++;
694 
695 	/*
696 	 * Note: \Sum (k_i * (w_i >> 1)) != (\Sum (k_i * w_i)) >> 1
697 	 */
698 	cfs_rq->sum_w_vruntime = 0;
699 	cfs_rq->sum_weight = 0;
700 
701 	for (struct rb_node *node = cfs_rq->tasks_timeline.rb_leftmost;
702 	     node; node = rb_next(node))
703 		__sum_w_vruntime_add(cfs_rq, __node_2_se(node));
704 
705 	goto again;
706 }
707 
708 static void
709 sum_w_vruntime_add(struct cfs_rq *cfs_rq, struct sched_entity *se)
710 {
711 	if (sched_feat(PARANOID_AVG))
712 		return sum_w_vruntime_add_paranoid(cfs_rq, se);
713 
714 	__sum_w_vruntime_add(cfs_rq, se);
715 }
716 
717 static void
718 sum_w_vruntime_sub(struct cfs_rq *cfs_rq, struct sched_entity *se)
719 {
720 	unsigned long weight = avg_vruntime_weight(cfs_rq, se->h_load.weight);
721 	s64 key = entity_key(cfs_rq, se);
722 
723 	cfs_rq->sum_w_vruntime -= key * weight;
724 	cfs_rq->sum_weight -= weight;
725 }
726 
727 static inline
728 void update_zero_vruntime(struct cfs_rq *cfs_rq, s64 delta)
729 {
730 	/*
731 	 * v' = v + d ==> sum_w_vruntime' = sum_w_vruntime - d*sum_weight
732 	 */
733 	cfs_rq->sum_w_vruntime -= cfs_rq->sum_weight * delta;
734 	cfs_rq->zero_vruntime += delta;
735 }
736 
737 /*
738  * Specifically: avg_vruntime() + 0 must result in entity_eligible() := true
739  * For this to be so, the result of this function must have a left bias.
740  *
741  * Called in:
742  *  - place_entity()      -- before enqueue
743  *  - update_entity_lag() -- before dequeue
744  *  - update_deadline()   -- slice expiration
745  *
746  * This means it is one entry 'behind' but that puts it close enough to where
747  * the bound on entity_key() is at most two lag bounds.
748  */
749 u64 avg_vruntime(struct cfs_rq *cfs_rq)
750 {
751 	struct sched_entity *curr = cfs_rq->curr;
752 	long weight = cfs_rq->sum_weight;
753 	s64 delta = 0;
754 
755 	if (curr && !curr->on_rq)
756 		curr = NULL;
757 
758 	if (weight) {
759 		s64 runtime = cfs_rq->sum_w_vruntime;
760 
761 		if (curr) {
762 			unsigned long w = avg_vruntime_weight(cfs_rq, curr->h_load.weight);
763 
764 			runtime += entity_key(cfs_rq, curr) * w;
765 			weight += w;
766 		}
767 
768 		/* sign flips effective floor / ceiling */
769 		if (runtime < 0)
770 			runtime -= (weight - 1);
771 
772 		delta = div64_long(runtime, weight);
773 	} else if (curr) {
774 		/*
775 		 * When there is but one element, it is the average.
776 		 */
777 		delta = curr->vruntime - cfs_rq->zero_vruntime;
778 	}
779 
780 	update_zero_vruntime(cfs_rq, delta);
781 
782 	return cfs_rq->zero_vruntime;
783 }
784 
785 /*
786  *     \Sum (v_i - v0)*w_i
787  * V = ------------------- + v0
788  *          \Sum w_i
789  *
790  * Let W = \Sum w_i, and move v_j such that 'v_j == V', thus:
791  *
792  * V = 1/W * {(v_j - v0)*w_j + \Sum_i!=j (v_i - v0)*w_i} + v0
793  *
794  * v_j = 1/W * {(v_j - v0)*w_j + \Sum_i!=j (v_i - v0)*w_i} + v0
795  *
796  * v_j = 1/W * (v_j - v0)*w_j + 1/W * \Sum_i!=j (v_i - v0)*w_i + v0
797  *
798  * v_j - 1/W * (v_j - v0)*w_j = 1/W * \Sum_i!=j (v_i - v0)*w_i + v0
799  *
800  * v_j*W - (v_j - v0)*w_j = \Sum_i!=j (v_i - v0)*w_i + v0*W
801  *
802  * v_j*(W - w_j) + v0*w_j = \Sum_i!=j (v_i - v0)*w_i + v0*W
803  *
804  * v_j*(W - w_j) = \Sum_i!=j (v_i - v0)*w_i + v0*(W - w_j)
805  *
806  *       \Sum_i!=j (v_i - v0)*w_i
807  * v_j = ------------------------ + v0
808  *               W - w_j
809  *
810  * When v_j happens to be curr, then '\Sum_i!=j (v_i - v0)*w_i'
811  * is cfs_rq->sum_w_runtime, and 'W - w_j' is cfs_rq->sum_weight, since curr
812  * is not included in the sum.
813  */
814 static u64 ineligible_vruntime(struct cfs_rq *cfs_rq)
815 {
816 	struct sched_entity *curr = cfs_rq->curr;
817 	long weight = cfs_rq->sum_weight;
818 	s64 delta = 0;
819 
820 	if (curr && !curr->on_rq)
821 		curr = NULL;
822 
823 	/*
824 	 * This is called from set_next_task_fair(.first=true) /
825 	 * set_protect_slice() so curr had better be set and on_rq.
826 	 */
827 	WARN_ON_ONCE(!curr);
828 
829 	if (weight) {
830 		s64 runtime = cfs_rq->sum_w_vruntime;
831 
832 		/*
833 		 * Do not add @curr to obtain the effective '- w_j' terms.
834 		 */
835 
836 		/* sign flips effective floor / ceiling */
837 		if (runtime < 0)
838 			runtime -= (weight - 1);
839 
840 		delta = div64_long(runtime, weight);
841 	}
842 
843 	return cfs_rq->zero_vruntime + delta + 1;
844 }
845 
846 static inline u64 cfs_rq_max_slice(struct cfs_rq *cfs_rq);
847 
848 /*
849  * lag_i = S - s_i = w_i * (V - v_i)
850  *
851  * However, since V is approximated by the weighted average of all entities it
852  * is possible -- by addition/removal/reweight to the tree -- to move V around
853  * and end up with a larger lag than we started with.
854  *
855  * Limit this to either double the slice length with a minimum of TICK_NSEC
856  * since that is the timing granularity.
857  *
858  * EEVDF gives the following limit for a steady state system:
859  *
860  *   -r_max < lag < max(r_max, q)
861  */
862 static s64 entity_lag(struct cfs_rq *cfs_rq, struct sched_entity *se, u64 avruntime)
863 {
864 	u64 max_slice = cfs_rq_max_slice(cfs_rq) + TICK_NSEC;
865 	s64 vlag, limit;
866 
867 	vlag = avruntime - se->vruntime;
868 	limit = calc_delta_fair(max_slice, se);
869 
870 	return clamp(vlag, -limit, limit);
871 }
872 
873 /*
874  * Delayed dequeue aims to reduce the negative lag of a dequeued task. While
875  * updating the lag of an entity, check that negative lag didn't increase
876  * during the delayed dequeue period which would be unfair.
877  * Similarly, check that the entity didn't gain positive lag when DELAY_ZERO
878  * is set.
879  *
880  * Return true if the vlag has been modified. Specifically:
881  *
882  *   se->vlag != avg_vruntime() - se->vruntime
883  *
884  * This can be due to clamping in entity_lag() or clamping due to
885  * sched_delayed. Either way, when vlag is modified and the entity is
886  * retained, the tree needs to be adjusted.
887  */
888 static __always_inline
889 bool update_entity_lag(struct cfs_rq *cfs_rq, struct sched_entity *se)
890 {
891 	u64 avruntime = avg_vruntime(cfs_rq);
892 	s64 vlag = entity_lag(cfs_rq, se, avruntime);
893 
894 	if (se->sched_delayed) {
895 		/* previous vlag < 0 otherwise se would not be delayed */
896 		vlag = max(vlag, se->vlag);
897 		if (sched_feat(DELAY_ZERO))
898 			vlag = min(vlag, 0);
899 	}
900 	se->vlag = vlag;
901 
902 	return avruntime - vlag != se->vruntime;
903 }
904 
905 /*
906  * Entity is eligible once it received less service than it ought to have,
907  * eg. lag >= 0.
908  *
909  * lag_i = S - s_i = w_i*(V - v_i)
910  *
911  * lag_i >= 0 -> V >= v_i
912  *
913  *     \Sum (v_i - v0)*w_i
914  * V = ------------------- + v0
915  *          \Sum w_i
916  *
917  * lag_i >= 0 -> \Sum (v_i - v0)*w_i >= (v_i - v0)*(\Sum w_i)
918  *
919  * Note: using 'avg_vruntime() > se->vruntime' is inaccurate due
920  *       to the loss in precision caused by the division.
921  */
922 static int vruntime_eligible(struct cfs_rq *cfs_rq, u64 vruntime)
923 {
924 	struct sched_entity *curr = cfs_rq->curr;
925 	s64 key, avg = cfs_rq->sum_w_vruntime;
926 	long load = cfs_rq->sum_weight;
927 
928 	if (curr && curr->on_rq) {
929 		unsigned long weight = avg_vruntime_weight(cfs_rq, curr->h_load.weight);
930 
931 		avg += entity_key(cfs_rq, curr) * weight;
932 		load += weight;
933 	}
934 
935 	key = vruntime_op(vruntime, "-", cfs_rq->zero_vruntime);
936 
937 	/*
938 	 * The worst case term for @key includes 'NSEC_TICK * NICE_0_LOAD'
939 	 * and @load obviously includes NICE_0_LOAD. NSEC_TICK is around 24
940 	 * bits, while NICE_0_LOAD is 20 on 64bit and 10 otherwise.
941 	 *
942 	 * This gives that on 64bit the product will be at least 64bit which
943 	 * overflows s64, while on 32bit it will only be 44bits and should fit
944 	 * comfortably.
945 	 */
946 #ifdef CONFIG_64BIT
947 #ifdef CONFIG_ARCH_SUPPORTS_INT128
948 	/* This often results in simpler code than __builtin_mul_overflow(). */
949 	return avg >= (__int128)key * load;
950 #else
951 	s64 rhs;
952 	/*
953 	 * On overflow, the sign of key tells us the correct answer: a large
954 	 * positive key means vruntime >> V, so not eligible; a large negative
955 	 * key means vruntime << V, so eligible.
956 	 */
957 	if (check_mul_overflow(key, load, &rhs))
958 		return key <= 0;
959 
960 	return avg >= rhs;
961 #endif
962 #else /* 32bit */
963 	return avg >= key * load;
964 #endif
965 }
966 
967 int entity_eligible(struct cfs_rq *cfs_rq, struct sched_entity *se)
968 {
969 	return vruntime_eligible(cfs_rq, se->vruntime);
970 }
971 
972 static inline u64 cfs_rq_min_slice(struct cfs_rq *cfs_rq)
973 {
974 	struct sched_entity *root = __pick_root_entity(cfs_rq);
975 	struct sched_entity *curr = cfs_rq->curr;
976 	u64 min_slice = ~0ULL;
977 
978 	if (curr && curr->on_rq)
979 		min_slice = curr->slice;
980 
981 	if (root)
982 		min_slice = min(min_slice, root->min_slice);
983 
984 	return min_slice;
985 }
986 
987 static inline u64 cfs_rq_max_slice(struct cfs_rq *cfs_rq)
988 {
989 	struct sched_entity *root = __pick_root_entity(cfs_rq);
990 	struct sched_entity *curr = cfs_rq->curr;
991 	u64 max_slice = 0ULL;
992 
993 	if (curr && curr->on_rq)
994 		max_slice = curr->slice;
995 
996 	if (root)
997 		max_slice = max(max_slice, root->max_slice);
998 
999 	return max_slice;
1000 }
1001 
1002 static inline bool __entity_less(struct rb_node *a, const struct rb_node *b)
1003 {
1004 	return entity_before(__node_2_se(a), __node_2_se(b));
1005 }
1006 
1007 static inline void __min_vruntime_update(struct sched_entity *se, struct rb_node *node)
1008 {
1009 	if (node) {
1010 		struct sched_entity *rse = __node_2_se(node);
1011 
1012 		if (vruntime_cmp(se->min_vruntime, ">", rse->min_vruntime))
1013 			se->min_vruntime = rse->min_vruntime;
1014 	}
1015 }
1016 
1017 static inline void __min_slice_update(struct sched_entity *se, struct rb_node *node)
1018 {
1019 	if (node) {
1020 		struct sched_entity *rse = __node_2_se(node);
1021 		if (rse->min_slice < se->min_slice)
1022 			se->min_slice = rse->min_slice;
1023 	}
1024 }
1025 
1026 static inline void __max_slice_update(struct sched_entity *se, struct rb_node *node)
1027 {
1028 	if (node) {
1029 		struct sched_entity *rse = __node_2_se(node);
1030 		if (rse->max_slice > se->max_slice)
1031 			se->max_slice = rse->max_slice;
1032 	}
1033 }
1034 
1035 static inline void min_vruntime_copy(struct sched_entity *new, struct sched_entity *old)
1036 {
1037 	new->min_vruntime = old->min_vruntime;
1038 	new->min_slice = old->min_slice;
1039 	new->max_slice = old->max_slice;
1040 }
1041 
1042 /*
1043  * se->min_vruntime = min(se->vruntime, {left,right}->min_vruntime)
1044  */
1045 static inline bool min_vruntime_update(struct sched_entity *se, bool exit)
1046 {
1047 	u64 old_min_vruntime = se->min_vruntime;
1048 	u64 old_min_slice = se->min_slice;
1049 	u64 old_max_slice = se->max_slice;
1050 	struct rb_node *node = &se->run_node;
1051 
1052 	se->min_vruntime = se->vruntime;
1053 	__min_vruntime_update(se, node->rb_right);
1054 	__min_vruntime_update(se, node->rb_left);
1055 
1056 	se->min_slice = se->slice;
1057 	__min_slice_update(se, node->rb_right);
1058 	__min_slice_update(se, node->rb_left);
1059 
1060 	se->max_slice = se->slice;
1061 	__max_slice_update(se, node->rb_right);
1062 	__max_slice_update(se, node->rb_left);
1063 
1064 	return se->min_vruntime == old_min_vruntime &&
1065 	       se->min_slice == old_min_slice &&
1066 	       se->max_slice == old_max_slice;
1067 }
1068 
1069 
1070 RB_DECLARE_CALLBACKS_MULTI(static, min_vruntime_cb, struct sched_entity,
1071 		     run_node, min_vruntime_copy, min_vruntime_update);
1072 
1073 /*
1074  * Enqueue an entity into the rb-tree:
1075  */
1076 static void __enqueue_entity(struct cfs_rq *cfs_rq, struct sched_entity *se)
1077 {
1078 	WARN_ON_ONCE(&rq_of(cfs_rq)->cfs != cfs_rq);
1079 	WARN_ON_ONCE(!entity_is_task(se));
1080 
1081 	sum_w_vruntime_add(cfs_rq, se);
1082 	se->min_vruntime = se->vruntime;
1083 	se->min_slice = se->slice;
1084 	se->max_slice = se->slice;
1085 
1086 	rb_add_augmented_cached(&se->run_node, &cfs_rq->tasks_timeline,
1087 				__entity_less, &min_vruntime_cb);
1088 }
1089 
1090 static void __dequeue_entity(struct cfs_rq *cfs_rq, struct sched_entity *se)
1091 {
1092 	WARN_ON_ONCE(&rq_of(cfs_rq)->cfs != cfs_rq);
1093 	WARN_ON_ONCE(!entity_is_task(se));
1094 
1095 	rb_erase_augmented_cached(&se->run_node, &cfs_rq->tasks_timeline,
1096 				  &min_vruntime_cb);
1097 	sum_w_vruntime_sub(cfs_rq, se);
1098 }
1099 
1100 struct sched_entity *__pick_root_entity(struct cfs_rq *cfs_rq)
1101 {
1102 	struct rb_node *root = cfs_rq->tasks_timeline.rb_root.rb_node;
1103 
1104 	if (!root)
1105 		return NULL;
1106 
1107 	return __node_2_se(root);
1108 }
1109 
1110 struct sched_entity *__pick_first_entity(struct cfs_rq *cfs_rq)
1111 {
1112 	struct rb_node *left = rb_first_cached(&cfs_rq->tasks_timeline);
1113 
1114 	if (!left)
1115 		return NULL;
1116 
1117 	return __node_2_se(left);
1118 }
1119 
1120 /*
1121  * Set the vruntime up to which an entity can run before looking
1122  * for another entity to pick.
1123  * In case of run to parity, we use the shortest slice of the enqueued
1124  * entities to set the protected period.
1125  * When run to parity is disabled, we give a minimum quantum to the running
1126  * entity to ensure progress.
1127  */
1128 static inline void set_protect_slice(struct cfs_rq *cfs_rq, struct sched_entity *se)
1129 {
1130 	u64 slice = normalized_sysctl_sched_base_slice;
1131 	u64 vprot = se->deadline;
1132 
1133 	if (sched_feat(RUN_TO_PARITY))
1134 		slice = cfs_rq_min_slice(cfs_rq);
1135 
1136 	slice = min(slice, se->slice);
1137 
1138 	/* If there are shorter slices than se's one */
1139 	if (slice != se->slice) {
1140 		if (sched_feat(PREEMPT_SHORT))
1141 			vprot = min_vruntime(vprot, ineligible_vruntime(cfs_rq));
1142 		else
1143 			vprot = min_vruntime(vprot, se->vruntime + calc_delta_fair(slice, se));
1144 	}
1145 
1146 	se->vprot = vprot;
1147 }
1148 
1149 static inline void update_protect_slice(struct cfs_rq *cfs_rq, struct sched_entity *se)
1150 {
1151 	u64 slice = cfs_rq_min_slice(cfs_rq);
1152 	u64 vruntime = min_vruntime(se->vruntime, avg_vruntime(cfs_rq));
1153 
1154 	se->vprot = min_vruntime(se->vprot, vruntime + calc_delta_fair(slice, se));
1155 }
1156 
1157 static inline bool protect_slice(struct sched_entity *se)
1158 {
1159 	return vruntime_cmp(se->vruntime, "<", se->vprot);
1160 }
1161 
1162 static inline void cancel_protect_slice(struct sched_entity *se)
1163 {
1164 	if (protect_slice(se))
1165 		se->vprot = se->vruntime;
1166 }
1167 
1168 /*
1169  * Earliest Eligible Virtual Deadline First
1170  *
1171  * In order to provide latency guarantees for different request sizes
1172  * EEVDF selects the best runnable task from two criteria:
1173  *
1174  *  1) the task must be eligible (must be owed service)
1175  *
1176  *  2) from those tasks that meet 1), we select the one
1177  *     with the earliest virtual deadline.
1178  *
1179  * We can do this in O(log n) time due to an augmented RB-tree. The
1180  * tree keeps the entries sorted on deadline, but also functions as a
1181  * heap based on the vruntime by keeping:
1182  *
1183  *  se->min_vruntime = min(se->vruntime, se->{left,right}->min_vruntime)
1184  *
1185  * Which allows tree pruning through eligibility.
1186  */
1187 static struct sched_entity *pick_eevdf(struct cfs_rq *cfs_rq, bool protect)
1188 {
1189 	struct rb_node *node = cfs_rq->tasks_timeline.rb_root.rb_node;
1190 	struct sched_entity *se = __pick_first_entity(cfs_rq);
1191 	struct sched_entity *curr = cfs_rq->curr;
1192 	struct sched_entity *best = NULL;
1193 
1194 	/*
1195 	 * We can safely skip eligibility check if there is only one entity
1196 	 * in this cfs_rq, saving some cycles.
1197 	 */
1198 	if (cfs_rq->h_nr_queued == 1)
1199 		return curr && curr->on_rq ? curr : se;
1200 
1201 	/*
1202 	 * Picking the ->next buddy will affect latency but not fairness.
1203 	 */
1204 	if (sched_feat(PICK_BUDDY) && protect &&
1205 	    cfs_rq->next && entity_eligible(cfs_rq, cfs_rq->next)) {
1206 		/* ->next will never be delayed */
1207 		WARN_ON_ONCE(cfs_rq->next->sched_delayed);
1208 		return cfs_rq->next;
1209 	}
1210 
1211 	if (curr && (!curr->on_rq || !entity_eligible(cfs_rq, curr)))
1212 		curr = NULL;
1213 
1214 	if (curr && protect && protect_slice(curr))
1215 		return curr;
1216 
1217 	/* Pick the leftmost entity if it's eligible */
1218 	if (se && entity_eligible(cfs_rq, se)) {
1219 		best = se;
1220 		goto found;
1221 	}
1222 
1223 	/* Heap search for the EEVD entity */
1224 	while (node) {
1225 		struct rb_node *left = node->rb_left;
1226 
1227 		/*
1228 		 * Eligible entities in left subtree are always better
1229 		 * choices, since they have earlier deadlines.
1230 		 */
1231 		if (left && vruntime_eligible(cfs_rq,
1232 					__node_2_se(left)->min_vruntime)) {
1233 			node = left;
1234 			continue;
1235 		}
1236 
1237 		se = __node_2_se(node);
1238 
1239 		/*
1240 		 * The left subtree either is empty or has no eligible
1241 		 * entity, so check the current node since it is the one
1242 		 * with earliest deadline that might be eligible.
1243 		 */
1244 		if (entity_eligible(cfs_rq, se)) {
1245 			best = se;
1246 			break;
1247 		}
1248 
1249 		node = node->rb_right;
1250 	}
1251 found:
1252 	if (!best || (curr && entity_before(curr, best)))
1253 		best = curr;
1254 
1255 	return best;
1256 }
1257 
1258 struct sched_entity *__pick_last_entity(struct cfs_rq *cfs_rq)
1259 {
1260 	struct rb_node *last = rb_last(&cfs_rq->tasks_timeline.rb_root);
1261 
1262 	if (!last)
1263 		return NULL;
1264 
1265 	return __node_2_se(last);
1266 }
1267 
1268 /**************************************************************
1269  * Scheduling class statistics methods:
1270  */
1271 int sched_update_scaling(void)
1272 {
1273 	unsigned int factor = get_update_sysctl_factor();
1274 
1275 #define WRT_SYSCTL(name) \
1276 	(normalized_sysctl_##name = sysctl_##name / (factor))
1277 	WRT_SYSCTL(sched_base_slice);
1278 #undef WRT_SYSCTL
1279 
1280 	return 0;
1281 }
1282 
1283 static void clear_buddies(struct cfs_rq *cfs_rq, struct sched_entity *se);
1284 
1285 /*
1286  * XXX: strictly: vd_i += N*r_i/w_i such that: vd_i > ve_i
1287  * this is probably good enough.
1288  */
1289 static bool update_deadline(struct cfs_rq *cfs_rq, struct sched_entity *se)
1290 {
1291 	if (vruntime_cmp(se->vruntime, "<", se->deadline))
1292 		return false;
1293 
1294 	/*
1295 	 * For EEVDF the virtual time slope is determined by w_i (iow.
1296 	 * nice) while the request time r_i is determined by
1297 	 * sysctl_sched_base_slice.
1298 	 */
1299 	if (!se->custom_slice)
1300 		se->slice = sysctl_sched_base_slice;
1301 
1302 	/*
1303 	 * EEVDF: vd_i = ve_i + r_i / w_i
1304 	 */
1305 	se->deadline = se->vruntime + calc_delta_fair(se->slice, se);
1306 	avg_vruntime(cfs_rq);
1307 
1308 	/*
1309 	 * The task has consumed its request, reschedule.
1310 	 */
1311 	return true;
1312 }
1313 
1314 #include "pelt.h"
1315 
1316 static int select_idle_sibling(struct task_struct *p, int prev_cpu, int cpu);
1317 static unsigned long task_h_load(struct task_struct *p);
1318 static unsigned long capacity_of(int cpu);
1319 
1320 /* Give new sched_entity start runnable values to heavy its load in infant time */
1321 void init_entity_runnable_average(struct sched_entity *se)
1322 {
1323 	struct sched_avg *sa = &se->avg;
1324 
1325 	memset(sa, 0, sizeof(*sa));
1326 
1327 	/*
1328 	 * Tasks are initialized with full load to be seen as heavy tasks until
1329 	 * they get a chance to stabilize to their real load level.
1330 	 * Group entities are initialized with zero load to reflect the fact that
1331 	 * nothing has been attached to the task group yet.
1332 	 */
1333 	if (entity_is_task(se))
1334 		sa->load_avg = scale_load_down(se->load.weight);
1335 
1336 	/* when this task is enqueued, it will contribute to its cfs_rq's load_avg */
1337 }
1338 
1339 /*
1340  * With new tasks being created, their initial util_avgs are extrapolated
1341  * based on the cfs_rq's current util_avg:
1342  *
1343  *   util_avg = cfs_rq->avg.util_avg / (cfs_rq->avg.load_avg + 1)
1344  *		* se_weight(se)
1345  *
1346  * However, in many cases, the above util_avg does not give a desired
1347  * value. Moreover, the sum of the util_avgs may be divergent, such
1348  * as when the series is a harmonic series.
1349  *
1350  * To solve this problem, we also cap the util_avg of successive tasks to
1351  * only 1/2 of the left utilization budget:
1352  *
1353  *   util_avg_cap = (cpu_scale - cfs_rq->avg.util_avg) / 2^n
1354  *
1355  * where n denotes the nth task and cpu_scale the CPU capacity.
1356  *
1357  * For example, for a CPU with 1024 of capacity, a simplest series from
1358  * the beginning would be like:
1359  *
1360  *  task  util_avg: 512, 256, 128,  64,  32,   16,    8, ...
1361  * cfs_rq util_avg: 512, 768, 896, 960, 992, 1008, 1016, ...
1362  *
1363  * Finally, that extrapolated util_avg is clamped to the cap (util_avg_cap)
1364  * if util_avg > util_avg_cap.
1365  */
1366 void post_init_entity_util_avg(struct task_struct *p)
1367 {
1368 	struct sched_entity *se = &p->se;
1369 	struct cfs_rq *cfs_rq = cfs_rq_of(se);
1370 	struct sched_avg *sa = &se->avg;
1371 	long cpu_scale = arch_scale_cpu_capacity(cpu_of(rq_of(cfs_rq)));
1372 	long cap = (long)(cpu_scale - cfs_rq->avg.util_avg) / 2;
1373 
1374 	if (p->sched_class != &fair_sched_class) {
1375 		/*
1376 		 * For !fair tasks do:
1377 		 *
1378 		update_cfs_rq_load_avg(now, cfs_rq);
1379 		attach_entity_load_avg(cfs_rq, se);
1380 		switched_from_fair(rq, p);
1381 		 *
1382 		 * such that the next switched_to_fair() has the
1383 		 * expected state.
1384 		 */
1385 		se->avg.last_update_time = cfs_rq_clock_pelt(cfs_rq);
1386 		return;
1387 	}
1388 
1389 	if (cap > 0) {
1390 		if (cfs_rq->avg.util_avg != 0) {
1391 			sa->util_avg  = cfs_rq->avg.util_avg * se_weight(se);
1392 			sa->util_avg /= (cfs_rq->avg.load_avg + 1);
1393 
1394 			if (sa->util_avg > cap)
1395 				sa->util_avg = cap;
1396 		} else {
1397 			sa->util_avg = cap;
1398 		}
1399 	}
1400 
1401 	sa->runnable_avg = sa->util_avg;
1402 }
1403 
1404 static inline void account_mm_sched(struct rq *rq, struct task_struct *p, s64 delta_exec);
1405 
1406 static s64 update_se(struct rq *rq, struct sched_entity *se)
1407 {
1408 	u64 now = rq_clock_task(rq);
1409 	s64 delta_exec;
1410 
1411 	delta_exec = now - se->exec_start;
1412 	if (unlikely(delta_exec <= 0))
1413 		return delta_exec;
1414 
1415 	se->exec_start = now;
1416 	if (entity_is_task(se)) {
1417 		struct task_struct *running = rq->curr;
1418 		/*
1419 		 * If se is a task, we account the time against the running
1420 		 * task, as w/ proxy-exec they may not be the same.
1421 		 */
1422 		running->se.exec_start = now;
1423 		running->se.sum_exec_runtime += delta_exec;
1424 
1425 		trace_sched_stat_runtime(running, delta_exec);
1426 		account_group_exec_runtime(running, delta_exec);
1427 		account_mm_sched(rq, running, delta_exec);
1428 
1429 		cgroup_account_cputime(running, delta_exec);
1430 	} else {
1431 		/* If not task, account the time against donor se  */
1432 		se->sum_exec_runtime += delta_exec;
1433 	}
1434 
1435 	if (schedstat_enabled()) {
1436 		struct sched_statistics *stats;
1437 
1438 		stats = __schedstats_from_se(se);
1439 		__schedstat_set(stats->exec_max,
1440 				max(delta_exec, stats->exec_max));
1441 	}
1442 
1443 	return delta_exec;
1444 }
1445 
1446 #ifdef CONFIG_SCHED_CACHE
1447 
1448 /*
1449  * XXX numbers come from a place the sun don't shine -- probably wants to be SD
1450  * tunable or so.
1451  */
1452 #define EPOCH_PERIOD	(HZ / 100)	/* 10 ms */
1453 #define EPOCH_LLC_AFFINITY_TIMEOUT	5	/* 50 ms */
1454 __read_mostly unsigned int llc_aggr_tolerance	= 1;
1455 __read_mostly unsigned int llc_epoch_period	= EPOCH_PERIOD;
1456 __read_mostly unsigned int llc_epoch_affinity_timeout = EPOCH_LLC_AFFINITY_TIMEOUT;
1457 __read_mostly unsigned int llc_imb_pct		= 20;
1458 __read_mostly unsigned int llc_overaggr_pct	= 50;
1459 
1460 static int llc_id(int cpu)
1461 {
1462 	if (cpu < 0)
1463 		return -1;
1464 
1465 	return per_cpu(sd_llc_id, cpu);
1466 }
1467 
1468 static inline int get_sched_cache_scale(int mul)
1469 {
1470 	unsigned int tol = READ_ONCE(llc_aggr_tolerance);
1471 
1472 	if (!tol)
1473 		return 0;
1474 
1475 	if (tol >= 100)
1476 		return INT_MAX;
1477 
1478 	return (1 + (tol - 1) * mul);
1479 }
1480 
1481 static bool exceed_llc_capacity(struct sched_cache_group *grp, int cpu)
1482 {
1483 #ifdef CONFIG_NUMA_BALANCING
1484 	unsigned long llc, footprint;
1485 	struct sched_domain *sd;
1486 	int scale;
1487 
1488 	guard(rcu)();
1489 
1490 	sd = rcu_dereference_sched_domain(cpu_rq(cpu)->sd);
1491 	if (!sd)
1492 		return true;
1493 
1494 	if (static_branch_likely(&sched_numa_balancing)) {
1495 		/*
1496 		 * TBD: RDT exclusive LLC ways reserved should be
1497 		 * excluded.
1498 		 */
1499 		llc = sd->llc_bytes;
1500 		footprint = READ_ONCE(grp->footprint);
1501 
1502 		/*
1503 		 * Scale the LLC size by 256*llc_aggr_tolerance
1504 		 * and compare it to the task's footprint.
1505 		 *
1506 		 * Suppose the L3 size is 32MB. If the
1507 		 * llc_aggr_tolerance is 1:
1508 		 * When the footprint is larger than 32MB, the
1509 		 * process is regarded as exceeding the LLC
1510 		 * capacity. If the llc_aggr_tolerance is 99:
1511 		 * When the footprint is larger than 784GB, the
1512 		 * process is regarded as exceeding the LLC
1513 		 * capacity:
1514 		 * 784GB = (1 + (99 - 1) * 256) * 32MB
1515 		 * If the llc_aggr_tolerance is 100:
1516 		 * ignore the footprint and do the aggregation
1517 		 * anyway.
1518 		 */
1519 		scale = get_sched_cache_scale(256);
1520 		if (scale == INT_MAX)
1521 			return false;
1522 
1523 		return ((llc * (u64)scale) < (footprint * PAGE_SIZE));
1524 	}
1525 #endif
1526 	return false;
1527 }
1528 
1529 static bool invalid_llc_nr(struct sched_cache_group *grp, struct task_struct *p,
1530 			   int cpu)
1531 {
1532 	int scale;
1533 
1534 	if (get_nr_threads(p) <= 1)
1535 		return true;
1536 
1537 	/*
1538 	 * Scale the number of 'cores' in a LLC by llc_aggr_tolerance
1539 	 * and compare it to the task's active threads.
1540 	 */
1541 	scale = get_sched_cache_scale(1);
1542 	if (scale == INT_MAX)
1543 		return false;
1544 
1545 	return !fits_capacity((READ_ONCE(grp->nr_running_avg) * cpu_smt_num_threads),
1546 			(scale * per_cpu(sd_llc_size, cpu)));
1547 }
1548 
1549 /*
1550  * A task counts in nr_pref_llc_running while it is queued on its preferred
1551  * LLC (pref_llc_queued) and runnable (!sched_delayed), keeping the counter in
1552  * the runnable domain so alb_break_llc() can compare it with h_nr_runnable.
1553  */
1554 static bool task_pref_llc_runnable(struct task_struct *p)
1555 {
1556 	return p->pref_llc_queued && !p->se.sched_delayed;
1557 }
1558 
1559 static void pref_llc_running_inc(struct rq *rq, struct task_struct *p)
1560 {
1561 	if (task_pref_llc_runnable(p))
1562 		rq->nr_pref_llc_running++;
1563 }
1564 
1565 static void pref_llc_running_dec(struct rq *rq, struct task_struct *p)
1566 {
1567 	if (task_pref_llc_runnable(p))
1568 		rq->nr_pref_llc_running--;
1569 }
1570 
1571 static void account_llc_enqueue(struct rq *rq, struct task_struct *p)
1572 {
1573 	int pref_llc, pref_llc_queued;
1574 	struct sched_domain *sd;
1575 
1576 	pref_llc = p->preferred_llc;
1577 	if (pref_llc < 0)
1578 		return;
1579 
1580 	pref_llc_queued = (pref_llc == task_llc(p));
1581 	rq->nr_llc_running++;
1582 
1583 	/*
1584 	 * Record whether p is enqueued on its preferred
1585 	 * LLC, in order to pair with account_llc_dequeue()
1586 	 * to maintain a consistent nr_pref_llc_running per
1587 	 * runqueue.
1588 	 * This is necessary because a race condition exists:
1589 	 * after a task is enqueued on a runqueue, task_llc(p)
1590 	 * may change due to CPU hotplug. Therefore, checking
1591 	 * task_llc(p) to determine whether the task is being
1592 	 * dequeued from its preferred LLC is unreliable and
1593 	 * can cause inconsistent values - checking the
1594 	 * p->pref_llc_queued in account_llc_dequeue() would
1595 	 * be reliable.
1596 	 */
1597 	p->pref_llc_queued = pref_llc_queued;
1598 
1599 	/* Skipped while delayed; clear_delayed() adds it back on wake. */
1600 	pref_llc_running_inc(rq, p);
1601 
1602 	sd = rcu_dereference_all(rq->sd);
1603 	if (sd && (unsigned int)pref_llc < sd->llc_max)
1604 		sd->llc_counts[pref_llc]++;
1605 }
1606 
1607 static void account_llc_dequeue(struct rq *rq, struct task_struct *p)
1608 {
1609 	struct sched_domain *sd;
1610 	int pref_llc;
1611 
1612 	pref_llc = p->preferred_llc;
1613 	if (pref_llc < 0)
1614 		return;
1615 
1616 	rq->nr_llc_running--;
1617 	if (p->pref_llc_queued) {
1618 		/*
1619 		 * Skipped if still delayed (set_delayed() already removed it);
1620 		 * clearing pref_llc_queued below also stops clear_delayed()
1621 		 * from re-adding it.
1622 		 */
1623 		pref_llc_running_dec(rq, p);
1624 		/*
1625 		 * Update the status in case
1626 		 * other logic might query
1627 		 * this.
1628 		 */
1629 		p->pref_llc_queued = 0;
1630 	}
1631 
1632 	sd = rcu_dereference_all(rq->sd);
1633 	if (sd && (unsigned int)pref_llc < sd->llc_max) {
1634 		/*
1635 		 * There is a race condition between dequeue
1636 		 * and CPU hotplug. After a task has been enqueued
1637 		 * on CPUx, a CPU hotplug event occurs, and all online
1638 		 * CPUs (including CPUx) rebuild their sched_domains
1639 		 * and reset statistics to zero(including sd->llc_counts).
1640 		 * This can cause temporary undercount and we have to
1641 		 * check for such underflow in sd->llc_counts.
1642 		 *
1643 		 * This undercount is temporary and accurate accounting
1644 		 * will resume once the rq has a chance to be idle.
1645 		 */
1646 		if (sd->llc_counts[pref_llc])
1647 			sd->llc_counts[pref_llc]--;
1648 	}
1649 }
1650 
1651 int mm_init_sched(struct mm_struct *mm,
1652 		  struct sched_cache_time __percpu *_pcpu_sched)
1653 {
1654 	struct sched_cache_group *grp;
1655 	unsigned long epoch = 0;
1656 	int i;
1657 
1658 	grp = kzalloc_obj(*grp);
1659 	if (!grp) {
1660 		free_percpu(_pcpu_sched);
1661 		mm->sched_cache_grp = NULL;
1662 		return -ENOMEM;
1663 	}
1664 
1665 	for_each_possible_cpu(i) {
1666 		struct sched_cache_time *pcpu_sched = per_cpu_ptr(_pcpu_sched, i);
1667 		struct rq *rq = cpu_rq(i);
1668 
1669 		pcpu_sched->runtime = 0;
1670 		/* a slightly stale cpu epoch is acceptible */
1671 		pcpu_sched->epoch = rq->cpu_epoch;
1672 		epoch = rq->cpu_epoch;
1673 	}
1674 
1675 	raw_spin_lock_init(&grp->lock);
1676 	grp->epoch = epoch;
1677 	grp->cpu = -1;
1678 	grp->next_scan = jiffies;
1679 	grp->nr_running_avg = 0;
1680 	grp->footprint = 0;
1681 	refcount_set(&grp->refcnt, 1);
1682 	/*
1683 	 * The update to grp->pcpu_sched should not be reordered
1684 	 * before initialization to grp's other fields, in case
1685 	 * the readers may get invalid mm_sched_epoch, etc.
1686 	 */
1687 	smp_store_release(&grp->pcpu_sched, _pcpu_sched);
1688 	/*
1689 	 * Publish the group last.  Not every reader qualifies it by
1690 	 * grp->pcpu_sched - can_migrate_llc_task() only checks that the
1691 	 * pointer is non-NULL before reading grp->footprint and
1692 	 * grp->nr_running_avg - so a reachable group must already be
1693 	 * fully initialized.
1694 	 */
1695 	smp_store_release(&mm->sched_cache_grp, grp);
1696 	return 0;
1697 }
1698 
1699 static void sched_cache_group_free_rcu(struct rcu_head *rcu)
1700 {
1701 	struct sched_cache_group *grp =
1702 		container_of(rcu, struct sched_cache_group, rcu);
1703 
1704 	free_percpu(grp->pcpu_sched);
1705 	kfree(grp);
1706 }
1707 
1708 static void sched_cache_group_put(struct sched_cache_group *grp)
1709 {
1710 	if (!grp || !refcount_dec_and_test(&grp->refcnt))
1711 		return;
1712 
1713 	call_rcu(&grp->rcu, sched_cache_group_free_rcu);
1714 }
1715 
1716 DEFINE_FREE(sched_cache_group_put, struct sched_cache_group *,
1717 	    sched_cache_group_put(_T));
1718 
1719 #define rcu_deref_sched_cache_grp(tsk) \
1720 	rcu_dereference_check((tsk)->sched_cache_grp, (tsk) == current)
1721 
1722 static struct sched_cache_group *sched_cache_replace_grp(struct task_struct *p,
1723 							 struct sched_cache_group *new)
1724 {
1725 	struct sched_cache_group *old;
1726 
1727 	old = rcu_deref_sched_cache_grp(p);
1728 	rcu_assign_pointer(p->sched_cache_grp, new);
1729 
1730 	return old;
1731 }
1732 
1733 struct sched_cache_group *sched_cache_group_get(struct sched_cache_group *grp)
1734 {
1735 	/*
1736 	 * refcount_inc_not_zero() is the acquire primitive for lockless
1737 	 * (RCU) lookups; plain refcount_inc() would scribble the count if
1738 	 * it already reached zero. Return NULL in that case.
1739 	 */
1740 	if (grp && !refcount_inc_not_zero(&grp->refcnt))
1741 		grp = NULL;
1742 
1743 	return grp;
1744 }
1745 
1746 struct sched_cache_group *task_cache_group_get(struct task_struct *p)
1747 {
1748 	guard(rcu)();
1749 	return sched_cache_group_get(rcu_dereference(p->sched_cache_grp));
1750 }
1751 
1752 void sched_cache_fork(struct task_struct *p)
1753 {
1754 	/*
1755 	 * The child takes its own reference on the mm's cache group, separate
1756 	 * from the reference held by the mm. @p is not yet visible to readers,
1757 	 * so a plain initializing store is enough.
1758 	 */
1759 	RCU_INIT_POINTER(p->sched_cache_grp,
1760 			 sched_cache_group_get(p->mm->sched_cache_grp));
1761 }
1762 
1763 void sched_cache_fork_cleanup(struct task_struct *p)
1764 {
1765 	/*
1766 	 * A fork that fails after sched_cache_fork() never reaches exit_mm(),
1767 	 * so drop the reference here. @p never became visible, so there are no
1768 	 * concurrent readers and the reference we hold keeps the group alive.
1769 	 */
1770 	sched_cache_group_put(rcu_access_pointer(p->sched_cache_grp));
1771 	RCU_INIT_POINTER(p->sched_cache_grp, NULL);
1772 }
1773 
1774 void sched_cache_exec_mmap(struct task_struct *p, struct mm_struct *mm)
1775 {
1776 	struct sched_cache_group *old;
1777 
1778 	/*
1779 	 * Acquire the new reference before publishing the pointer, then drop
1780 	 * the old one. @p is current and the only writer of its own pointer.
1781 	 */
1782 	old = sched_cache_replace_grp(p, sched_cache_group_get(mm->sched_cache_grp));
1783 	sched_cache_group_put(old);
1784 }
1785 
1786 void sched_cache_exit_mm(struct task_struct *p)
1787 {
1788 	struct sched_cache_group *grp = sched_cache_replace_grp(p, NULL);
1789 
1790 #ifdef CONFIG_NUMA_BALANCING
1791 	/*
1792 	 * Subtract this task's footprint from the group before dropping the
1793 	 * reference, so the group footprint converges as its threads exit.
1794 	 * Unlocked for performance; clamp to avoid underflow.
1795 	 */
1796 	if (grp && p->total_numa_faults) {
1797 		unsigned long fp = READ_ONCE(grp->footprint);
1798 		unsigned long sub = min(fp, p->total_numa_faults);
1799 
1800 		WRITE_ONCE(grp->footprint, fp - sub);
1801 	}
1802 #endif
1803 	sched_cache_group_put(grp);
1804 }
1805 
1806 void mm_destroy_sched(struct mm_struct *mm)
1807 {
1808 	sched_cache_group_put(mm->sched_cache_grp);
1809 	mm->sched_cache_grp = NULL;
1810 }
1811 
1812 /* because why would C be fully specified */
1813 static __always_inline void __shr_u64(u64 *val, unsigned int n)
1814 {
1815 	if (n >= 64) {
1816 		*val = 0;
1817 		return;
1818 	}
1819 	*val >>= n;
1820 }
1821 
1822 static inline void __update_mm_sched(struct rq *rq,
1823 				     struct sched_cache_time *pcpu_sched)
1824 {
1825 	lockdep_assert_held(&rq->cpu_epoch_lock);
1826 
1827 	unsigned int period = max(READ_ONCE(llc_epoch_period), 1U);
1828 	unsigned long n, now = jiffies;
1829 	long delta = now - rq->cpu_epoch_next;
1830 
1831 	if (delta > 0) {
1832 		n = (delta + period - 1) / period;
1833 		rq->cpu_epoch += n;
1834 		rq->cpu_epoch_next += n * period;
1835 		__shr_u64(&rq->cpu_runtime, n);
1836 	}
1837 
1838 	n = rq->cpu_epoch - pcpu_sched->epoch;
1839 	if (n) {
1840 		pcpu_sched->epoch += n;
1841 		__shr_u64(&pcpu_sched->runtime, n);
1842 	}
1843 }
1844 
1845 static unsigned long fraction_mm_sched(struct rq *rq,
1846 				       struct sched_cache_time *pcpu_sched)
1847 {
1848 	guard(raw_spinlock_irqsave)(&rq->cpu_epoch_lock);
1849 
1850 	__update_mm_sched(rq, pcpu_sched);
1851 
1852 	/*
1853 	 * Runtime is a geometric series (r=0.5) and as such will sum to twice
1854 	 * the accumulation period, this means the multiplcation here should
1855 	 * not overflow.
1856 	 */
1857 	return div64_u64(NICE_0_LOAD * pcpu_sched->runtime, rq->cpu_runtime + 1);
1858 }
1859 
1860 static int get_pref_llc(struct task_struct *p, struct sched_cache_group *grp)
1861 {
1862 	int mm_sched_llc = -1, mm_sched_cpu;
1863 
1864 	if (!grp)
1865 		return -1;
1866 
1867 	mm_sched_cpu = READ_ONCE(grp->cpu);
1868 	if (mm_sched_cpu != -1) {
1869 		mm_sched_llc = llc_id(mm_sched_cpu);
1870 
1871 #ifdef CONFIG_NUMA_BALANCING
1872 		/*
1873 		 * Don't assign preferred LLC if it
1874 		 * conflicts with NUMA balancing.
1875 		 * This can happen when sched_setnuma() gets
1876 		 * called, however it is not much of an issue
1877 		 * because we expect account_mm_sched() to get
1878 		 * called fairly regularly -- at a higher rate
1879 		 * than sched_setnuma() at least -- and thus the
1880 		 * conflict only exists for a short period of time.
1881 		 */
1882 		if (static_branch_likely(&sched_numa_balancing) &&
1883 		    p->numa_preferred_nid >= 0 &&
1884 		    cpu_to_node(mm_sched_cpu) != p->numa_preferred_nid)
1885 			mm_sched_llc = -1;
1886 #endif
1887 	}
1888 
1889 	return mm_sched_llc;
1890 }
1891 
1892 static unsigned int task_running_on_cpu(int cpu, struct task_struct *p);
1893 
1894 static inline
1895 void account_mm_sched(struct rq *rq, struct task_struct *p, s64 delta_exec)
1896 {
1897 	struct sched_cache_group *grp = rcu_dereference_all(p->sched_cache_grp);
1898 	struct sched_cache_time *pcpu_sched;
1899 	int mm_sched_llc = -1;
1900 	unsigned long epoch;
1901 
1902 	if (!sched_cache_enabled())
1903 		return;
1904 
1905 	if (p->sched_class != &fair_sched_class)
1906 		return;
1907 	/*
1908 	 * init_task, kthreads and user thread created
1909 	 * by user_mode_thread() don't have a cache group.
1910 	 * In theory a kernel thread does not have any valid
1911 	 * cache group, because sched_cache_fork() is not
1912 	 * invoked for a kernel thread - !grp should gate the
1913 	 * kernel thread. Use the PF_KTHREAD check explicitly
1914 	 * here for safety reasons, to guard against future
1915 	 * modifications and to pair with task_tick_cache().
1916 	 */
1917 	if (p->flags & PF_KTHREAD || !grp || !grp->pcpu_sched)
1918 		return;
1919 
1920 	pcpu_sched = per_cpu_ptr(grp->pcpu_sched, cpu_of(rq));
1921 
1922 	scoped_guard (raw_spinlock, &rq->cpu_epoch_lock) {
1923 		__update_mm_sched(rq, pcpu_sched);
1924 		pcpu_sched->runtime += delta_exec;
1925 		rq->cpu_runtime += delta_exec;
1926 		epoch = rq->cpu_epoch;
1927 	}
1928 
1929 	/*
1930 	 * If this process hasn't hit task_cache_work() for a while invalidate
1931 	 * its preferred state.
1932 	 */
1933 	if ((long)(epoch - READ_ONCE(grp->epoch)) > llc_epoch_affinity_timeout ||
1934 	    invalid_llc_nr(grp, p, cpu_of(rq)) ||
1935 	    exceed_llc_capacity(grp, cpu_of(rq))) {
1936 		if (READ_ONCE(grp->cpu) != -1)
1937 			WRITE_ONCE(grp->cpu, -1);
1938 	}
1939 
1940 	mm_sched_llc = get_pref_llc(p, grp);
1941 
1942 	/* task not on rq accounted later in account_entity_enqueue() */
1943 	if (task_running_on_cpu(rq->cpu, p) &&
1944 	    READ_ONCE(p->preferred_llc) != mm_sched_llc) {
1945 		account_llc_dequeue(rq, p);
1946 		WRITE_ONCE(p->preferred_llc, mm_sched_llc);
1947 		account_llc_enqueue(rq, p);
1948 	}
1949 }
1950 
1951 static void task_tick_cache(struct rq *rq, struct task_struct *p)
1952 {
1953 	struct sched_cache_group *grp = rcu_dereference_all(p->sched_cache_grp);
1954 	struct callback_head *work = &p->cache_work;
1955 	unsigned long epoch;
1956 
1957 	if (!sched_cache_enabled())
1958 		return;
1959 
1960 	if (!grp || p->flags & PF_KTHREAD ||
1961 	    !grp->pcpu_sched)
1962 		return;
1963 
1964 	epoch = rq->cpu_epoch;
1965 	/* avoid moving backwards */
1966 	if (time_after_eq(grp->epoch, epoch))
1967 		return;
1968 
1969 	guard(raw_spinlock)(&grp->lock);
1970 
1971 	if (work->next == work) {
1972 		task_work_add(p, work, TWA_RESUME);
1973 		WRITE_ONCE(grp->epoch, epoch);
1974 	}
1975 }
1976 
1977 static void get_scan_cpumasks(cpumask_var_t cpus, struct task_struct *p,
1978 			      struct sched_cache_group *grp)
1979 {
1980 #ifdef CONFIG_NUMA_BALANCING
1981 	int cpu, curr_cpu, nid, pref_nid;
1982 
1983 	if (!static_branch_likely(&sched_numa_balancing))
1984 		goto out;
1985 
1986 	cpu = READ_ONCE(grp->cpu);
1987 	if (cpu != -1)
1988 		nid = cpu_to_node(cpu);
1989 	curr_cpu = task_cpu(p);
1990 
1991 	/*
1992 	 * Scanning in the preferred NUMA node is ideal. However, the NUMA
1993 	 * preferred node is per-task rather than per-process. It is possible
1994 	 * for different threads of the process to have distinct preferred
1995 	 * nodes; consequently, the process-wide preferred LLC may bounce
1996 	 * between different nodes. As a workaround, maintain the scan
1997 	 * CPU mask to also cover the process's current preferred LLC and the
1998 	 * current running node to mitigate the bouncing risk.
1999 	 * TBD: numa_group should be considered during task aggregation.
2000 	 */
2001 	pref_nid = p->numa_preferred_nid;
2002 	/* honor the task's preferred node */
2003 	if (pref_nid == NUMA_NO_NODE)
2004 		goto out;
2005 
2006 	cpumask_or(cpus, cpus, cpumask_of_node(pref_nid));
2007 
2008 	/* honor the task's preferred LLC CPU */
2009 	if (cpu != -1 && !cpumask_test_cpu(cpu, cpus) && nid != NUMA_NO_NODE)
2010 		cpumask_or(cpus, cpus, cpumask_of_node(nid));
2011 
2012 	/* make sure the task's current running node is included */
2013 	if (!cpumask_test_cpu(curr_cpu, cpus))
2014 		cpumask_or(cpus, cpus, cpumask_of_node(cpu_to_node(curr_cpu)));
2015 
2016 	return;
2017 
2018 out:
2019 #endif
2020 	cpumask_copy(cpus, cpu_online_mask);
2021 }
2022 
2023 static inline void update_avg_scale(u64 *avg, u64 sample)
2024 {
2025 	int factor = per_cpu(sd_llc_size, raw_smp_processor_id());
2026 	s64 diff = sample - *avg;
2027 	u32 divisor;
2028 
2029 	/*
2030 	 * Scale the divisor based on the number of CPUs contained
2031 	 * in the LLC. This scaling ensures smaller LLC domains use
2032 	 * a smaller divisor to achieve more precise sensitivity to
2033 	 * changes in nr_running, while larger LLC domains are capped
2034 	 * at a maximum divisor of 8 which is the default smoothing
2035 	 * factor of EWMA in update_avg().
2036 	 */
2037 	divisor = clamp_t(u32, (factor >> 2), 2, 8);
2038 	*avg += div64_s64(diff, divisor);
2039 }
2040 
2041 static void task_cache_work(struct callback_head *work)
2042 {
2043 	struct sched_cache_group *grp __free(sched_cache_group_put) = NULL;
2044 	cpumask_var_t cpus __free(free_cpumask_var) = CPUMASK_VAR_NULL;
2045 	int cpu, m_a_cpu = -1, nr_running = 0, curr_cpu;
2046 	unsigned long next_scan, now = jiffies;
2047 	struct task_struct *p = current, *cur;
2048 	unsigned long curr_m_a_occ = 0;
2049 	unsigned long m_a_occ = 0;
2050 
2051 	WARN_ON_ONCE(work != &p->cache_work);
2052 
2053 	work->next = work;
2054 
2055 	if (p->flags & PF_EXITING)
2056 		return;
2057 
2058 	/*
2059 	 * A reference makes sure grp is not released by others. The rcu
2060 	 * lock can not be held till after zalloc_cpumask_var() below,
2061 	 * because the latter might sleep.
2062 	 */
2063 	grp = task_cache_group_get(p);
2064 	if (!grp)
2065 		return;
2066 
2067 	next_scan = READ_ONCE(grp->next_scan);
2068 	if (time_before(now, next_scan))
2069 		return;
2070 
2071 	/* only 1 thread is allowed to scan */
2072 	if (!try_cmpxchg(&grp->next_scan, &next_scan,
2073 			 now + max_t(unsigned long,
2074 				     READ_ONCE(llc_epoch_period), 1)))
2075 		return;
2076 
2077 	curr_cpu = task_cpu(p);
2078 	if (invalid_llc_nr(grp, p, curr_cpu) ||
2079 	    exceed_llc_capacity(grp, curr_cpu)) {
2080 		if (READ_ONCE(grp->cpu) != -1)
2081 			WRITE_ONCE(grp->cpu, -1);
2082 
2083 		return;
2084 	}
2085 
2086 	if (!zalloc_cpumask_var(&cpus, GFP_KERNEL))
2087 		return;
2088 
2089 	scoped_guard (cpus_read_lock) {
2090 		guard(rcu)();
2091 
2092 		get_scan_cpumasks(cpus, p, grp);
2093 
2094 		for_each_cpu(cpu, cpus) {
2095 			/* XXX sched_cluster_active */
2096 			struct sched_domain *sd = rcu_dereference_all(per_cpu(sd_llc, cpu));
2097 			unsigned long occ, m_occ = 0, a_occ = 0;
2098 			int m_cpu = -1, i;
2099 
2100 			if (!sd)
2101 				continue;
2102 
2103 			for_each_cpu(i, sched_domain_span(sd)) {
2104 				occ = fraction_mm_sched(cpu_rq(i),
2105 							per_cpu_ptr(grp->pcpu_sched, i));
2106 				a_occ += occ;
2107 				if (occ > m_occ) {
2108 					m_occ = occ;
2109 					m_cpu = i;
2110 				}
2111 
2112 				/*
2113 				 * rcu_access_pointer() is used because the
2114 				 * pointer is only compared, never dereferenced.
2115 				 */
2116 				cur = rcu_dereference_all(cpu_rq(i)->curr);
2117 				if (cur && !(cur->flags & (PF_EXITING | PF_KTHREAD)) &&
2118 				    rcu_access_pointer(cur->sched_cache_grp) == grp)
2119 					nr_running++;
2120 			}
2121 
2122 			/*
2123 			 * Compare the accumulated occupancy of each LLC. The
2124 			 * reason for using accumulated occupancy rather than average
2125 			 * per CPU occupancy is that it works better in asymmetric LLC
2126 			 * scenarios.
2127 			 * For example, if there are 2 threads in a 4CPU LLC and 3
2128 			 * threads in an 8CPU LLC, it might be better to choose the one
2129 			 * with 3 threads. However, this would not be the case if the
2130 			 * occupancy is divided by the number of CPUs in an LLC (i.e.,
2131 			 * if average per CPU occupancy is used).
2132 			 * Besides, NUMA balancing fault statistics behave similarly:
2133 			 * the total number of faults per node is compared rather than
2134 			 * the average number of faults per CPU. This strategy is also
2135 			 * followed here.
2136 			 */
2137 			if (a_occ > m_a_occ) {
2138 				m_a_occ = a_occ;
2139 				m_a_cpu = m_cpu;
2140 			}
2141 
2142 			if (llc_id(cpu) == llc_id(READ_ONCE(grp->cpu)))
2143 				curr_m_a_occ = a_occ;
2144 
2145 			cpumask_andnot(cpus, cpus, sched_domain_span(sd));
2146 		}
2147 	}
2148 
2149 	if (m_a_occ > (2 * curr_m_a_occ)) {
2150 		/*
2151 		 * Avoid switching sched_cache_grp->cpu too fast.
2152 		 * The reason to choose 2X is because:
2153 		 * 1. It is better to keep the preferred LLC stable,
2154 		 *    rather than changing it frequently and cause migrations
2155 		 * 2. 2X means the new preferred LLC has at least 1 more
2156 		 *    busy CPU than the old one(200% vs 100%, eg)
2157 		 * 3. 2X is chosen based on test results, as it delivers
2158 		 *    the optimal performance gain so far.
2159 		 */
2160 		WRITE_ONCE(grp->cpu, m_a_cpu);
2161 	}
2162 
2163 	update_avg_scale(&grp->nr_running_avg, nr_running);
2164 }
2165 
2166 void init_sched_mm(struct task_struct *p)
2167 {
2168 	struct callback_head *work = &p->cache_work;
2169 
2170 	init_task_work(work, task_cache_work);
2171 	work->next = work;
2172 	/*
2173 	 * dup_task_struct() copies the parent's task_struct, including its
2174 	 * sched_cache_grp, for which the child holds no reference.  Clear it
2175 	 * here - before copy_mm() runs - so the child never carries a
2176 	 * borrowed pointer that the fork error path would put.
2177 	 */
2178 	RCU_INIT_POINTER(p->sched_cache_grp, NULL);
2179 	/*
2180 	 * Reset new task's preference to avoid
2181 	 * polluting account_llc_enqueue().
2182 	 */
2183 	p->preferred_llc = -1;
2184 	p->pref_llc_queued = 0;
2185 }
2186 
2187 #else /* CONFIG_SCHED_CACHE */
2188 
2189 static inline void account_mm_sched(struct rq *rq, struct task_struct *p,
2190 				    s64 delta_exec) { }
2191 
2192 void init_sched_mm(struct task_struct *p) { }
2193 
2194 static void task_tick_cache(struct rq *rq, struct task_struct *p) { }
2195 
2196 static inline int get_pref_llc(struct task_struct *p,
2197 			       struct mm_struct *mm)
2198 {
2199 	return -1;
2200 }
2201 
2202 static void account_llc_enqueue(struct rq *rq, struct task_struct *p) {}
2203 
2204 static void account_llc_dequeue(struct rq *rq, struct task_struct *p) {}
2205 
2206 static void pref_llc_running_inc(struct rq *rq, struct task_struct *p) {}
2207 
2208 static void pref_llc_running_dec(struct rq *rq, struct task_struct *p) {}
2209 
2210 #endif /* CONFIG_SCHED_CACHE */
2211 
2212 /*
2213  * Used by other classes to account runtime.
2214  */
2215 s64 update_curr_common(struct rq *rq)
2216 {
2217 	return update_se(rq, &rq->donor->se);
2218 }
2219 
2220 /*
2221  * Update the current task's runtime statistics.
2222  */
2223 static void update_curr(struct cfs_rq *cfs_rq)
2224 {
2225 	/*
2226 	 * Note: cfs_rq->curr corresponds to the task picked to
2227 	 * run (ie: rq->donor.se) which due to proxy-exec may
2228 	 * not necessarily be the actual task running
2229 	 * (rq->curr.se). This is easy to confuse!
2230 	 */
2231 	struct sched_entity *curr = cfs_rq->h_curr;
2232 	struct rq *rq = rq_of(cfs_rq);
2233 	s64 delta_exec;
2234 	bool resched;
2235 
2236 	if (unlikely(!curr))
2237 		return;
2238 
2239 	delta_exec = update_se(rq, curr);
2240 	if (unlikely(delta_exec <= 0))
2241 		return;
2242 
2243 	account_cfs_rq_runtime(cfs_rq, delta_exec);
2244 
2245 	if (!entity_is_task(curr))
2246 		return;
2247 
2248 	cfs_rq = &rq->cfs;
2249 
2250 	curr->vruntime += calc_delta_fair(delta_exec, curr);
2251 	resched = update_deadline(cfs_rq, curr);
2252 
2253 	/*
2254 	 * If the fair_server is active, we need to account for the
2255 	 * fair_server time whether or not the task is running on
2256 	 * behalf of fair_server or not:
2257 	 *  - If the task is running on behalf of fair_server, we need
2258 	 *    to limit its time based on the assigned runtime.
2259 	 *  - Fair task that runs outside of fair_server should account
2260 	 *    against fair_server such that it can account for this time
2261 	 *    and possibly avoid running this period.
2262 	 */
2263 	dl_server_update(&rq->fair_server, delta_exec);
2264 
2265 	if (cfs_rq->h_nr_queued == 1)
2266 		return;
2267 
2268 	if (resched || !protect_slice(curr)) {
2269 		resched_curr_lazy(rq);
2270 		clear_buddies(cfs_rq, curr);
2271 	}
2272 }
2273 
2274 static void update_curr_fair(struct rq *rq)
2275 {
2276 	struct sched_entity *se = &rq->donor->se;
2277 
2278 	for_each_sched_entity(se)
2279 		update_curr(cfs_rq_of(se));
2280 }
2281 
2282 static inline void
2283 update_stats_wait_start_fair(struct cfs_rq *cfs_rq, struct sched_entity *se)
2284 {
2285 	struct sched_statistics *stats;
2286 	struct task_struct *p = NULL;
2287 
2288 	if (!schedstat_enabled())
2289 		return;
2290 
2291 	stats = __schedstats_from_se(se);
2292 
2293 	if (entity_is_task(se))
2294 		p = task_of(se);
2295 
2296 	__update_stats_wait_start(rq_of(cfs_rq), p, stats);
2297 }
2298 
2299 static inline void
2300 update_stats_wait_end_fair(struct cfs_rq *cfs_rq, struct sched_entity *se)
2301 {
2302 	struct sched_statistics *stats;
2303 	struct task_struct *p = NULL;
2304 
2305 	if (!schedstat_enabled())
2306 		return;
2307 
2308 	stats = __schedstats_from_se(se);
2309 
2310 	/*
2311 	 * When the sched_schedstat changes from 0 to 1, some sched se
2312 	 * maybe already in the runqueue, the se->statistics.wait_start
2313 	 * will be 0.So it will let the delta wrong. We need to avoid this
2314 	 * scenario.
2315 	 */
2316 	if (unlikely(!schedstat_val(stats->wait_start)))
2317 		return;
2318 
2319 	if (entity_is_task(se))
2320 		p = task_of(se);
2321 
2322 	__update_stats_wait_end(rq_of(cfs_rq), p, stats);
2323 }
2324 
2325 static inline void
2326 update_stats_enqueue_sleeper_fair(struct cfs_rq *cfs_rq, struct sched_entity *se)
2327 {
2328 	struct sched_statistics *stats;
2329 	struct task_struct *tsk = NULL;
2330 
2331 	if (!schedstat_enabled())
2332 		return;
2333 
2334 	stats = __schedstats_from_se(se);
2335 
2336 	if (entity_is_task(se))
2337 		tsk = task_of(se);
2338 
2339 	__update_stats_enqueue_sleeper(rq_of(cfs_rq), tsk, stats);
2340 }
2341 
2342 /*
2343  * Task is being enqueued - update stats:
2344  */
2345 static inline void
2346 update_stats_enqueue_fair(struct cfs_rq *cfs_rq, struct sched_entity *se, int flags)
2347 {
2348 	if (!schedstat_enabled())
2349 		return;
2350 
2351 	/*
2352 	 * Are we enqueueing a waiting task? (for current tasks
2353 	 * a dequeue/enqueue event is a NOP)
2354 	 */
2355 	if (se != cfs_rq->h_curr)
2356 		update_stats_wait_start_fair(cfs_rq, se);
2357 
2358 	if (flags & ENQUEUE_WAKEUP)
2359 		update_stats_enqueue_sleeper_fair(cfs_rq, se);
2360 }
2361 
2362 static inline void
2363 update_stats_dequeue_fair(struct cfs_rq *cfs_rq, struct sched_entity *se, int flags)
2364 {
2365 
2366 	if (!schedstat_enabled())
2367 		return;
2368 
2369 	/*
2370 	 * Mark the end of the wait period if dequeueing a
2371 	 * waiting task:
2372 	 */
2373 	if (se != cfs_rq->h_curr)
2374 		update_stats_wait_end_fair(cfs_rq, se);
2375 
2376 	if ((flags & DEQUEUE_SLEEP) && entity_is_task(se)) {
2377 		struct task_struct *tsk = task_of(se);
2378 		unsigned int state;
2379 
2380 		/* XXX racy against TTWU */
2381 		state = READ_ONCE(tsk->__state);
2382 		if (state & TASK_INTERRUPTIBLE)
2383 			__schedstat_set(tsk->stats.sleep_start,
2384 				      rq_clock(rq_of(cfs_rq)));
2385 		if (state & TASK_UNINTERRUPTIBLE)
2386 			__schedstat_set(tsk->stats.block_start,
2387 				      rq_clock(rq_of(cfs_rq)));
2388 	}
2389 }
2390 
2391 /*
2392  * We are picking a new current task - update its stats:
2393  */
2394 static inline void
2395 update_stats_curr_start(struct cfs_rq *cfs_rq, struct sched_entity *se)
2396 {
2397 	/*
2398 	 * We are starting a new run period:
2399 	 */
2400 	se->exec_start = rq_clock_task(rq_of(cfs_rq));
2401 }
2402 
2403 /* Check sched_smt_active before calling this to avoid overheads in fastpaths */
2404 static inline bool is_core_idle(int cpu)
2405 {
2406 	int sibling;
2407 
2408 	for_each_cpu(sibling, cpu_smt_mask(cpu)) {
2409 		if (cpu == sibling)
2410 			continue;
2411 
2412 		if (!idle_cpu(sibling))
2413 			return false;
2414 	}
2415 
2416 	return true;
2417 }
2418 
2419 #ifdef CONFIG_NUMA
2420 #define NUMA_IMBALANCE_MIN 2
2421 
2422 static inline long
2423 adjust_numa_imbalance(int imbalance, int dst_running, int imb_numa_nr)
2424 {
2425 	/*
2426 	 * Allow a NUMA imbalance if busy CPUs is less than the maximum
2427 	 * threshold. Above this threshold, individual tasks may be contending
2428 	 * for both memory bandwidth and any shared HT resources.  This is an
2429 	 * approximation as the number of running tasks may not be related to
2430 	 * the number of busy CPUs due to sched_setaffinity.
2431 	 */
2432 	if (dst_running > imb_numa_nr)
2433 		return imbalance;
2434 
2435 	/*
2436 	 * Allow a small imbalance based on a simple pair of communicating
2437 	 * tasks that remain local when the destination is lightly loaded.
2438 	 */
2439 	if (imbalance <= NUMA_IMBALANCE_MIN)
2440 		return 0;
2441 
2442 	return imbalance;
2443 }
2444 #endif /* CONFIG_NUMA */
2445 
2446 #ifdef CONFIG_NUMA_BALANCING
2447 /*
2448  * Approximate time to scan a full NUMA task in ms. The task scan period is
2449  * calculated based on the tasks virtual memory size and
2450  * numa_balancing_scan_size.
2451  */
2452 unsigned int sysctl_numa_balancing_scan_period_min = 1000;
2453 unsigned int sysctl_numa_balancing_scan_period_max = 60000;
2454 
2455 /* Portion of address space to scan in MB */
2456 unsigned int sysctl_numa_balancing_scan_size = 256;
2457 
2458 /* Scan @scan_size MB every @scan_period after an initial @scan_delay in ms */
2459 unsigned int sysctl_numa_balancing_scan_delay = 1000;
2460 
2461 /* The page with hint page fault latency < threshold in ms is considered hot */
2462 unsigned int sysctl_numa_balancing_hot_threshold = MSEC_PER_SEC;
2463 
2464 struct numa_group {
2465 	refcount_t refcount;
2466 
2467 	spinlock_t lock; /* nr_tasks, tasks */
2468 	int nr_tasks;
2469 	pid_t gid;
2470 	int active_nodes;
2471 
2472 	struct rcu_head rcu;
2473 	unsigned long total_faults;
2474 	unsigned long max_faults_cpu;
2475 	/*
2476 	 * faults[] array is split into two regions: faults_mem and faults_cpu.
2477 	 *
2478 	 * Faults_cpu is used to decide whether memory should move
2479 	 * towards the CPU. As a consequence, these stats are weighted
2480 	 * more by CPU use than by memory faults.
2481 	 */
2482 	unsigned long faults[];
2483 };
2484 
2485 /*
2486  * For functions that can be called in multiple contexts that permit reading
2487  * ->numa_group (see struct task_struct for locking rules).
2488  */
2489 static struct numa_group *deref_task_numa_group(struct task_struct *p)
2490 {
2491 	return rcu_dereference_check(p->numa_group, p == current ||
2492 		(lockdep_is_held(__rq_lockp(task_rq(p))) && !READ_ONCE(p->on_cpu)));
2493 }
2494 
2495 static struct numa_group *deref_curr_numa_group(struct task_struct *p)
2496 {
2497 	return rcu_dereference_protected(p->numa_group, p == current);
2498 }
2499 
2500 static inline unsigned long group_faults_priv(struct numa_group *ng);
2501 static inline unsigned long group_faults_shared(struct numa_group *ng);
2502 
2503 static unsigned int task_nr_scan_windows(struct task_struct *p)
2504 {
2505 	unsigned long rss = 0;
2506 	unsigned long nr_scan_pages;
2507 
2508 	/*
2509 	 * Calculations based on RSS as non-present and empty pages are skipped
2510 	 * by the PTE scanner and NUMA hinting faults should be trapped based
2511 	 * on resident pages
2512 	 */
2513 	nr_scan_pages = MB_TO_PAGES(sysctl_numa_balancing_scan_size);
2514 	rss = get_mm_rss(p->mm);
2515 	if (!rss)
2516 		rss = nr_scan_pages;
2517 
2518 	rss = round_up(rss, nr_scan_pages);
2519 	return rss / nr_scan_pages;
2520 }
2521 
2522 /* For sanity's sake, never scan more PTEs than MAX_SCAN_WINDOW MB/sec. */
2523 #define MAX_SCAN_WINDOW 2560
2524 
2525 static unsigned int task_scan_min(struct task_struct *p)
2526 {
2527 	unsigned int scan_size = READ_ONCE(sysctl_numa_balancing_scan_size);
2528 	unsigned int scan, floor;
2529 	unsigned int windows = 1;
2530 
2531 	if (scan_size < MAX_SCAN_WINDOW)
2532 		windows = MAX_SCAN_WINDOW / scan_size;
2533 	floor = 1000 / windows;
2534 
2535 	scan = sysctl_numa_balancing_scan_period_min / task_nr_scan_windows(p);
2536 	return max_t(unsigned int, floor, scan);
2537 }
2538 
2539 static unsigned int task_scan_start(struct task_struct *p)
2540 {
2541 	unsigned long smin = task_scan_min(p);
2542 	unsigned long period = smin;
2543 	struct numa_group *ng;
2544 
2545 	/* Scale the maximum scan period with the amount of shared memory. */
2546 	rcu_read_lock();
2547 	ng = rcu_dereference_all(p->numa_group);
2548 	if (ng) {
2549 		unsigned long shared = group_faults_shared(ng);
2550 		unsigned long private = group_faults_priv(ng);
2551 
2552 		period *= refcount_read(&ng->refcount);
2553 		period *= shared + 1;
2554 		period /= private + shared + 1;
2555 	}
2556 	rcu_read_unlock();
2557 
2558 	return max(smin, period);
2559 }
2560 
2561 static unsigned int task_scan_max(struct task_struct *p)
2562 {
2563 	unsigned long smin = task_scan_min(p);
2564 	unsigned long smax;
2565 	struct numa_group *ng;
2566 
2567 	/* Watch for min being lower than max due to floor calculations */
2568 	smax = sysctl_numa_balancing_scan_period_max / task_nr_scan_windows(p);
2569 
2570 	/* Scale the maximum scan period with the amount of shared memory. */
2571 	ng = deref_curr_numa_group(p);
2572 	if (ng) {
2573 		unsigned long shared = group_faults_shared(ng);
2574 		unsigned long private = group_faults_priv(ng);
2575 		unsigned long period = smax;
2576 
2577 		period *= refcount_read(&ng->refcount);
2578 		period *= shared + 1;
2579 		period /= private + shared + 1;
2580 
2581 		smax = max(smax, period);
2582 	}
2583 
2584 	return max(smin, smax);
2585 }
2586 
2587 static void account_numa_enqueue(struct rq *rq, struct task_struct *p)
2588 {
2589 	rq->nr_numa_running += (p->numa_preferred_nid != NUMA_NO_NODE);
2590 	rq->nr_preferred_running += (p->numa_preferred_nid == task_node(p));
2591 }
2592 
2593 static void account_numa_dequeue(struct rq *rq, struct task_struct *p)
2594 {
2595 	rq->nr_numa_running -= (p->numa_preferred_nid != NUMA_NO_NODE);
2596 	rq->nr_preferred_running -= (p->numa_preferred_nid == task_node(p));
2597 }
2598 
2599 /* Shared or private faults. */
2600 #define NR_NUMA_HINT_FAULT_TYPES 2
2601 
2602 /* Memory and CPU locality */
2603 #define NR_NUMA_HINT_FAULT_STATS (NR_NUMA_HINT_FAULT_TYPES * 2)
2604 
2605 /* Averaged statistics, and temporary buffers. */
2606 #define NR_NUMA_HINT_FAULT_BUCKETS (NR_NUMA_HINT_FAULT_STATS * 2)
2607 
2608 pid_t task_numa_group_id(struct task_struct *p)
2609 {
2610 	struct numa_group *ng;
2611 	pid_t gid = 0;
2612 
2613 	rcu_read_lock();
2614 	ng = rcu_dereference_all(p->numa_group);
2615 	if (ng)
2616 		gid = ng->gid;
2617 	rcu_read_unlock();
2618 
2619 	return gid;
2620 }
2621 
2622 /*
2623  * The averaged statistics, shared & private, memory & CPU,
2624  * occupy the first half of the array. The second half of the
2625  * array is for current counters, which are averaged into the
2626  * first set by task_numa_placement.
2627  */
2628 static inline int task_faults_idx(enum numa_faults_stats s, int nid, int priv)
2629 {
2630 	return NR_NUMA_HINT_FAULT_TYPES * (s * nr_node_ids + nid) + priv;
2631 }
2632 
2633 static inline unsigned long task_faults(struct task_struct *p, int nid)
2634 {
2635 	if (!p->numa_faults)
2636 		return 0;
2637 
2638 	return p->numa_faults[task_faults_idx(NUMA_MEM, nid, 0)] +
2639 		p->numa_faults[task_faults_idx(NUMA_MEM, nid, 1)];
2640 }
2641 
2642 static inline unsigned long group_faults(struct task_struct *p, int nid)
2643 {
2644 	struct numa_group *ng = deref_task_numa_group(p);
2645 
2646 	if (!ng)
2647 		return 0;
2648 
2649 	return ng->faults[task_faults_idx(NUMA_MEM, nid, 0)] +
2650 		ng->faults[task_faults_idx(NUMA_MEM, nid, 1)];
2651 }
2652 
2653 static inline unsigned long group_faults_cpu(struct numa_group *group, int nid)
2654 {
2655 	return group->faults[task_faults_idx(NUMA_CPU, nid, 0)] +
2656 		group->faults[task_faults_idx(NUMA_CPU, nid, 1)];
2657 }
2658 
2659 static inline unsigned long group_faults_priv(struct numa_group *ng)
2660 {
2661 	unsigned long faults = 0;
2662 	int node;
2663 
2664 	for_each_online_node(node) {
2665 		faults += ng->faults[task_faults_idx(NUMA_MEM, node, 1)];
2666 	}
2667 
2668 	return faults;
2669 }
2670 
2671 static inline unsigned long group_faults_shared(struct numa_group *ng)
2672 {
2673 	unsigned long faults = 0;
2674 	int node;
2675 
2676 	for_each_online_node(node) {
2677 		faults += ng->faults[task_faults_idx(NUMA_MEM, node, 0)];
2678 	}
2679 
2680 	return faults;
2681 }
2682 
2683 /*
2684  * A node triggering more than 1/3 as many NUMA faults as the maximum is
2685  * considered part of a numa group's pseudo-interleaving set. Migrations
2686  * between these nodes are slowed down, to allow things to settle down.
2687  */
2688 #define ACTIVE_NODE_FRACTION 3
2689 
2690 static bool numa_is_active_node(int nid, struct numa_group *ng)
2691 {
2692 	return group_faults_cpu(ng, nid) * ACTIVE_NODE_FRACTION > ng->max_faults_cpu;
2693 }
2694 
2695 /* Handle placement on systems where not all nodes are directly connected. */
2696 static unsigned long score_nearby_nodes(struct task_struct *p, int nid,
2697 					int lim_dist, bool task)
2698 {
2699 	unsigned long score = 0;
2700 	int node, max_dist;
2701 
2702 	/*
2703 	 * All nodes are directly connected, and the same distance
2704 	 * from each other. No need for fancy placement algorithms.
2705 	 */
2706 	if (sched_numa_topology_type == NUMA_DIRECT)
2707 		return 0;
2708 
2709 	/* sched_max_numa_distance may be changed in parallel. */
2710 	max_dist = READ_ONCE(sched_max_numa_distance);
2711 	/*
2712 	 * This code is called for each node, introducing N^2 complexity,
2713 	 * which should be OK given the number of nodes rarely exceeds 8.
2714 	 */
2715 	for_each_online_node(node) {
2716 		unsigned long faults;
2717 		int dist = node_distance(nid, node);
2718 
2719 		/*
2720 		 * The furthest away nodes in the system are not interesting
2721 		 * for placement; nid was already counted.
2722 		 */
2723 		if (dist >= max_dist || node == nid)
2724 			continue;
2725 
2726 		/*
2727 		 * On systems with a backplane NUMA topology, compare groups
2728 		 * of nodes, and move tasks towards the group with the most
2729 		 * memory accesses. When comparing two nodes at distance
2730 		 * "hoplimit", only nodes closer by than "hoplimit" are part
2731 		 * of each group. Skip other nodes.
2732 		 */
2733 		if (sched_numa_topology_type == NUMA_BACKPLANE && dist >= lim_dist)
2734 			continue;
2735 
2736 		/* Add up the faults from nearby nodes. */
2737 		if (task)
2738 			faults = task_faults(p, node);
2739 		else
2740 			faults = group_faults(p, node);
2741 
2742 		/*
2743 		 * On systems with a glueless mesh NUMA topology, there are
2744 		 * no fixed "groups of nodes". Instead, nodes that are not
2745 		 * directly connected bounce traffic through intermediate
2746 		 * nodes; a numa_group can occupy any set of nodes.
2747 		 * The further away a node is, the less the faults count.
2748 		 * This seems to result in good task placement.
2749 		 */
2750 		if (sched_numa_topology_type == NUMA_GLUELESS_MESH) {
2751 			faults *= (max_dist - dist);
2752 			faults /= (max_dist - LOCAL_DISTANCE);
2753 		}
2754 
2755 		score += faults;
2756 	}
2757 
2758 	return score;
2759 }
2760 
2761 /*
2762  * These return the fraction of accesses done by a particular task, or
2763  * task group, on a particular numa node.  The group weight is given a
2764  * larger multiplier, in order to group tasks together that are almost
2765  * evenly spread out between numa nodes.
2766  */
2767 static inline unsigned long task_weight(struct task_struct *p, int nid,
2768 					int dist)
2769 {
2770 	unsigned long faults, total_faults;
2771 
2772 	if (!p->numa_faults)
2773 		return 0;
2774 
2775 	total_faults = p->total_numa_faults;
2776 
2777 	if (!total_faults)
2778 		return 0;
2779 
2780 	faults = task_faults(p, nid);
2781 	faults += score_nearby_nodes(p, nid, dist, true);
2782 
2783 	return 1000 * faults / total_faults;
2784 }
2785 
2786 static inline unsigned long group_weight(struct task_struct *p, int nid,
2787 					 int dist)
2788 {
2789 	struct numa_group *ng = deref_task_numa_group(p);
2790 	unsigned long faults, total_faults;
2791 
2792 	if (!ng)
2793 		return 0;
2794 
2795 	total_faults = ng->total_faults;
2796 
2797 	if (!total_faults)
2798 		return 0;
2799 
2800 	faults = group_faults(p, nid);
2801 	faults += score_nearby_nodes(p, nid, dist, false);
2802 
2803 	return 1000 * faults / total_faults;
2804 }
2805 
2806 /*
2807  * If memory tiering mode is enabled, cpupid of slow memory page is
2808  * used to record scan time instead of CPU and PID.  When tiering mode
2809  * is disabled at run time, the scan time (in cpupid) will be
2810  * interpreted as CPU and PID.  So CPU needs to be checked to avoid to
2811  * access out of array bound.
2812  */
2813 static inline bool cpupid_valid(int cpupid)
2814 {
2815 	return cpupid_to_cpu(cpupid) < nr_cpu_ids;
2816 }
2817 
2818 /*
2819  * For memory tiering mode, if there are enough free pages (more than
2820  * enough watermark defined here) in fast memory node, to take full
2821  * advantage of fast memory capacity, all recently accessed slow
2822  * memory pages will be migrated to fast memory node without
2823  * considering hot threshold.
2824  */
2825 static bool pgdat_free_space_enough(struct pglist_data *pgdat)
2826 {
2827 	int z;
2828 	unsigned long enough_wmark;
2829 
2830 	enough_wmark = max(1UL * 1024 * 1024 * 1024 >> PAGE_SHIFT,
2831 			   pgdat->node_present_pages >> 4);
2832 	for (z = pgdat->nr_zones - 1; z >= 0; z--) {
2833 		struct zone *zone = pgdat->node_zones + z;
2834 
2835 		if (!populated_zone(zone))
2836 			continue;
2837 
2838 		if (zone_watermark_ok(zone, 0,
2839 				      promo_wmark_pages(zone) + enough_wmark,
2840 				      ZONE_MOVABLE, 0))
2841 			return true;
2842 	}
2843 	return false;
2844 }
2845 
2846 /*
2847  * For memory tiering mode, when page tables are scanned, the scan
2848  * time will be recorded in struct page in addition to make page
2849  * PROT_NONE for slow memory page.  So when the page is accessed, in
2850  * hint page fault handler, the hint page fault latency is calculated
2851  * via,
2852  *
2853  *	hint page fault latency = hint page fault time - scan time
2854  *
2855  * The smaller the hint page fault latency, the higher the possibility
2856  * for the page to be hot.
2857  */
2858 static int numa_hint_fault_latency(struct folio *folio)
2859 {
2860 	int last_time, time;
2861 
2862 	time = jiffies_to_msecs(jiffies);
2863 	last_time = folio_xchg_access_time(folio, time);
2864 
2865 	return (time - last_time) & PAGE_ACCESS_TIME_MASK;
2866 }
2867 
2868 /*
2869  * For memory tiering mode, too high promotion/demotion throughput may
2870  * hurt application latency.  So we provide a mechanism to rate limit
2871  * the number of pages that are tried to be promoted.
2872  */
2873 static bool numa_promotion_rate_limit(struct pglist_data *pgdat,
2874 				      unsigned long rate_limit, int nr)
2875 {
2876 	unsigned long nr_cand;
2877 	unsigned int now, start;
2878 
2879 	now = jiffies_to_msecs(jiffies);
2880 	mod_node_page_state(pgdat, PGPROMOTE_CANDIDATE, nr);
2881 	nr_cand = node_page_state(pgdat, PGPROMOTE_CANDIDATE);
2882 	start = pgdat->nbp_rl_start;
2883 	if (now - start > MSEC_PER_SEC &&
2884 	    cmpxchg(&pgdat->nbp_rl_start, start, now) == start)
2885 		pgdat->nbp_rl_nr_cand = nr_cand;
2886 	if (nr_cand - pgdat->nbp_rl_nr_cand >= rate_limit)
2887 		return true;
2888 	return false;
2889 }
2890 
2891 #define NUMA_MIGRATION_ADJUST_STEPS	16
2892 
2893 static void numa_promotion_adjust_threshold(struct pglist_data *pgdat,
2894 					    unsigned long rate_limit,
2895 					    unsigned int ref_th)
2896 {
2897 	unsigned int now, start, th_period, unit_th, th;
2898 	unsigned long nr_cand, ref_cand, diff_cand;
2899 
2900 	now = jiffies_to_msecs(jiffies);
2901 	th_period = sysctl_numa_balancing_scan_period_max;
2902 	start = pgdat->nbp_th_start;
2903 	if (now - start > th_period &&
2904 	    cmpxchg(&pgdat->nbp_th_start, start, now) == start) {
2905 		ref_cand = rate_limit *
2906 			sysctl_numa_balancing_scan_period_max / MSEC_PER_SEC;
2907 		nr_cand = node_page_state(pgdat, PGPROMOTE_CANDIDATE);
2908 		diff_cand = nr_cand - pgdat->nbp_th_nr_cand;
2909 		unit_th = ref_th * 2 / NUMA_MIGRATION_ADJUST_STEPS;
2910 		th = pgdat->nbp_threshold ? : ref_th;
2911 		if (diff_cand > ref_cand * 11 / 10)
2912 			th = max(th - unit_th, unit_th);
2913 		else if (diff_cand < ref_cand * 9 / 10)
2914 			th = min(th + unit_th, ref_th * 2);
2915 		pgdat->nbp_th_nr_cand = nr_cand;
2916 		pgdat->nbp_threshold = th;
2917 	}
2918 }
2919 
2920 bool should_numa_migrate_memory(struct task_struct *p, struct folio *folio,
2921 				int src_nid, int dst_cpu)
2922 {
2923 	struct numa_group *ng = deref_curr_numa_group(p);
2924 	int dst_nid = cpu_to_node(dst_cpu);
2925 	int last_cpupid, this_cpupid;
2926 
2927 	/*
2928 	 * Cannot migrate to memoryless nodes.
2929 	 */
2930 	if (!node_state(dst_nid, N_MEMORY))
2931 		return false;
2932 
2933 	/*
2934 	 * The pages in slow memory node should be migrated according
2935 	 * to hot/cold instead of private/shared.
2936 	 */
2937 	if (folio_use_access_time(folio)) {
2938 		struct pglist_data *pgdat;
2939 		unsigned long rate_limit;
2940 		unsigned int latency, th, def_th;
2941 		long nr = folio_nr_pages(folio);
2942 
2943 		pgdat = NODE_DATA(dst_nid);
2944 		if (pgdat_free_space_enough(pgdat)) {
2945 			/* workload changed, reset hot threshold */
2946 			pgdat->nbp_threshold = 0;
2947 			mod_node_page_state(pgdat, PGPROMOTE_CANDIDATE_NRL, nr);
2948 			return true;
2949 		}
2950 
2951 		def_th = sysctl_numa_balancing_hot_threshold;
2952 		rate_limit = MB_TO_PAGES(sysctl_numa_balancing_promote_rate_limit);
2953 		numa_promotion_adjust_threshold(pgdat, rate_limit, def_th);
2954 
2955 		th = pgdat->nbp_threshold ? : def_th;
2956 		latency = numa_hint_fault_latency(folio);
2957 		if (latency >= th)
2958 			return false;
2959 
2960 		return !numa_promotion_rate_limit(pgdat, rate_limit, nr);
2961 	}
2962 
2963 	this_cpupid = cpu_pid_to_cpupid(dst_cpu, current->pid);
2964 	last_cpupid = folio_xchg_last_cpupid(folio, this_cpupid);
2965 
2966 	if (!(sysctl_numa_balancing_mode & NUMA_BALANCING_MEMORY_TIERING) &&
2967 	    !node_is_toptier(src_nid) && !cpupid_valid(last_cpupid))
2968 		return false;
2969 
2970 	/*
2971 	 * Allow first faults or private faults to migrate immediately early in
2972 	 * the lifetime of a task. The magic number 4 is based on waiting for
2973 	 * two full passes of the "multi-stage node selection" test that is
2974 	 * executed below.
2975 	 */
2976 	if ((p->numa_preferred_nid == NUMA_NO_NODE || p->numa_scan_seq <= 4) &&
2977 	    (cpupid_pid_unset(last_cpupid) || cpupid_match_pid(p, last_cpupid)))
2978 		return true;
2979 
2980 	/*
2981 	 * Multi-stage node selection is used in conjunction with a periodic
2982 	 * migration fault to build a temporal task<->page relation. By using
2983 	 * a two-stage filter we remove short/unlikely relations.
2984 	 *
2985 	 * Using P(p) ~ n_p / n_t as per frequentist probability, we can equate
2986 	 * a task's usage of a particular page (n_p) per total usage of this
2987 	 * page (n_t) (in a given time-span) to a probability.
2988 	 *
2989 	 * Our periodic faults will sample this probability and getting the
2990 	 * same result twice in a row, given these samples are fully
2991 	 * independent, is then given by P(n)^2, provided our sample period
2992 	 * is sufficiently short compared to the usage pattern.
2993 	 *
2994 	 * This quadric squishes small probabilities, making it less likely we
2995 	 * act on an unlikely task<->page relation.
2996 	 */
2997 	if (!cpupid_pid_unset(last_cpupid) &&
2998 				cpupid_to_nid(last_cpupid) != dst_nid)
2999 		return false;
3000 
3001 	/* Always allow migrate on private faults */
3002 	if (cpupid_match_pid(p, last_cpupid))
3003 		return true;
3004 
3005 	/* A shared fault, but p->numa_group has not been set up yet. */
3006 	if (!ng)
3007 		return true;
3008 
3009 	/*
3010 	 * Destination node is much more heavily used than the source
3011 	 * node? Allow migration.
3012 	 */
3013 	if (group_faults_cpu(ng, dst_nid) > group_faults_cpu(ng, src_nid) *
3014 					ACTIVE_NODE_FRACTION)
3015 		return true;
3016 
3017 	/*
3018 	 * Distribute memory according to CPU & memory use on each node,
3019 	 * with 3/4 hysteresis to avoid unnecessary memory migrations:
3020 	 *
3021 	 * faults_cpu(dst)   3   faults_cpu(src)
3022 	 * --------------- * - > ---------------
3023 	 * faults_mem(dst)   4   faults_mem(src)
3024 	 */
3025 	return group_faults_cpu(ng, dst_nid) * group_faults(p, src_nid) * 3 >
3026 	       group_faults_cpu(ng, src_nid) * group_faults(p, dst_nid) * 4;
3027 }
3028 
3029 /*
3030  * 'numa_type' describes the node at the moment of load balancing.
3031  */
3032 enum numa_type {
3033 	/* The node has spare capacity that can be used to run more tasks.  */
3034 	node_has_spare = 0,
3035 	/*
3036 	 * The node is fully used and the tasks don't compete for more CPU
3037 	 * cycles. Nevertheless, some tasks might wait before running.
3038 	 */
3039 	node_fully_busy,
3040 	/*
3041 	 * The node is overloaded and can't provide expected CPU cycles to all
3042 	 * tasks.
3043 	 */
3044 	node_overloaded
3045 };
3046 
3047 /* Cached statistics for all CPUs within a node */
3048 struct numa_stats {
3049 	unsigned long load;
3050 	unsigned long runnable;
3051 	unsigned long util;
3052 	/* Total compute capacity of CPUs on a node */
3053 	unsigned long compute_capacity;
3054 	unsigned int nr_running;
3055 	unsigned int weight;
3056 	enum numa_type node_type;
3057 	int idle_cpu;
3058 };
3059 
3060 struct task_numa_env {
3061 	struct task_struct *p;
3062 
3063 	int src_cpu, src_nid;
3064 	int dst_cpu, dst_nid;
3065 	int imb_numa_nr;
3066 
3067 	struct numa_stats src_stats, dst_stats;
3068 
3069 	int imbalance_pct;
3070 	int dist;
3071 
3072 	struct task_struct *best_task;
3073 	long best_imp;
3074 	int best_cpu;
3075 };
3076 
3077 static unsigned long cpu_load(struct rq *rq);
3078 static unsigned long cpu_runnable(struct rq *rq);
3079 
3080 static inline enum
3081 numa_type numa_classify(unsigned int imbalance_pct,
3082 			 struct numa_stats *ns)
3083 {
3084 	if ((ns->nr_running > ns->weight) &&
3085 	    (((ns->compute_capacity * 100) < (ns->util * imbalance_pct)) ||
3086 	     ((ns->compute_capacity * imbalance_pct) < (ns->runnable * 100))))
3087 		return node_overloaded;
3088 
3089 	if ((ns->nr_running < ns->weight) ||
3090 	    (((ns->compute_capacity * 100) > (ns->util * imbalance_pct)) &&
3091 	     ((ns->compute_capacity * imbalance_pct) > (ns->runnable * 100))))
3092 		return node_has_spare;
3093 
3094 	return node_fully_busy;
3095 }
3096 
3097 /* Forward declarations of select_idle_sibling helpers */
3098 static inline bool test_idle_cores(int cpu);
3099 static inline int numa_idle_core(int idle_core, int cpu)
3100 {
3101 	if (!sched_smt_active() ||
3102 	    idle_core >= 0 || !test_idle_cores(cpu))
3103 		return idle_core;
3104 
3105 	/*
3106 	 * Prefer cores instead of packing HT siblings
3107 	 * and triggering future load balancing.
3108 	 */
3109 	if (is_core_idle(cpu))
3110 		idle_core = cpu;
3111 
3112 	return idle_core;
3113 }
3114 
3115 /*
3116  * Gather all necessary information to make NUMA balancing placement
3117  * decisions that are compatible with standard load balancer. This
3118  * borrows code and logic from update_sg_lb_stats but sharing a
3119  * common implementation is impractical.
3120  */
3121 static void update_numa_stats(struct task_numa_env *env,
3122 			      struct numa_stats *ns, int nid,
3123 			      bool find_idle)
3124 {
3125 	int cpu, idle_core = -1;
3126 
3127 	memset(ns, 0, sizeof(*ns));
3128 	ns->idle_cpu = -1;
3129 
3130 	rcu_read_lock();
3131 	for_each_cpu(cpu, cpumask_of_node(nid)) {
3132 		struct rq *rq = cpu_rq(cpu);
3133 
3134 		ns->load += cpu_load(rq);
3135 		ns->runnable += cpu_runnable(rq);
3136 		ns->util += cpu_util_cfs(cpu);
3137 		ns->nr_running += rq->cfs.h_nr_runnable;
3138 		ns->compute_capacity += capacity_of(cpu);
3139 
3140 		if (find_idle && idle_core < 0 && !rq->nr_running && idle_cpu(cpu)) {
3141 			if (READ_ONCE(rq->numa_migrate_on) ||
3142 			    !cpumask_test_cpu(cpu, env->p->cpus_ptr))
3143 				continue;
3144 
3145 			if (ns->idle_cpu == -1)
3146 				ns->idle_cpu = cpu;
3147 
3148 			idle_core = numa_idle_core(idle_core, cpu);
3149 		}
3150 	}
3151 	rcu_read_unlock();
3152 
3153 	ns->weight = cpumask_weight(cpumask_of_node(nid));
3154 
3155 	ns->node_type = numa_classify(env->imbalance_pct, ns);
3156 
3157 	if (idle_core >= 0)
3158 		ns->idle_cpu = idle_core;
3159 }
3160 
3161 static void task_numa_assign(struct task_numa_env *env,
3162 			     struct task_struct *p, long imp)
3163 {
3164 	struct rq *rq = cpu_rq(env->dst_cpu);
3165 
3166 	/* Check if run-queue part of active NUMA balance. */
3167 	if (env->best_cpu != env->dst_cpu && xchg(&rq->numa_migrate_on, 1)) {
3168 		int cpu;
3169 		int start = env->dst_cpu;
3170 
3171 		/* Find alternative idle CPU. */
3172 		for_each_cpu_wrap(cpu, cpumask_of_node(env->dst_nid), start + 1) {
3173 			if (cpu == env->best_cpu || !idle_cpu(cpu) ||
3174 			    !cpumask_test_cpu(cpu, env->p->cpus_ptr)) {
3175 				continue;
3176 			}
3177 
3178 			env->dst_cpu = cpu;
3179 			rq = cpu_rq(env->dst_cpu);
3180 			if (!xchg(&rq->numa_migrate_on, 1))
3181 				goto assign;
3182 		}
3183 
3184 		/* Failed to find an alternative idle CPU */
3185 		return;
3186 	}
3187 
3188 assign:
3189 	/*
3190 	 * Clear previous best_cpu/rq numa-migrate flag, since task now
3191 	 * found a better CPU to move/swap.
3192 	 */
3193 	if (env->best_cpu != -1 && env->best_cpu != env->dst_cpu) {
3194 		rq = cpu_rq(env->best_cpu);
3195 		WRITE_ONCE(rq->numa_migrate_on, 0);
3196 	}
3197 
3198 	if (env->best_task)
3199 		put_task_struct(env->best_task);
3200 	if (p)
3201 		get_task_struct(p);
3202 
3203 	env->best_task = p;
3204 	env->best_imp = imp;
3205 	env->best_cpu = env->dst_cpu;
3206 }
3207 
3208 static bool load_too_imbalanced(long src_load, long dst_load,
3209 				struct task_numa_env *env)
3210 {
3211 	long imb, old_imb;
3212 	long orig_src_load, orig_dst_load;
3213 	long src_capacity, dst_capacity;
3214 
3215 	/*
3216 	 * The load is corrected for the CPU capacity available on each node.
3217 	 *
3218 	 * src_load        dst_load
3219 	 * ------------ vs ---------
3220 	 * src_capacity    dst_capacity
3221 	 */
3222 	src_capacity = env->src_stats.compute_capacity;
3223 	dst_capacity = env->dst_stats.compute_capacity;
3224 
3225 	imb = abs(dst_load * src_capacity - src_load * dst_capacity);
3226 
3227 	orig_src_load = env->src_stats.load;
3228 	orig_dst_load = env->dst_stats.load;
3229 
3230 	old_imb = abs(orig_dst_load * src_capacity - orig_src_load * dst_capacity);
3231 
3232 	/* Would this change make things worse? */
3233 	return (imb > old_imb);
3234 }
3235 
3236 /*
3237  * Maximum NUMA importance can be 1998 (2*999);
3238  * SMALLIMP @ 30 would be close to 1998/64.
3239  * Used to deter task migration.
3240  */
3241 #define SMALLIMP	30
3242 
3243 /*
3244  * This checks if the overall compute and NUMA accesses of the system would
3245  * be improved if the source tasks was migrated to the target dst_cpu taking
3246  * into account that it might be best if task running on the dst_cpu should
3247  * be exchanged with the source task
3248  */
3249 static bool task_numa_compare(struct task_numa_env *env,
3250 			      long taskimp, long groupimp, bool maymove)
3251 {
3252 	struct numa_group *cur_ng, *p_ng = deref_curr_numa_group(env->p);
3253 	struct rq *dst_rq = cpu_rq(env->dst_cpu);
3254 	long imp = p_ng ? groupimp : taskimp;
3255 	struct task_struct *cur;
3256 	long src_load, dst_load;
3257 	int dist = env->dist;
3258 	long moveimp = imp;
3259 	long load;
3260 	bool stopsearch = false;
3261 
3262 	if (READ_ONCE(dst_rq->numa_migrate_on))
3263 		return false;
3264 
3265 	rcu_read_lock();
3266 	cur = rcu_dereference_all(dst_rq->curr);
3267 	if (cur && ((cur->flags & (PF_EXITING | PF_KTHREAD)) ||
3268 		    !cur->mm))
3269 		cur = NULL;
3270 
3271 	/*
3272 	 * Because we have preemption enabled we can get migrated around and
3273 	 * end try selecting ourselves (current == env->p) as a swap candidate.
3274 	 */
3275 	if (cur == env->p) {
3276 		stopsearch = true;
3277 		goto unlock;
3278 	}
3279 
3280 	if (!cur) {
3281 		if (maymove && moveimp >= env->best_imp)
3282 			goto assign;
3283 		else
3284 			goto unlock;
3285 	}
3286 
3287 	/* Skip this swap candidate if cannot move to the source cpu. */
3288 	if (!cpumask_test_cpu(env->src_cpu, cur->cpus_ptr))
3289 		goto unlock;
3290 
3291 	/*
3292 	 * Skip this swap candidate if it is not moving to its preferred
3293 	 * node and the best task is.
3294 	 */
3295 	if (env->best_task &&
3296 	    env->best_task->numa_preferred_nid == env->src_nid &&
3297 	    cur->numa_preferred_nid != env->src_nid) {
3298 		goto unlock;
3299 	}
3300 
3301 	/*
3302 	 * "imp" is the fault differential for the source task between the
3303 	 * source and destination node. Calculate the total differential for
3304 	 * the source task and potential destination task. The more negative
3305 	 * the value is, the more remote accesses that would be expected to
3306 	 * be incurred if the tasks were swapped.
3307 	 *
3308 	 * If dst and source tasks are in the same NUMA group, or not
3309 	 * in any group then look only at task weights.
3310 	 */
3311 	cur_ng = rcu_dereference_all(cur->numa_group);
3312 	if (cur_ng == p_ng) {
3313 		/*
3314 		 * Do not swap within a group or between tasks that have
3315 		 * no group if there is spare capacity. Swapping does
3316 		 * not address the load imbalance and helps one task at
3317 		 * the cost of punishing another.
3318 		 */
3319 		if (env->dst_stats.node_type == node_has_spare)
3320 			goto unlock;
3321 
3322 		imp = taskimp + task_weight(cur, env->src_nid, dist) -
3323 		      task_weight(cur, env->dst_nid, dist);
3324 		/*
3325 		 * Add some hysteresis to prevent swapping the
3326 		 * tasks within a group over tiny differences.
3327 		 */
3328 		if (cur_ng)
3329 			imp -= imp / 16;
3330 	} else {
3331 		/*
3332 		 * Compare the group weights. If a task is all by itself
3333 		 * (not part of a group), use the task weight instead.
3334 		 */
3335 		if (cur_ng && p_ng)
3336 			imp += group_weight(cur, env->src_nid, dist) -
3337 			       group_weight(cur, env->dst_nid, dist);
3338 		else
3339 			imp += task_weight(cur, env->src_nid, dist) -
3340 			       task_weight(cur, env->dst_nid, dist);
3341 	}
3342 
3343 	/* Discourage picking a task already on its preferred node */
3344 	if (cur->numa_preferred_nid == env->dst_nid)
3345 		imp -= imp / 16;
3346 
3347 	/*
3348 	 * Encourage picking a task that moves to its preferred node.
3349 	 * This potentially makes imp larger than it's maximum of
3350 	 * 1998 (see SMALLIMP and task_weight for why) but in this
3351 	 * case, it does not matter.
3352 	 */
3353 	if (cur->numa_preferred_nid == env->src_nid)
3354 		imp += imp / 8;
3355 
3356 	if (maymove && moveimp > imp && moveimp > env->best_imp) {
3357 		imp = moveimp;
3358 		cur = NULL;
3359 		goto assign;
3360 	}
3361 
3362 	/*
3363 	 * Prefer swapping with a task moving to its preferred node over a
3364 	 * task that is not.
3365 	 */
3366 	if (env->best_task && cur->numa_preferred_nid == env->src_nid &&
3367 	    env->best_task->numa_preferred_nid != env->src_nid) {
3368 		goto assign;
3369 	}
3370 
3371 	/*
3372 	 * If the NUMA importance is less than SMALLIMP,
3373 	 * task migration might only result in ping pong
3374 	 * of tasks and also hurt performance due to cache
3375 	 * misses.
3376 	 */
3377 	if (imp < SMALLIMP || imp <= env->best_imp + SMALLIMP / 2)
3378 		goto unlock;
3379 
3380 	/*
3381 	 * In the overloaded case, try and keep the load balanced.
3382 	 */
3383 	load = task_h_load(env->p) - task_h_load(cur);
3384 	if (!load)
3385 		goto assign;
3386 
3387 	dst_load = env->dst_stats.load + load;
3388 	src_load = env->src_stats.load - load;
3389 
3390 	if (load_too_imbalanced(src_load, dst_load, env))
3391 		goto unlock;
3392 
3393 assign:
3394 	/* Evaluate an idle CPU for a task numa move. */
3395 	if (!cur) {
3396 		int cpu = env->dst_stats.idle_cpu;
3397 
3398 		/* Nothing cached so current CPU went idle since the search. */
3399 		if (cpu < 0)
3400 			cpu = env->dst_cpu;
3401 
3402 		/*
3403 		 * If the CPU is no longer truly idle and the previous best CPU
3404 		 * is, keep using it.
3405 		 */
3406 		if (!idle_cpu(cpu) && env->best_cpu >= 0 &&
3407 		    idle_cpu(env->best_cpu)) {
3408 			cpu = env->best_cpu;
3409 		}
3410 
3411 		env->dst_cpu = cpu;
3412 	}
3413 
3414 	task_numa_assign(env, cur, imp);
3415 
3416 	/*
3417 	 * If a move to idle is allowed because there is capacity or load
3418 	 * balance improves then stop the search. While a better swap
3419 	 * candidate may exist, a search is not free.
3420 	 */
3421 	if (maymove && !cur && env->best_cpu >= 0 && idle_cpu(env->best_cpu))
3422 		stopsearch = true;
3423 
3424 	/*
3425 	 * If a swap candidate must be identified and the current best task
3426 	 * moves its preferred node then stop the search.
3427 	 */
3428 	if (!maymove && env->best_task &&
3429 	    env->best_task->numa_preferred_nid == env->src_nid) {
3430 		stopsearch = true;
3431 	}
3432 unlock:
3433 	rcu_read_unlock();
3434 
3435 	return stopsearch;
3436 }
3437 
3438 static void task_numa_find_cpu(struct task_numa_env *env,
3439 				long taskimp, long groupimp)
3440 {
3441 	bool maymove = false;
3442 	int cpu;
3443 
3444 	/*
3445 	 * If dst node has spare capacity, then check if there is an
3446 	 * imbalance that would be overruled by the load balancer.
3447 	 */
3448 	if (env->dst_stats.node_type == node_has_spare) {
3449 		unsigned int imbalance;
3450 		int src_running, dst_running;
3451 
3452 		/*
3453 		 * Would movement cause an imbalance? Note that if src has
3454 		 * more running tasks that the imbalance is ignored as the
3455 		 * move improves the imbalance from the perspective of the
3456 		 * CPU load balancer.
3457 		 * */
3458 		src_running = env->src_stats.nr_running - 1;
3459 		dst_running = env->dst_stats.nr_running + 1;
3460 		imbalance = max(0, dst_running - src_running);
3461 		imbalance = adjust_numa_imbalance(imbalance, dst_running,
3462 						  env->imb_numa_nr);
3463 
3464 		/* Use idle CPU if there is no imbalance */
3465 		if (!imbalance) {
3466 			maymove = true;
3467 			if (env->dst_stats.idle_cpu >= 0) {
3468 				env->dst_cpu = env->dst_stats.idle_cpu;
3469 				task_numa_assign(env, NULL, 0);
3470 				return;
3471 			}
3472 		}
3473 	} else {
3474 		long src_load, dst_load, load;
3475 		/*
3476 		 * If the improvement from just moving env->p direction is better
3477 		 * than swapping tasks around, check if a move is possible.
3478 		 */
3479 		load = task_h_load(env->p);
3480 		dst_load = env->dst_stats.load + load;
3481 		src_load = env->src_stats.load - load;
3482 		maymove = !load_too_imbalanced(src_load, dst_load, env);
3483 	}
3484 
3485 	/* Skip CPUs if the source task cannot migrate */
3486 	for_each_cpu_and(cpu, cpumask_of_node(env->dst_nid), env->p->cpus_ptr) {
3487 		env->dst_cpu = cpu;
3488 		if (task_numa_compare(env, taskimp, groupimp, maymove))
3489 			break;
3490 	}
3491 }
3492 
3493 static int task_numa_migrate(struct task_struct *p)
3494 {
3495 	struct task_numa_env env = {
3496 		.p = p,
3497 
3498 		.src_cpu = task_cpu(p),
3499 		.src_nid = task_node(p),
3500 
3501 		.imbalance_pct = 112,
3502 
3503 		.best_task = NULL,
3504 		.best_imp = 0,
3505 		.best_cpu = -1,
3506 	};
3507 	unsigned long taskweight, groupweight;
3508 	struct sched_domain *sd;
3509 	long taskimp, groupimp;
3510 	struct numa_group *ng;
3511 	struct rq *best_rq;
3512 	int nid, ret, dist;
3513 
3514 	/*
3515 	 * Pick the lowest SD_NUMA domain, as that would have the smallest
3516 	 * imbalance and would be the first to start moving tasks about.
3517 	 *
3518 	 * And we want to avoid any moving of tasks about, as that would create
3519 	 * random movement of tasks -- counter the numa conditions we're trying
3520 	 * to satisfy here.
3521 	 */
3522 	rcu_read_lock();
3523 	sd = rcu_dereference_all(per_cpu(sd_numa, env.src_cpu));
3524 	if (sd) {
3525 		env.imbalance_pct = 100 + (sd->imbalance_pct - 100) / 2;
3526 		env.imb_numa_nr = sd->imb_numa_nr;
3527 	}
3528 	rcu_read_unlock();
3529 
3530 	/*
3531 	 * Cpusets can break the scheduler domain tree into smaller
3532 	 * balance domains, some of which do not cross NUMA boundaries.
3533 	 * Tasks that are "trapped" in such domains cannot be migrated
3534 	 * elsewhere, so there is no point in (re)trying.
3535 	 */
3536 	if (unlikely(!sd)) {
3537 		sched_setnuma(p, task_node(p));
3538 		return -EINVAL;
3539 	}
3540 
3541 	env.dst_nid = p->numa_preferred_nid;
3542 	dist = env.dist = node_distance(env.src_nid, env.dst_nid);
3543 	taskweight = task_weight(p, env.src_nid, dist);
3544 	groupweight = group_weight(p, env.src_nid, dist);
3545 	update_numa_stats(&env, &env.src_stats, env.src_nid, false);
3546 	taskimp = task_weight(p, env.dst_nid, dist) - taskweight;
3547 	groupimp = group_weight(p, env.dst_nid, dist) - groupweight;
3548 	update_numa_stats(&env, &env.dst_stats, env.dst_nid, true);
3549 
3550 	/* Try to find a spot on the preferred nid. */
3551 	task_numa_find_cpu(&env, taskimp, groupimp);
3552 
3553 	/*
3554 	 * Look at other nodes in these cases:
3555 	 * - there is no space available on the preferred_nid
3556 	 * - the task is part of a numa_group that is interleaved across
3557 	 *   multiple NUMA nodes; in order to better consolidate the group,
3558 	 *   we need to check other locations.
3559 	 */
3560 	ng = deref_curr_numa_group(p);
3561 	if (env.best_cpu == -1 || (ng && ng->active_nodes > 1)) {
3562 		for_each_node_state(nid, N_CPU) {
3563 			if (nid == env.src_nid || nid == p->numa_preferred_nid)
3564 				continue;
3565 
3566 			dist = node_distance(env.src_nid, env.dst_nid);
3567 			if (sched_numa_topology_type == NUMA_BACKPLANE &&
3568 						dist != env.dist) {
3569 				taskweight = task_weight(p, env.src_nid, dist);
3570 				groupweight = group_weight(p, env.src_nid, dist);
3571 			}
3572 
3573 			/* Only consider nodes where both task and groups benefit */
3574 			taskimp = task_weight(p, nid, dist) - taskweight;
3575 			groupimp = group_weight(p, nid, dist) - groupweight;
3576 			if (taskimp < 0 && groupimp < 0)
3577 				continue;
3578 
3579 			env.dist = dist;
3580 			env.dst_nid = nid;
3581 			update_numa_stats(&env, &env.dst_stats, env.dst_nid, true);
3582 			task_numa_find_cpu(&env, taskimp, groupimp);
3583 		}
3584 	}
3585 
3586 	/*
3587 	 * If the task is part of a workload that spans multiple NUMA nodes,
3588 	 * and is migrating into one of the workload's active nodes, remember
3589 	 * this node as the task's preferred numa node, so the workload can
3590 	 * settle down.
3591 	 * A task that migrated to a second choice node will be better off
3592 	 * trying for a better one later. Do not set the preferred node here.
3593 	 */
3594 	if (ng) {
3595 		if (env.best_cpu == -1)
3596 			nid = env.src_nid;
3597 		else
3598 			nid = cpu_to_node(env.best_cpu);
3599 
3600 		if (nid != p->numa_preferred_nid)
3601 			sched_setnuma(p, nid);
3602 	}
3603 
3604 	/* No better CPU than the current one was found. */
3605 	if (env.best_cpu == -1) {
3606 		trace_sched_stick_numa(p, env.src_cpu, NULL, -1);
3607 		return -EAGAIN;
3608 	}
3609 
3610 	best_rq = cpu_rq(env.best_cpu);
3611 	if (env.best_task == NULL) {
3612 		ret = migrate_task_to(p, env.best_cpu);
3613 		WRITE_ONCE(best_rq->numa_migrate_on, 0);
3614 		if (ret != 0)
3615 			trace_sched_stick_numa(p, env.src_cpu, NULL, env.best_cpu);
3616 		return ret;
3617 	}
3618 
3619 	ret = migrate_swap(p, env.best_task, env.best_cpu, env.src_cpu);
3620 	WRITE_ONCE(best_rq->numa_migrate_on, 0);
3621 
3622 	if (ret != 0)
3623 		trace_sched_stick_numa(p, env.src_cpu, env.best_task, env.best_cpu);
3624 	put_task_struct(env.best_task);
3625 	return ret;
3626 }
3627 
3628 /* Attempt to migrate a task to a CPU on the preferred node. */
3629 static void numa_migrate_preferred(struct task_struct *p)
3630 {
3631 	unsigned long interval = HZ;
3632 
3633 	/* This task has no NUMA fault statistics yet */
3634 	if (unlikely(p->numa_preferred_nid == NUMA_NO_NODE || !p->numa_faults))
3635 		return;
3636 
3637 	/* Periodically retry migrating the task to the preferred node */
3638 	interval = min(interval, msecs_to_jiffies(p->numa_scan_period) / 16);
3639 	p->numa_migrate_retry = jiffies + interval;
3640 
3641 	/* Success if task is already running on preferred CPU */
3642 	if (task_node(p) == p->numa_preferred_nid)
3643 		return;
3644 
3645 	/* Otherwise, try migrate to a CPU on the preferred node */
3646 	task_numa_migrate(p);
3647 }
3648 
3649 /*
3650  * Find out how many nodes the workload is actively running on. Do this by
3651  * tracking the nodes from which NUMA hinting faults are triggered. This can
3652  * be different from the set of nodes where the workload's memory is currently
3653  * located.
3654  */
3655 static void numa_group_count_active_nodes(struct numa_group *numa_group)
3656 {
3657 	unsigned long faults, max_faults = 0;
3658 	int nid, active_nodes = 0;
3659 
3660 	for_each_node_state(nid, N_CPU) {
3661 		faults = group_faults_cpu(numa_group, nid);
3662 		if (faults > max_faults)
3663 			max_faults = faults;
3664 	}
3665 
3666 	for_each_node_state(nid, N_CPU) {
3667 		faults = group_faults_cpu(numa_group, nid);
3668 		if (faults * ACTIVE_NODE_FRACTION > max_faults)
3669 			active_nodes++;
3670 	}
3671 
3672 	numa_group->max_faults_cpu = max_faults;
3673 	numa_group->active_nodes = active_nodes;
3674 }
3675 
3676 /*
3677  * When adapting the scan rate, the period is divided into NUMA_PERIOD_SLOTS
3678  * increments. The more local the fault statistics are, the higher the scan
3679  * period will be for the next scan window. If local/(local+remote) ratio is
3680  * below NUMA_PERIOD_THRESHOLD (where range of ratio is 1..NUMA_PERIOD_SLOTS)
3681  * the scan period will decrease. Aim for 70% local accesses.
3682  */
3683 #define NUMA_PERIOD_SLOTS 10
3684 #define NUMA_PERIOD_THRESHOLD 7
3685 
3686 /*
3687  * Increase the scan period (slow down scanning) if the majority of
3688  * our memory is already on our local node, or if the majority of
3689  * the page accesses are shared with other processes.
3690  * Otherwise, decrease the scan period.
3691  */
3692 static void update_task_scan_period(struct task_struct *p,
3693 			unsigned long shared, unsigned long private)
3694 {
3695 	unsigned int period_slot;
3696 	int lr_ratio, ps_ratio;
3697 	int diff;
3698 
3699 	unsigned long remote = p->numa_faults_locality[0];
3700 	unsigned long local = p->numa_faults_locality[1];
3701 
3702 	/*
3703 	 * If there were no record hinting faults then either the task is
3704 	 * completely idle or all activity is in areas that are not of interest
3705 	 * to automatic numa balancing. Related to that, if there were failed
3706 	 * migration then it implies we are migrating too quickly or the local
3707 	 * node is overloaded. In either case, scan slower
3708 	 */
3709 	if (local + shared == 0 || p->numa_faults_locality[2]) {
3710 		p->numa_scan_period = min(p->numa_scan_period_max,
3711 			p->numa_scan_period << 1);
3712 
3713 		p->mm->numa_next_scan = jiffies +
3714 			msecs_to_jiffies(p->numa_scan_period);
3715 
3716 		return;
3717 	}
3718 
3719 	/*
3720 	 * Prepare to scale scan period relative to the current period.
3721 	 *	 == NUMA_PERIOD_THRESHOLD scan period stays the same
3722 	 *       <  NUMA_PERIOD_THRESHOLD scan period decreases (scan faster)
3723 	 *	 >= NUMA_PERIOD_THRESHOLD scan period increases (scan slower)
3724 	 */
3725 	period_slot = DIV_ROUND_UP(p->numa_scan_period, NUMA_PERIOD_SLOTS);
3726 	lr_ratio = (local * NUMA_PERIOD_SLOTS) / (local + remote);
3727 	ps_ratio = (private * NUMA_PERIOD_SLOTS) / (private + shared);
3728 
3729 	if (ps_ratio >= NUMA_PERIOD_THRESHOLD) {
3730 		/*
3731 		 * Most memory accesses are local. There is no need to
3732 		 * do fast NUMA scanning, since memory is already local.
3733 		 */
3734 		int slot = ps_ratio - NUMA_PERIOD_THRESHOLD;
3735 		if (!slot)
3736 			slot = 1;
3737 		diff = slot * period_slot;
3738 	} else if (lr_ratio >= NUMA_PERIOD_THRESHOLD) {
3739 		/*
3740 		 * Most memory accesses are shared with other tasks.
3741 		 * There is no point in continuing fast NUMA scanning,
3742 		 * since other tasks may just move the memory elsewhere.
3743 		 */
3744 		int slot = lr_ratio - NUMA_PERIOD_THRESHOLD;
3745 		if (!slot)
3746 			slot = 1;
3747 		diff = slot * period_slot;
3748 	} else {
3749 		/*
3750 		 * Private memory faults exceed (SLOTS-THRESHOLD)/SLOTS,
3751 		 * yet they are not on the local NUMA node. Speed up
3752 		 * NUMA scanning to get the memory moved over.
3753 		 */
3754 		int ratio = max(lr_ratio, ps_ratio);
3755 		diff = -(NUMA_PERIOD_THRESHOLD - ratio) * period_slot;
3756 	}
3757 
3758 	p->numa_scan_period = clamp(p->numa_scan_period + diff,
3759 			task_scan_min(p), task_scan_max(p));
3760 	memset(p->numa_faults_locality, 0, sizeof(p->numa_faults_locality));
3761 }
3762 
3763 /*
3764  * Get the fraction of time the task has been running since the last
3765  * NUMA placement cycle. The scheduler keeps similar statistics, but
3766  * decays those on a 32ms period, which is orders of magnitude off
3767  * from the dozens-of-seconds NUMA balancing period. Use the scheduler
3768  * stats only if the task is so new there are no NUMA statistics yet.
3769  */
3770 static u64 numa_get_avg_runtime(struct task_struct *p, u64 *period)
3771 {
3772 	u64 runtime, delta, now;
3773 	/* Use the start of this time slice to avoid calculations. */
3774 	now = p->se.exec_start;
3775 	runtime = p->se.sum_exec_runtime;
3776 
3777 	if (p->last_task_numa_placement) {
3778 		delta = runtime - p->last_sum_exec_runtime;
3779 		*period = now - p->last_task_numa_placement;
3780 
3781 		/* Avoid time going backwards, prevent potential divide error: */
3782 		if (unlikely((s64)*period < 0))
3783 			*period = 0;
3784 	} else {
3785 		delta = p->se.avg.load_sum;
3786 		*period = LOAD_AVG_MAX;
3787 	}
3788 
3789 	p->last_sum_exec_runtime = runtime;
3790 	p->last_task_numa_placement = now;
3791 
3792 	return delta;
3793 }
3794 
3795 /*
3796  * Determine the preferred nid for a task in a numa_group. This needs to
3797  * be done in a way that produces consistent results with group_weight,
3798  * otherwise workloads might not converge.
3799  */
3800 static int preferred_group_nid(struct task_struct *p, int nid)
3801 {
3802 	nodemask_t nodes;
3803 	int dist;
3804 
3805 	/* Direct connections between all NUMA nodes. */
3806 	if (sched_numa_topology_type == NUMA_DIRECT)
3807 		return nid;
3808 
3809 	/*
3810 	 * On a system with glueless mesh NUMA topology, group_weight
3811 	 * scores nodes according to the number of NUMA hinting faults on
3812 	 * both the node itself, and on nearby nodes.
3813 	 */
3814 	if (sched_numa_topology_type == NUMA_GLUELESS_MESH) {
3815 		unsigned long score, max_score = 0;
3816 		int node, max_node = nid;
3817 
3818 		dist = sched_max_numa_distance;
3819 
3820 		for_each_node_state(node, N_CPU) {
3821 			score = group_weight(p, node, dist);
3822 			if (score > max_score) {
3823 				max_score = score;
3824 				max_node = node;
3825 			}
3826 		}
3827 		return max_node;
3828 	}
3829 
3830 	/*
3831 	 * Finding the preferred nid in a system with NUMA backplane
3832 	 * interconnect topology is more involved. The goal is to locate
3833 	 * tasks from numa_groups near each other in the system, and
3834 	 * untangle workloads from different sides of the system. This requires
3835 	 * searching down the hierarchy of node groups, recursively searching
3836 	 * inside the highest scoring group of nodes. The nodemask tricks
3837 	 * keep the complexity of the search down.
3838 	 */
3839 	nodes = node_states[N_CPU];
3840 	for (dist = sched_max_numa_distance; dist > LOCAL_DISTANCE; dist--) {
3841 		unsigned long max_faults = 0;
3842 		nodemask_t max_group = NODE_MASK_NONE;
3843 		int a, b;
3844 
3845 		/* Are there nodes at this distance from each other? */
3846 		if (!find_numa_distance(dist))
3847 			continue;
3848 
3849 		for_each_node_mask(a, nodes) {
3850 			unsigned long faults = 0;
3851 			nodemask_t this_group;
3852 			nodes_clear(this_group);
3853 
3854 			/* Sum group's NUMA faults; includes a==b case. */
3855 			for_each_node_mask(b, nodes) {
3856 				if (node_distance(a, b) < dist) {
3857 					faults += group_faults(p, b);
3858 					node_set(b, this_group);
3859 					node_clear(b, nodes);
3860 				}
3861 			}
3862 
3863 			/* Remember the top group. */
3864 			if (faults > max_faults) {
3865 				max_faults = faults;
3866 				max_group = this_group;
3867 				/*
3868 				 * subtle: at the smallest distance there is
3869 				 * just one node left in each "group", the
3870 				 * winner is the preferred nid.
3871 				 */
3872 				nid = a;
3873 			}
3874 		}
3875 		/* Next round, evaluate the nodes within max_group. */
3876 		if (!max_faults)
3877 			break;
3878 		nodes = max_group;
3879 	}
3880 	return nid;
3881 }
3882 
3883 static void task_numa_placement(struct task_struct *p)
3884 	__context_unsafe(/* conditional locking */)
3885 {
3886 	struct sched_cache_group __maybe_unused *grp;
3887 	int seq, nid, max_nid = NUMA_NO_NODE;
3888 	unsigned long max_faults = 0;
3889 	unsigned long fault_types[2] = { 0, 0 };
3890 	unsigned long total_faults;
3891 	u64 runtime, period;
3892 	spinlock_t *group_lock = NULL;
3893 	long __maybe_unused new_fp;
3894 	struct numa_group *ng;
3895 
3896 	/*
3897 	 * The p->mm->numa_scan_seq field gets updated without
3898 	 * exclusive access. Use READ_ONCE() here to ensure
3899 	 * that the field is read in a single access:
3900 	 */
3901 	seq = READ_ONCE(p->mm->numa_scan_seq);
3902 	if (p->numa_scan_seq == seq)
3903 		return;
3904 	p->numa_scan_seq = seq;
3905 	p->numa_scan_period_max = task_scan_max(p);
3906 
3907 	total_faults = p->numa_faults_locality[0] +
3908 		       p->numa_faults_locality[1];
3909 	runtime = numa_get_avg_runtime(p, &period);
3910 
3911 	/* If the task is part of a group prevent parallel updates to group stats */
3912 	ng = deref_curr_numa_group(p);
3913 	if (ng) {
3914 		group_lock = &ng->lock;
3915 		spin_lock_irq(group_lock);
3916 	}
3917 
3918 	/* Find the node with the highest number of faults */
3919 	for_each_online_node(nid) {
3920 		/* Keep track of the offsets in numa_faults array */
3921 		int mem_idx, membuf_idx, cpu_idx, cpubuf_idx;
3922 		unsigned long faults = 0, group_faults = 0;
3923 		int priv;
3924 
3925 		for (priv = 0; priv < NR_NUMA_HINT_FAULT_TYPES; priv++) {
3926 			long diff, f_diff, f_weight;
3927 
3928 			mem_idx = task_faults_idx(NUMA_MEM, nid, priv);
3929 			membuf_idx = task_faults_idx(NUMA_MEMBUF, nid, priv);
3930 			cpu_idx = task_faults_idx(NUMA_CPU, nid, priv);
3931 			cpubuf_idx = task_faults_idx(NUMA_CPUBUF, nid, priv);
3932 
3933 			/* Decay existing window, copy faults since last scan */
3934 			diff = p->numa_faults[membuf_idx] - p->numa_faults[mem_idx] / 2;
3935 			fault_types[priv] += p->numa_faults[membuf_idx];
3936 			p->numa_faults[membuf_idx] = 0;
3937 
3938 			/*
3939 			 * Normalize the faults_from, so all tasks in a group
3940 			 * count according to CPU use, instead of by the raw
3941 			 * number of faults. Tasks with little runtime have
3942 			 * little over-all impact on throughput, and thus their
3943 			 * faults are less important.
3944 			 */
3945 			f_weight = div64_u64(runtime << 16, period + 1);
3946 			f_weight = (f_weight * p->numa_faults[cpubuf_idx]) /
3947 				   (total_faults + 1);
3948 			f_diff = f_weight - p->numa_faults[cpu_idx] / 2;
3949 			p->numa_faults[cpubuf_idx] = 0;
3950 
3951 			p->numa_faults[mem_idx] += diff;
3952 			p->numa_faults[cpu_idx] += f_diff;
3953 			faults += p->numa_faults[mem_idx];
3954 			p->total_numa_faults += diff;
3955 			if (ng) {
3956 				/*
3957 				 * safe because we can only change our own group
3958 				 *
3959 				 * mem_idx represents the offset for a given
3960 				 * nid and priv in a specific region because it
3961 				 * is at the beginning of the numa_faults array.
3962 				 */
3963 				ng->faults[mem_idx] += diff;
3964 				ng->faults[cpu_idx] += f_diff;
3965 				ng->total_faults += diff;
3966 				group_faults += ng->faults[mem_idx];
3967 			}
3968 #ifdef CONFIG_SCHED_CACHE
3969 			/*
3970 			 * Per task p->numa_faults[mem_idx] converges,
3971 			 * so the accumulation of each task's faults
3972 			 * converges too - Given the number of threads,
3973 			 * it cannot overflow an unsigned long.
3974 			 * Racy with concurrent updates from other threads
3975 			 * sharing this mm. Acceptable since footprint is a
3976 			 * heuristic and occasional lost updates are tolerable.
3977 			 *
3978 			 * If a task exits, its corresponding footprint must
3979 			 * be subtracted from p->sched_cache_grp->footprint,
3980 			 * otherwise the footprint will not converge: the
3981 			 * exiting thread's footprint remains unchanged/undecayed.
3982 			 * See exit_mm().
3983 			 *
3984 			 * Lost updates and unsynchronized subtraction
3985 			 * in exit_mm() can cause footprint + diff to
3986 			 * go negative. Clamp to zero to prevent the
3987 			 * unsigned footprint from wrapping.
3988 			 */
3989 			scoped_guard(rcu) {
3990 				grp = rcu_dereference(p->sched_cache_grp);
3991 
3992 				if (grp) {
3993 					new_fp = (long)READ_ONCE(grp->footprint) + diff;
3994 					WRITE_ONCE(grp->footprint, max(new_fp, 0L));
3995 				}
3996 			}
3997 #endif
3998 		}
3999 
4000 		if (!ng) {
4001 			if (faults > max_faults) {
4002 				max_faults = faults;
4003 				max_nid = nid;
4004 			}
4005 		} else if (group_faults > max_faults) {
4006 			max_faults = group_faults;
4007 			max_nid = nid;
4008 		}
4009 	}
4010 
4011 	/* Cannot migrate task to CPU-less node */
4012 	max_nid = numa_nearest_node(max_nid, N_CPU);
4013 
4014 	if (ng) {
4015 		numa_group_count_active_nodes(ng);
4016 		spin_unlock_irq(group_lock);
4017 		max_nid = preferred_group_nid(p, max_nid);
4018 	}
4019 
4020 	if (max_faults) {
4021 		/* Set the new preferred node */
4022 		if (max_nid != p->numa_preferred_nid)
4023 			sched_setnuma(p, max_nid);
4024 	}
4025 
4026 	update_task_scan_period(p, fault_types[0], fault_types[1]);
4027 }
4028 
4029 static inline int get_numa_group(struct numa_group *grp)
4030 {
4031 	return refcount_inc_not_zero(&grp->refcount);
4032 }
4033 
4034 static inline void put_numa_group(struct numa_group *grp)
4035 {
4036 	if (refcount_dec_and_test(&grp->refcount))
4037 		kfree_rcu(grp, rcu);
4038 }
4039 
4040 static void task_numa_group(struct task_struct *p, int cpupid, int flags,
4041 			int *priv)
4042 {
4043 	struct numa_group *grp, *my_grp;
4044 	struct task_struct *tsk;
4045 	bool join = false;
4046 	int cpu = cpupid_to_cpu(cpupid);
4047 	int i;
4048 
4049 	if (unlikely(!deref_curr_numa_group(p))) {
4050 		unsigned int size = sizeof(struct numa_group) +
4051 				    NR_NUMA_HINT_FAULT_STATS *
4052 				    nr_node_ids * sizeof(unsigned long);
4053 
4054 		grp = kzalloc(size, GFP_KERNEL | __GFP_NOWARN);
4055 		if (!grp)
4056 			return;
4057 
4058 		refcount_set(&grp->refcount, 1);
4059 		grp->active_nodes = 1;
4060 		grp->max_faults_cpu = 0;
4061 		spin_lock_init(&grp->lock);
4062 		grp->gid = p->pid;
4063 
4064 		for (i = 0; i < NR_NUMA_HINT_FAULT_STATS * nr_node_ids; i++)
4065 			grp->faults[i] = p->numa_faults[i];
4066 
4067 		grp->total_faults = p->total_numa_faults;
4068 
4069 		grp->nr_tasks++;
4070 		rcu_assign_pointer(p->numa_group, grp);
4071 	}
4072 
4073 	rcu_read_lock();
4074 	tsk = READ_ONCE(cpu_rq(cpu)->curr);
4075 
4076 	if (!cpupid_match_pid(tsk, cpupid))
4077 		goto no_join;
4078 
4079 	grp = rcu_dereference_all(tsk->numa_group);
4080 	if (!grp)
4081 		goto no_join;
4082 
4083 	my_grp = deref_curr_numa_group(p);
4084 	if (grp == my_grp)
4085 		goto no_join;
4086 
4087 	/*
4088 	 * Only join the other group if its bigger; if we're the bigger group,
4089 	 * the other task will join us.
4090 	 */
4091 	if (my_grp->nr_tasks > grp->nr_tasks)
4092 		goto no_join;
4093 
4094 	/*
4095 	 * Tie-break on the grp address.
4096 	 */
4097 	if (my_grp->nr_tasks == grp->nr_tasks && my_grp > grp)
4098 		goto no_join;
4099 
4100 	/* Always join threads in the same process. */
4101 	if (tsk->mm == current->mm)
4102 		join = true;
4103 
4104 	/* Simple filter to avoid false positives due to PID collisions */
4105 	if (flags & TNF_SHARED)
4106 		join = true;
4107 
4108 	/* Update priv based on whether false sharing was detected */
4109 	*priv = !join;
4110 
4111 	if (join && !get_numa_group(grp))
4112 		goto no_join;
4113 
4114 	rcu_read_unlock();
4115 
4116 	if (!join)
4117 		return;
4118 
4119 	WARN_ON_ONCE(irqs_disabled());
4120 	double_lock_irq(&my_grp->lock, &grp->lock);
4121 
4122 	for (i = 0; i < NR_NUMA_HINT_FAULT_STATS * nr_node_ids; i++) {
4123 		my_grp->faults[i] -= p->numa_faults[i];
4124 		grp->faults[i] += p->numa_faults[i];
4125 	}
4126 	my_grp->total_faults -= p->total_numa_faults;
4127 	grp->total_faults += p->total_numa_faults;
4128 
4129 	my_grp->nr_tasks--;
4130 	grp->nr_tasks++;
4131 
4132 	spin_unlock(&my_grp->lock);
4133 	spin_unlock_irq(&grp->lock);
4134 
4135 	rcu_assign_pointer(p->numa_group, grp);
4136 
4137 	put_numa_group(my_grp);
4138 	return;
4139 
4140 no_join:
4141 	rcu_read_unlock();
4142 	return;
4143 }
4144 
4145 /*
4146  * Get rid of NUMA statistics associated with a task (either current or dead).
4147  * If @final is set, the task is dead and has reached refcount zero, so we can
4148  * safely free all relevant data structures. Otherwise, there might be
4149  * concurrent reads from places like load balancing and procfs, and we should
4150  * reset the data back to default state without freeing ->numa_faults.
4151  */
4152 void task_numa_free(struct task_struct *p, bool final)
4153 {
4154 	/* safe: p either is current or is being freed by current */
4155 	struct numa_group *grp = rcu_dereference_raw(p->numa_group);
4156 	unsigned long *numa_faults = p->numa_faults;
4157 	unsigned long flags;
4158 	int i;
4159 
4160 	if (!numa_faults)
4161 		return;
4162 
4163 	if (grp) {
4164 		spin_lock_irqsave(&grp->lock, flags);
4165 		for (i = 0; i < NR_NUMA_HINT_FAULT_STATS * nr_node_ids; i++)
4166 			grp->faults[i] -= p->numa_faults[i];
4167 		grp->total_faults -= p->total_numa_faults;
4168 
4169 		grp->nr_tasks--;
4170 		spin_unlock_irqrestore(&grp->lock, flags);
4171 		RCU_INIT_POINTER(p->numa_group, NULL);
4172 		put_numa_group(grp);
4173 	}
4174 
4175 	if (final) {
4176 		p->numa_faults = NULL;
4177 		kfree(numa_faults);
4178 	} else {
4179 		p->total_numa_faults = 0;
4180 		for (i = 0; i < NR_NUMA_HINT_FAULT_STATS * nr_node_ids; i++)
4181 			numa_faults[i] = 0;
4182 	}
4183 }
4184 
4185 /*
4186  * Got a PROT_NONE fault for a page on @node.
4187  */
4188 void task_numa_fault(int last_cpupid, int mem_node, int pages, int flags)
4189 {
4190 	struct task_struct *p = current;
4191 	bool migrated = flags & TNF_MIGRATED;
4192 	int cpu_node = task_node(current);
4193 	int local = !!(flags & TNF_FAULT_LOCAL);
4194 	struct numa_group *ng;
4195 	int priv;
4196 
4197 	if (!static_branch_likely(&sched_numa_balancing))
4198 		return;
4199 
4200 	/* for example, ksmd faulting in a user's mm */
4201 	if (!p->mm)
4202 		return;
4203 
4204 	/*
4205 	 * NUMA faults statistics are unnecessary for the slow memory
4206 	 * node for memory tiering mode.
4207 	 */
4208 	if (!node_is_toptier(mem_node) &&
4209 	    (sysctl_numa_balancing_mode & NUMA_BALANCING_MEMORY_TIERING ||
4210 	     !cpupid_valid(last_cpupid)))
4211 		return;
4212 
4213 	/* Allocate buffer to track faults on a per-node basis */
4214 	if (unlikely(!p->numa_faults)) {
4215 		int size = sizeof(*p->numa_faults) *
4216 			   NR_NUMA_HINT_FAULT_BUCKETS * nr_node_ids;
4217 
4218 		p->numa_faults = kzalloc(size, GFP_KERNEL|__GFP_NOWARN);
4219 		if (!p->numa_faults)
4220 			return;
4221 
4222 		p->total_numa_faults = 0;
4223 		memset(p->numa_faults_locality, 0, sizeof(p->numa_faults_locality));
4224 	}
4225 
4226 	/*
4227 	 * First accesses are treated as private, otherwise consider accesses
4228 	 * to be private if the accessing pid has not changed
4229 	 */
4230 	if (unlikely(last_cpupid == (-1 & LAST_CPUPID_MASK))) {
4231 		priv = 1;
4232 	} else {
4233 		priv = cpupid_match_pid(p, last_cpupid);
4234 		if (!priv && !(flags & TNF_NO_GROUP))
4235 			task_numa_group(p, last_cpupid, flags, &priv);
4236 	}
4237 
4238 	/*
4239 	 * If a workload spans multiple NUMA nodes, a shared fault that
4240 	 * occurs wholly within the set of nodes that the workload is
4241 	 * actively using should be counted as local. This allows the
4242 	 * scan rate to slow down when a workload has settled down.
4243 	 */
4244 	ng = deref_curr_numa_group(p);
4245 	if (!priv && !local && ng && ng->active_nodes > 1 &&
4246 				numa_is_active_node(cpu_node, ng) &&
4247 				numa_is_active_node(mem_node, ng))
4248 		local = 1;
4249 
4250 	/*
4251 	 * Retry to migrate task to preferred node periodically, in case it
4252 	 * previously failed, or the scheduler moved us.
4253 	 */
4254 	if (time_after(jiffies, p->numa_migrate_retry)) {
4255 		task_numa_placement(p);
4256 		numa_migrate_preferred(p);
4257 	}
4258 
4259 	if (migrated)
4260 		p->numa_pages_migrated += pages;
4261 	if (flags & TNF_MIGRATE_FAIL)
4262 		p->numa_faults_locality[2] += pages;
4263 
4264 	p->numa_faults[task_faults_idx(NUMA_MEMBUF, mem_node, priv)] += pages;
4265 	p->numa_faults[task_faults_idx(NUMA_CPUBUF, cpu_node, priv)] += pages;
4266 	p->numa_faults_locality[local] += pages;
4267 }
4268 
4269 static void reset_ptenuma_scan(struct task_struct *p)
4270 {
4271 	/*
4272 	 * We only did a read acquisition of the mmap sem, so
4273 	 * p->mm->numa_scan_seq is written to without exclusive access
4274 	 * and the update is not guaranteed to be atomic. That's not
4275 	 * much of an issue though, since this is just used for
4276 	 * statistical sampling. Use READ_ONCE/WRITE_ONCE, which are not
4277 	 * expensive, to avoid any form of compiler optimizations:
4278 	 */
4279 	WRITE_ONCE(p->mm->numa_scan_seq, READ_ONCE(p->mm->numa_scan_seq) + 1);
4280 	p->mm->numa_scan_offset = 0;
4281 }
4282 
4283 static bool vma_is_accessed(struct mm_struct *mm, struct vm_area_struct *vma)
4284 {
4285 	unsigned long pids;
4286 	/*
4287 	 * Allow unconditional access first two times, so that all the (pages)
4288 	 * of VMAs get prot_none fault introduced irrespective of accesses.
4289 	 * This is also done to avoid any side effect of task scanning
4290 	 * amplifying the unfairness of disjoint set of VMAs' access.
4291 	 */
4292 	if ((READ_ONCE(current->mm->numa_scan_seq) - vma->numab_state->start_scan_seq) < 2)
4293 		return true;
4294 
4295 	pids = vma->numab_state->pids_active[0] | vma->numab_state->pids_active[1];
4296 	if (test_bit(hash_32(current->pid, ilog2(BITS_PER_LONG)), &pids))
4297 		return true;
4298 
4299 	/*
4300 	 * Complete a scan that has already started regardless of PID access, or
4301 	 * some VMAs may never be scanned in multi-threaded applications:
4302 	 */
4303 	if (mm->numa_scan_offset > vma->vm_start) {
4304 		trace_sched_skip_vma_numa(mm, vma, NUMAB_SKIP_IGNORE_PID);
4305 		return true;
4306 	}
4307 
4308 	/*
4309 	 * This vma has not been accessed for a while, and if the number
4310 	 * the threads in the same process is low, which means no other
4311 	 * threads can help scan this vma, force a vma scan.
4312 	 */
4313 	if (READ_ONCE(mm->numa_scan_seq) >
4314 	   (vma->numab_state->prev_scan_seq + get_nr_threads(current)))
4315 		return true;
4316 
4317 	return false;
4318 }
4319 
4320 #define VMA_PID_RESET_PERIOD (4 * sysctl_numa_balancing_scan_delay)
4321 
4322 /*
4323  * The expensive part of numa migration is done from task_work context.
4324  * Triggered from task_tick_numa().
4325  */
4326 static void task_numa_work(struct callback_head *work)
4327 {
4328 	unsigned long migrate, next_scan, now = jiffies;
4329 	struct task_struct *p = current;
4330 	struct mm_struct *mm = p->mm;
4331 	u64 runtime = p->se.sum_exec_runtime;
4332 	struct vm_area_struct *vma;
4333 	unsigned long start, end;
4334 	unsigned long nr_pte_updates = 0;
4335 	long pages, virtpages;
4336 	struct vma_iterator vmi;
4337 	bool vma_pids_skipped;
4338 	bool vma_pids_forced = false;
4339 
4340 	WARN_ON_ONCE(p != container_of(work, struct task_struct, numa_work));
4341 
4342 	work->next = work;
4343 	/*
4344 	 * Who cares about NUMA placement when they're dying.
4345 	 *
4346 	 * NOTE: make sure not to dereference p->mm before this check,
4347 	 * exit_task_work() happens _after_ exit_mm() so we could be called
4348 	 * without p->mm even though we still had it when we enqueued this
4349 	 * work.
4350 	 */
4351 	if (p->flags & PF_EXITING)
4352 		return;
4353 
4354 	/*
4355 	 * Memory is pinned to only one NUMA node via cpuset.mems, naturally
4356 	 * no page can be migrated.
4357 	 */
4358 	if (cpusets_enabled() && nodes_weight(cpuset_current_mems_allowed) == 1) {
4359 		trace_sched_skip_cpuset_numa(current, &cpuset_current_mems_allowed);
4360 		return;
4361 	}
4362 
4363 	if (!mm->numa_next_scan) {
4364 		mm->numa_next_scan = now +
4365 			msecs_to_jiffies(sysctl_numa_balancing_scan_delay);
4366 	}
4367 
4368 	/*
4369 	 * Enforce maximal scan/migration frequency..
4370 	 */
4371 	migrate = mm->numa_next_scan;
4372 	if (time_before(now, migrate))
4373 		return;
4374 
4375 	if (p->numa_scan_period == 0) {
4376 		p->numa_scan_period_max = task_scan_max(p);
4377 		p->numa_scan_period = task_scan_start(p);
4378 	}
4379 
4380 	next_scan = now + msecs_to_jiffies(p->numa_scan_period);
4381 	if (!try_cmpxchg(&mm->numa_next_scan, &migrate, next_scan))
4382 		return;
4383 
4384 	/*
4385 	 * Delay this task enough that another task of this mm will likely win
4386 	 * the next time around.
4387 	 */
4388 	p->node_stamp += 2 * TICK_NSEC;
4389 
4390 	pages = sysctl_numa_balancing_scan_size;
4391 	pages <<= 20 - PAGE_SHIFT; /* MB in pages */
4392 	virtpages = pages * 8;	   /* Scan up to this much virtual space */
4393 	if (!pages)
4394 		return;
4395 
4396 
4397 	if (!mmap_read_trylock(mm))
4398 		return;
4399 
4400 	/*
4401 	 * VMAs are skipped if the current PID has not trapped a fault within
4402 	 * the VMA recently. Allow scanning to be forced if there is no
4403 	 * suitable VMA remaining.
4404 	 */
4405 	vma_pids_skipped = false;
4406 
4407 retry_pids:
4408 	start = mm->numa_scan_offset;
4409 	vma_iter_init(&vmi, mm, start);
4410 	vma = vma_next(&vmi);
4411 	if (!vma) {
4412 		reset_ptenuma_scan(p);
4413 		start = 0;
4414 		vma_iter_set(&vmi, start);
4415 		vma = vma_next(&vmi);
4416 	}
4417 
4418 	for (; vma; vma = vma_next(&vmi)) {
4419 		if (!vma_migratable(vma) || !vma_policy_mof(vma) ||
4420 			is_vm_hugetlb_page(vma) || (vma->vm_flags & VM_MIXEDMAP)) {
4421 			trace_sched_skip_vma_numa(mm, vma, NUMAB_SKIP_UNSUITABLE);
4422 			continue;
4423 		}
4424 
4425 		/*
4426 		 * Shared library pages mapped by multiple processes are not
4427 		 * migrated as it is expected they are cache replicated. Avoid
4428 		 * hinting faults in read-only file-backed mappings or the vDSO
4429 		 * as migrating the pages will be of marginal benefit.
4430 		 */
4431 		if (!vma->vm_mm ||
4432 		    (vma->vm_file && (vma->vm_flags & (VM_READ|VM_WRITE)) == (VM_READ))) {
4433 			trace_sched_skip_vma_numa(mm, vma, NUMAB_SKIP_SHARED_RO);
4434 			continue;
4435 		}
4436 
4437 		/*
4438 		 * Skip inaccessible VMAs to avoid any confusion between
4439 		 * PROT_NONE and NUMA hinting PTEs
4440 		 */
4441 		if (!vma_is_accessible(vma)) {
4442 			trace_sched_skip_vma_numa(mm, vma, NUMAB_SKIP_INACCESSIBLE);
4443 			continue;
4444 		}
4445 
4446 		/* Initialise new per-VMA NUMAB state. */
4447 		if (!vma->numab_state) {
4448 			struct vma_numab_state *ptr;
4449 
4450 			ptr = kzalloc_obj(*ptr);
4451 			if (!ptr)
4452 				continue;
4453 
4454 			if (cmpxchg(&vma->numab_state, NULL, ptr)) {
4455 				kfree(ptr);
4456 				continue;
4457 			}
4458 
4459 			vma->numab_state->start_scan_seq = mm->numa_scan_seq;
4460 
4461 			vma->numab_state->next_scan = now +
4462 				msecs_to_jiffies(sysctl_numa_balancing_scan_delay);
4463 
4464 			/* Reset happens after 4 times scan delay of scan start */
4465 			vma->numab_state->pids_active_reset =  vma->numab_state->next_scan +
4466 				msecs_to_jiffies(VMA_PID_RESET_PERIOD);
4467 
4468 			/*
4469 			 * Ensure prev_scan_seq does not match numa_scan_seq,
4470 			 * to prevent VMAs being skipped prematurely on the
4471 			 * first scan:
4472 			 */
4473 			 vma->numab_state->prev_scan_seq = mm->numa_scan_seq - 1;
4474 		}
4475 
4476 		/*
4477 		 * Scanning the VMAs of short lived tasks add more overhead. So
4478 		 * delay the scan for new VMAs.
4479 		 */
4480 		if (mm->numa_scan_seq && time_before(jiffies,
4481 						vma->numab_state->next_scan)) {
4482 			trace_sched_skip_vma_numa(mm, vma, NUMAB_SKIP_SCAN_DELAY);
4483 			continue;
4484 		}
4485 
4486 		/* RESET access PIDs regularly for old VMAs. */
4487 		if (mm->numa_scan_seq &&
4488 				time_after(jiffies, vma->numab_state->pids_active_reset)) {
4489 			vma->numab_state->pids_active_reset = vma->numab_state->pids_active_reset +
4490 				msecs_to_jiffies(VMA_PID_RESET_PERIOD);
4491 			vma->numab_state->pids_active[0] = READ_ONCE(vma->numab_state->pids_active[1]);
4492 			vma->numab_state->pids_active[1] = 0;
4493 		}
4494 
4495 		/* Do not rescan VMAs twice within the same sequence. */
4496 		if (vma->numab_state->prev_scan_seq == mm->numa_scan_seq) {
4497 			mm->numa_scan_offset = vma->vm_end;
4498 			trace_sched_skip_vma_numa(mm, vma, NUMAB_SKIP_SEQ_COMPLETED);
4499 			continue;
4500 		}
4501 
4502 		/*
4503 		 * Do not scan the VMA if task has not accessed it, unless no other
4504 		 * VMA candidate exists.
4505 		 */
4506 		if (!vma_pids_forced && !vma_is_accessed(mm, vma)) {
4507 			vma_pids_skipped = true;
4508 			trace_sched_skip_vma_numa(mm, vma, NUMAB_SKIP_PID_INACTIVE);
4509 			continue;
4510 		}
4511 
4512 		do {
4513 			start = max(start, vma->vm_start);
4514 			end = ALIGN(start + (pages << PAGE_SHIFT), HPAGE_SIZE);
4515 			end = min(end, vma->vm_end);
4516 			nr_pte_updates = change_prot_numa(vma, start, end);
4517 
4518 			/*
4519 			 * Try to scan sysctl_numa_balancing_size worth of
4520 			 * hpages that have at least one present PTE that
4521 			 * is not already PTE-numa. If the VMA contains
4522 			 * areas that are unused or already full of prot_numa
4523 			 * PTEs, scan up to virtpages, to skip through those
4524 			 * areas faster.
4525 			 */
4526 			if (nr_pte_updates)
4527 				pages -= (end - start) >> PAGE_SHIFT;
4528 			virtpages -= (end - start) >> PAGE_SHIFT;
4529 
4530 			start = end;
4531 			if (pages <= 0 || virtpages <= 0)
4532 				goto out;
4533 
4534 			cond_resched();
4535 		} while (end != vma->vm_end);
4536 
4537 		/* VMA scan is complete, do not scan until next sequence. */
4538 		vma->numab_state->prev_scan_seq = mm->numa_scan_seq;
4539 
4540 		/*
4541 		 * Only force scan within one VMA at a time, to limit the
4542 		 * cost of scanning a potentially uninteresting VMA.
4543 		 */
4544 		if (vma_pids_forced)
4545 			break;
4546 	}
4547 
4548 	/*
4549 	 * If no VMAs are remaining and VMAs were skipped due to the PID
4550 	 * not accessing the VMA previously, then force a scan to ensure
4551 	 * forward progress:
4552 	 */
4553 	if (!vma && !vma_pids_forced && vma_pids_skipped) {
4554 		vma_pids_forced = true;
4555 		goto retry_pids;
4556 	}
4557 
4558 out:
4559 	/*
4560 	 * It is possible to reach the end of the VMA list but the last few
4561 	 * VMAs are not guaranteed to the vma_migratable. If they are not, we
4562 	 * would find the !migratable VMA on the next scan but not reset the
4563 	 * scanner to the start so check it now.
4564 	 */
4565 	if (vma)
4566 		mm->numa_scan_offset = start;
4567 	else
4568 		reset_ptenuma_scan(p);
4569 	mmap_read_unlock(mm);
4570 
4571 	/*
4572 	 * Make sure tasks use at least 32x as much time to run other code
4573 	 * than they used here, to limit NUMA PTE scanning overhead to 3% max.
4574 	 * Usually update_task_scan_period slows down scanning enough; on an
4575 	 * overloaded system we need to limit overhead on a per task basis.
4576 	 */
4577 	if (unlikely(p->se.sum_exec_runtime != runtime)) {
4578 		u64 diff = p->se.sum_exec_runtime - runtime;
4579 		p->node_stamp += 32 * diff;
4580 	}
4581 }
4582 
4583 void init_numa_balancing(u64 clone_flags, struct task_struct *p)
4584 {
4585 	int mm_users = 0;
4586 	struct mm_struct *mm = p->mm;
4587 
4588 	if (mm) {
4589 		mm_users = atomic_read(&mm->mm_users);
4590 		if (mm_users == 1) {
4591 			mm->numa_next_scan = jiffies + msecs_to_jiffies(sysctl_numa_balancing_scan_delay);
4592 			mm->numa_scan_seq = 0;
4593 		}
4594 	}
4595 	p->node_stamp			= 0;
4596 	p->numa_scan_seq		= mm ? mm->numa_scan_seq : 0;
4597 	p->numa_scan_period		= sysctl_numa_balancing_scan_delay;
4598 	p->numa_migrate_retry		= 0;
4599 	/* Protect against double add, see task_tick_numa and task_numa_work */
4600 	p->numa_work.next		= &p->numa_work;
4601 	p->numa_faults			= NULL;
4602 	p->numa_pages_migrated		= 0;
4603 	p->total_numa_faults		= 0;
4604 	RCU_INIT_POINTER(p->numa_group, NULL);
4605 	p->last_task_numa_placement	= 0;
4606 	p->last_sum_exec_runtime	= 0;
4607 
4608 	init_task_work(&p->numa_work, task_numa_work);
4609 
4610 	/* New address space, reset the preferred nid */
4611 	if (!(clone_flags & CLONE_VM)) {
4612 		p->numa_preferred_nid = NUMA_NO_NODE;
4613 		return;
4614 	}
4615 
4616 	/*
4617 	 * New thread, keep existing numa_preferred_nid which should be copied
4618 	 * already by arch_dup_task_struct but stagger when scans start.
4619 	 */
4620 	if (mm) {
4621 		unsigned int delay;
4622 
4623 		delay = min_t(unsigned int, task_scan_max(current),
4624 			current->numa_scan_period * mm_users * NSEC_PER_MSEC);
4625 		delay += 2 * TICK_NSEC;
4626 		p->node_stamp = delay;
4627 	}
4628 }
4629 
4630 /*
4631  * Drive the periodic memory faults..
4632  */
4633 static void task_tick_numa(struct rq *rq, struct task_struct *curr)
4634 {
4635 	struct callback_head *work = &curr->numa_work;
4636 	u64 period, now;
4637 
4638 	/*
4639 	 * We don't care about NUMA placement if we don't have memory.
4640 	 */
4641 	if (!curr->mm || (curr->flags & (PF_EXITING | PF_KTHREAD)) || work->next != work)
4642 		return;
4643 
4644 	/*
4645 	 * Using runtime rather than walltime has the dual advantage that
4646 	 * we (mostly) drive the selection from busy threads and that the
4647 	 * task needs to have done some actual work before we bother with
4648 	 * NUMA placement.
4649 	 */
4650 	now = curr->se.sum_exec_runtime;
4651 	period = (u64)curr->numa_scan_period * NSEC_PER_MSEC;
4652 
4653 	if (now > curr->node_stamp + period) {
4654 		if (!curr->node_stamp)
4655 			curr->numa_scan_period = task_scan_start(curr);
4656 		curr->node_stamp += period;
4657 
4658 		if (!time_before(jiffies, curr->mm->numa_next_scan))
4659 			task_work_add(curr, work, TWA_RESUME);
4660 	}
4661 }
4662 
4663 static void update_scan_period(struct task_struct *p, int new_cpu)
4664 {
4665 	int src_nid = cpu_to_node(task_cpu(p));
4666 	int dst_nid = cpu_to_node(new_cpu);
4667 
4668 	if (!static_branch_likely(&sched_numa_balancing))
4669 		return;
4670 
4671 	if (!p->mm || !p->numa_faults || (p->flags & PF_EXITING))
4672 		return;
4673 
4674 	if (src_nid == dst_nid)
4675 		return;
4676 
4677 	/*
4678 	 * Allow resets if faults have been trapped before one scan
4679 	 * has completed. This is most likely due to a new task that
4680 	 * is pulled cross-node due to wakeups or load balancing.
4681 	 */
4682 	if (p->numa_scan_seq) {
4683 		/*
4684 		 * Avoid scan adjustments if moving to the preferred
4685 		 * node or if the task was not previously running on
4686 		 * the preferred node.
4687 		 */
4688 		if (dst_nid == p->numa_preferred_nid ||
4689 		    (p->numa_preferred_nid != NUMA_NO_NODE &&
4690 			src_nid != p->numa_preferred_nid))
4691 			return;
4692 	}
4693 
4694 	p->numa_scan_period = task_scan_start(p);
4695 }
4696 
4697 #else /* !CONFIG_NUMA_BALANCING: */
4698 
4699 static void task_tick_numa(struct rq *rq, struct task_struct *curr)
4700 {
4701 }
4702 
4703 static inline void account_numa_enqueue(struct rq *rq, struct task_struct *p)
4704 {
4705 }
4706 
4707 static inline void account_numa_dequeue(struct rq *rq, struct task_struct *p)
4708 {
4709 }
4710 
4711 static inline void update_scan_period(struct task_struct *p, int new_cpu)
4712 {
4713 }
4714 
4715 #endif /* !CONFIG_NUMA_BALANCING */
4716 
4717 static void
4718 account_entity_enqueue(struct cfs_rq *cfs_rq, struct sched_entity *se)
4719 {
4720 	WARN_ON_ONCE(cfs_rq != cfs_rq_of(se));
4721 	update_load_add(&cfs_rq->load, se->load.weight);
4722 	if (entity_is_task(se)) {
4723 		struct task_struct *p = task_of(se);
4724 		struct rq *rq = rq_of(cfs_rq);
4725 
4726 		account_numa_enqueue(rq, p);
4727 		account_llc_enqueue(rq, p);
4728 		list_add(&se->group_node, &rq->cfs_tasks);
4729 	}
4730 	cfs_rq->nr_queued++;
4731 }
4732 
4733 static void
4734 account_entity_dequeue(struct cfs_rq *cfs_rq, struct sched_entity *se)
4735 {
4736 	WARN_ON_ONCE(cfs_rq != cfs_rq_of(se));
4737 	update_load_sub(&cfs_rq->load, se->load.weight);
4738 	if (entity_is_task(se)) {
4739 		struct task_struct *p = task_of(se);
4740 		struct rq *rq = rq_of(cfs_rq);
4741 
4742 		account_numa_dequeue(rq, p);
4743 		account_llc_dequeue(rq, p);
4744 		list_del_init(&se->group_node);
4745 	}
4746 	cfs_rq->nr_queued--;
4747 }
4748 
4749 /*
4750  * Signed add and clamp on underflow.
4751  *
4752  * Explicitly do a load-store to ensure the intermediate value never hits
4753  * memory. This allows lockless observations without ever seeing the negative
4754  * values.
4755  */
4756 #define add_positive(_ptr, _val) do {                           \
4757 	typeof(_ptr) ptr = (_ptr);                              \
4758 	__signed_scalar_typeof(*ptr) val = (_val);              \
4759 	typeof(*ptr) res, var = READ_ONCE(*ptr);                \
4760 								\
4761 	res = var + val;                                        \
4762 								\
4763 	if (val < 0 && res > var)                               \
4764 		res = 0;                                        \
4765 								\
4766 	WRITE_ONCE(*ptr, res);                                  \
4767 } while (0)
4768 
4769 /*
4770  * Remove and clamp on negative, from a local variable.
4771  *
4772  * A variant of sub_positive(), which does not use explicit load-store
4773  * and is thus optimized for local variable updates.
4774  */
4775 #define lsub_positive(_ptr, _val) do {				\
4776 	typeof(_ptr) ptr = (_ptr);				\
4777 	*ptr -= min_t(typeof(*ptr), *ptr, _val);		\
4778 } while (0)
4779 
4780 
4781 /*
4782  * Because of rounding, se->util_sum might ends up being +1 more than
4783  * cfs->util_sum. Although this is not a problem by itself, detaching
4784  * a lot of tasks with the rounding problem between 2 updates of
4785  * util_avg (~1ms) can make cfs->util_sum becoming null whereas
4786  * cfs_util_avg is not.
4787  *
4788  * Check that util_sum is still above its lower bound for the new
4789  * util_avg. Given that period_contrib might have moved since the last
4790  * sync, we are only sure that util_sum must be above or equal to
4791  *    util_avg * minimum possible divider
4792  */
4793 #define __update_sa(sa, name, delta_avg, delta_sum) do {	\
4794 	add_positive(&(sa)->name##_avg, delta_avg);		\
4795 	add_positive(&(sa)->name##_sum, delta_sum);		\
4796 	(sa)->name##_sum = max_t(typeof((sa)->name##_sum),	\
4797 			       (sa)->name##_sum,		\
4798 			       (sa)->name##_avg * PELT_MIN_DIVIDER); \
4799 } while (0)
4800 
4801 static inline void
4802 enqueue_load_avg(struct cfs_rq *cfs_rq, struct sched_entity *se)
4803 {
4804 	__update_sa(&cfs_rq->avg, load, se->avg.load_avg,
4805 		    se_weight(se) * se->avg.load_sum);
4806 }
4807 
4808 static inline void
4809 dequeue_load_avg(struct cfs_rq *cfs_rq, struct sched_entity *se)
4810 {
4811 	__update_sa(&cfs_rq->avg, load, -se->avg.load_avg,
4812 		    se_weight(se) * -se->avg.load_sum);
4813 }
4814 
4815 static void
4816 rescale_entity(struct sched_entity *se, unsigned long weight, bool rel_vprot)
4817 {
4818 	long old_weight = se->h_load.weight;
4819 
4820 	/*
4821 	 * VRUNTIME
4822 	 * --------
4823 	 *
4824 	 * COROLLARY #1: The virtual runtime of the entity needs to be
4825 	 * adjusted if re-weight at !0-lag point.
4826 	 *
4827 	 * Proof: For contradiction assume this is not true, so we can
4828 	 * re-weight without changing vruntime at !0-lag point.
4829 	 *
4830 	 *             Weight	VRuntime   Avg-VRuntime
4831 	 *     before    w          v            V
4832 	 *      after    w'         v'           V'
4833 	 *
4834 	 * Since lag needs to be preserved through re-weight:
4835 	 *
4836 	 *	lag = (V - v)*w = (V'- v')*w', where v = v'
4837 	 *	==>	V' = (V - v)*w/w' + v		(1)
4838 	 *
4839 	 * Let W be the total weight of the entities before reweight,
4840 	 * since V' is the new weighted average of entities:
4841 	 *
4842 	 *	V' = (WV + w'v - wv) / (W + w' - w)	(2)
4843 	 *
4844 	 * by using (1) & (2) we obtain:
4845 	 *
4846 	 *	(WV + w'v - wv) / (W + w' - w) = (V - v)*w/w' + v
4847 	 *	==> (WV-Wv+Wv+w'v-wv)/(W+w'-w) = (V - v)*w/w' + v
4848 	 *	==> (WV - Wv)/(W + w' - w) + v = (V - v)*w/w' + v
4849 	 *	==>	(V - v)*W/(W + w' - w) = (V - v)*w/w' (3)
4850 	 *
4851 	 * Since we are doing at !0-lag point which means V != v, we
4852 	 * can simplify (3):
4853 	 *
4854 	 *	==>	W / (W + w' - w) = w / w'
4855 	 *	==>	Ww' = Ww + ww' - ww
4856 	 *	==>	W * (w' - w) = w * (w' - w)
4857 	 *	==>	W = w	(re-weight indicates w' != w)
4858 	 *
4859 	 * So the cfs_rq contains only one entity, hence vruntime of
4860 	 * the entity @v should always equal to the cfs_rq's weighted
4861 	 * average vruntime @V, which means we will always re-weight
4862 	 * at 0-lag point, thus breach assumption. Proof completed.
4863 	 *
4864 	 *
4865 	 * COROLLARY #2: Re-weight does NOT affect weighted average
4866 	 * vruntime of all the entities.
4867 	 *
4868 	 * Proof: According to corollary #1, Eq. (1) should be:
4869 	 *
4870 	 *	(V - v)*w = (V' - v')*w'
4871 	 *	==>    v' = V' - (V - v)*w/w'		(4)
4872 	 *
4873 	 * According to the weighted average formula, we have:
4874 	 *
4875 	 *	V' = (WV - wv + w'v') / (W - w + w')
4876 	 *	   = (WV - wv + w'(V' - (V - v)w/w')) / (W - w + w')
4877 	 *	   = (WV - wv + w'V' - Vw + wv) / (W - w + w')
4878 	 *	   = (WV + w'V' - Vw) / (W - w + w')
4879 	 *
4880 	 *	==>  V'*(W - w + w') = WV + w'V' - Vw
4881 	 *	==>	V' * (W - w) = (W - w) * V	(5)
4882 	 *
4883 	 * If the entity is the only one in the cfs_rq, then reweight
4884 	 * always occurs at 0-lag point, so V won't change. Or else
4885 	 * there are other entities, hence W != w, then Eq. (5) turns
4886 	 * into V' = V. So V won't change in either case, proof done.
4887 	 *
4888 	 *
4889 	 * So according to corollary #1 & #2, the effect of re-weight
4890 	 * on vruntime should be:
4891 	 *
4892 	 *	v' = V' - (V - v) * w / w'		(4)
4893 	 *	   = V  - (V - v) * w / w'
4894 	 *	   = V  - vl * w / w'
4895 	 *	   = V  - vl'
4896 	 */
4897 	se->vlag = div64_long(se->vlag * old_weight, weight);
4898 
4899 	/*
4900 	 * DEADLINE
4901 	 * --------
4902 	 *
4903 	 * When the weight changes, the virtual time slope changes and
4904 	 * we should adjust the relative virtual deadline accordingly.
4905 	 *
4906 	 *	d' = v' + (d - v)*w/w'
4907 	 *	   = V' - (V - v)*w/w' + (d - v)*w/w'
4908 	 *	   = V  - (V - v)*w/w' + (d - v)*w/w'
4909 	 *	   = V  + (d - V)*w/w'
4910 	 */
4911 	if (se->rel_deadline)
4912 		se->deadline = div64_long(se->deadline * old_weight, weight);
4913 
4914 	if (rel_vprot)
4915 		se->vprot = div64_long(se->vprot * old_weight, weight);
4916 }
4917 
4918 static void reweight_eevdf(struct cfs_rq *cfs_rq, struct sched_entity *se,
4919 			   unsigned long weight, bool on_rq)
4920 {
4921 	bool curr = cfs_rq->curr == se;
4922 	bool rel_vprot = false;
4923 	u64 avruntime = 0;
4924 
4925 	if (se->h_load.weight == weight)
4926 		return;
4927 
4928 	if (on_rq) {
4929 		avruntime = avg_vruntime(cfs_rq);
4930 		se->vlag = entity_lag(cfs_rq, se, avruntime);
4931 		se->deadline -= avruntime;
4932 		se->rel_deadline = 1;
4933 		if (curr && protect_slice(se)) {
4934 			se->vprot -= avruntime;
4935 			rel_vprot = true;
4936 		}
4937 
4938 		cfs_rq->h_nr_queued--;
4939 		if (!curr)
4940 			__dequeue_entity(cfs_rq, se);
4941 	}
4942 
4943 	rescale_entity(se, weight, rel_vprot);
4944 
4945 	update_load_set(&se->h_load, weight);
4946 
4947 	if (on_rq) {
4948 		if (rel_vprot)
4949 			se->vprot += avruntime;
4950 		se->deadline += avruntime;
4951 		se->rel_deadline = 0;
4952 		se->vruntime = avruntime - se->vlag;
4953 
4954 		if (!curr)
4955 			__enqueue_entity(cfs_rq, se);
4956 		cfs_rq->h_nr_queued++;
4957 	}
4958 }
4959 
4960 static void reweight_entity(struct cfs_rq *cfs_rq, struct sched_entity *se,
4961 			    unsigned long weight)
4962 {
4963 	if (se->load.weight == weight)
4964 		return;
4965 
4966 	if (se->on_rq) {
4967 		WARN_ON_ONCE(cfs_rq != cfs_rq_of(se));
4968 		update_load_sub(&cfs_rq->load, se->load.weight);
4969 	}
4970 	dequeue_load_avg(cfs_rq, se);
4971 
4972 	update_load_set(&se->load, weight);
4973 
4974 	do {
4975 		u32 divider = get_pelt_divider(&se->avg);
4976 		se->avg.load_avg = div_u64(se_weight(se) * se->avg.load_sum, divider);
4977 	} while (0);
4978 
4979 	enqueue_load_avg(cfs_rq, se);
4980 
4981 	if (se->on_rq)
4982 		update_load_add(&cfs_rq->load, se->load.weight);
4983 }
4984 
4985 /*
4986  * weight = NICE_0_LOAD;
4987  * for_each_entity_se(se)
4988  *   weight = __calc_prop_weight(cfs_rq_of(se), se, weight);
4989  */
4990 static __always_inline
4991 unsigned long __calc_prop_weight(struct cfs_rq *cfs_rq, struct sched_entity *se,
4992 				 unsigned long weight)
4993 {
4994 	weight *= se->load.weight;
4995 	if (parent_entity(se))
4996 		weight /= cfs_rq->load.weight;
4997 	else
4998 		weight /= NICE_0_LOAD;
4999 
5000 	return max(weight, MIN_SHARES);
5001 }
5002 
5003 static void reweight_task_fair(struct rq *rq, struct task_struct *p,
5004 			       const struct load_weight *lw)
5005 {
5006 	struct sched_entity *se = &p->se;
5007 	unsigned long weight = NICE_0_LOAD;
5008 
5009 	if (se->on_rq)
5010 		update_curr_fair(rq);
5011 
5012 	reweight_entity(cfs_rq_of(se), se, lw->weight);
5013 	se->load.inv_weight = lw->inv_weight;
5014 
5015 	if (!se->on_rq)
5016 		return;
5017 
5018 	for_each_sched_entity(se)
5019 		weight = __calc_prop_weight(cfs_rq_of(se), se, weight);
5020 
5021 	reweight_eevdf(&rq->cfs, &p->se, weight, p->se.on_rq);
5022 }
5023 
5024 static inline int throttled_hierarchy(struct cfs_rq *cfs_rq);
5025 
5026 #ifdef CONFIG_FAIR_GROUP_SCHED
5027 /*
5028  * All this does is approximate the hierarchical proportion which includes that
5029  * global sum we all love to hate.
5030  *
5031  * That is, the weight of a group entity, is the proportional share of the
5032  * group weight based on the group runqueue weights. That is:
5033  *
5034  *                     tg->weight * grq->load.weight
5035  *   ge->load.weight = -----------------------------               (1)
5036  *                       \Sum grq->load.weight
5037  *
5038  * Now, because computing that sum is prohibitively expensive to compute (been
5039  * there, done that) we approximate it with this average stuff. The average
5040  * moves slower and therefore the approximation is cheaper and more stable.
5041  *
5042  * So instead of the above, we substitute:
5043  *
5044  *   grq->load.weight -> grq->avg.load_avg                         (2)
5045  *
5046  * which yields the following:
5047  *
5048  *                     tg->weight * grq->avg.load_avg
5049  *   ge->load.weight = ------------------------------              (3)
5050  *                             tg->load_avg
5051  *
5052  * Where: tg->load_avg ~= \Sum grq->avg.load_avg
5053  *
5054  * That is shares_avg, and it is right (given the approximation (2)).
5055  *
5056  * The problem with it is that because the average is slow -- it was designed
5057  * to be exactly that of course -- this leads to transients in boundary
5058  * conditions. In specific, the case where the group was idle and we start the
5059  * one task. It takes time for our CPU's grq->avg.load_avg to build up,
5060  * yielding bad latency etc..
5061  *
5062  * Now, in that special case (1) reduces to:
5063  *
5064  *                     tg->weight * grq->load.weight
5065  *   ge->load.weight = ----------------------------- = tg->weight   (4)
5066  *                         grp->load.weight
5067  *
5068  * That is, the sum collapses because all other CPUs are idle; the UP scenario.
5069  *
5070  * So what we do is modify our approximation (3) to approach (4) in the (near)
5071  * UP case, like:
5072  *
5073  *   ge->load.weight =
5074  *
5075  *              tg->weight * grq->load.weight
5076  *     ---------------------------------------------------         (5)
5077  *     tg->load_avg - grq->avg.load_avg + grq->load.weight
5078  *
5079  * But because grq->load.weight can drop to 0, resulting in a divide by zero,
5080  * we need to use grq->avg.load_avg as its lower bound, which then gives:
5081  *
5082  *
5083  *                     tg->weight * grq->load.weight
5084  *   ge->load.weight = -----------------------------		   (6)
5085  *                             tg_load_avg'
5086  *
5087  * Where:
5088  *
5089  *   tg_load_avg' = tg->load_avg - grq->avg.load_avg +
5090  *                  max(grq->load.weight, grq->avg.load_avg)
5091  *
5092  * And that is shares_weight and is icky. In the (near) UP case it approaches
5093  * (4) while in the normal case it approaches (3). It consistently
5094  * overestimates the ge->load.weight and therefore:
5095  *
5096  *   \Sum ge->load.weight >= tg->weight
5097  *
5098  * hence icky!
5099  */
5100 static long __calc_smp_shares(struct cfs_rq *cfs_rq, long tg_shares, long shares_max)
5101 {
5102 	struct task_group *tg = cfs_rq->tg;
5103 	long tg_weight, load, shares;
5104 
5105 	load = max(scale_load_down(cfs_rq->load.weight), cfs_rq->avg.load_avg);
5106 
5107 	tg_weight = atomic_long_read(&tg->load_avg);
5108 
5109 	/* Ensure tg_weight >= load */
5110 	tg_weight -= cfs_rq->tg_load_avg_contrib;
5111 	tg_weight += load;
5112 
5113 	shares = (tg_shares * load);
5114 	if (tg_weight)
5115 		shares /= tg_weight;
5116 
5117 	/*
5118 	 * MIN_SHARES has to be unscaled here to support per-CPU partitioning
5119 	 * of a group with small tg->shares value. It is a floor value which is
5120 	 * assigned as a minimum load.weight to the sched_entity representing
5121 	 * the group on a CPU.
5122 	 *
5123 	 * E.g. on 64-bit for a group with tg->shares of scale_load(15)=15*1024
5124 	 * on an 8-core system with 8 tasks each runnable on one CPU shares has
5125 	 * to be 15*1024*1/8=1920 instead of scale_load(MIN_SHARES)=2*1024. In
5126 	 * case no task is runnable on a CPU MIN_SHARES=2 should be returned
5127 	 * instead of 0.
5128 	 */
5129 	return clamp_t(long, shares, MIN_SHARES, shares_max);
5130 }
5131 
5132 static int tg_cpus(struct task_group *tg)
5133 {
5134 	int nr = num_online_cpus();
5135 
5136 	if (cpusets_enabled()) {
5137 		struct cgroup *cgrp = tg->css.cgroup;
5138 		if (cgrp)
5139 			nr = cpuset_num_cpus(cgrp);
5140 	}
5141 
5142 	/*
5143 	 * An empty cpuset would propagate a 0 shares_max into
5144 	 * __calc_smp_shares(), where clamp() yields hi when hi < lo and so
5145 	 * defeats the MIN_SHARES floor. Match tg_tasks(), which floors at 1.
5146 	 */
5147 	return max(nr, 1);
5148 }
5149 
5150 static inline int tg_tasks(struct task_group *tg)
5151 {
5152 	return max(1, atomic_long_read(&tg->runnable_avg) >> SCHED_CAPACITY_SHIFT);
5153 }
5154 
5155 /*
5156  * Func: fraction(nr_tasks * tg->shares)
5157  *
5158  * Scale tg->shares by the number of tasks.
5159  */
5160 static long calc_tasks_shares(struct cfs_rq *cfs_rq)
5161 {
5162 	struct task_group *tg = cfs_rq->tg;
5163 	int nr = tg_tasks(tg);
5164 	long tg_shares = READ_ONCE(tg->shares);
5165 	return __calc_smp_shares(cfs_rq, nr * tg_shares, nr * tg_shares);
5166 }
5167 
5168 /*
5169  * Func: min(fraction(nr_cpus * tg->shares), nice -20)
5170  *
5171  * Scale tg->shares by the maximal number of CPUs; but clip the max shares at
5172  * nice -20, otherwise a single spinner on a 512 CPU machine would result in
5173  * 512*NICE_0_LOAD, which is also crazy.
5174  */
5175 static long calc_max_shares(struct cfs_rq *cfs_rq)
5176 {
5177 	struct task_group *tg = cfs_rq->tg;
5178 	int nr = tg_cpus(tg);
5179 	long tg_shares = READ_ONCE(tg->shares);
5180 	long max_shares = scale_load(sched_prio_to_weight[0]);
5181 	return __calc_smp_shares(cfs_rq, tg_shares * nr, max_shares);
5182 }
5183 
5184 /*
5185  * Func: fraction(nr * tg->shares); nr = min(nr_tasks, nr_cpus)
5186  *
5187  * Scales between "smp" and "max" in a natural way. No longer needs clipping
5188  * since there are no unnatural inflations like with "max".
5189  */
5190 static long calc_concur_shares(struct cfs_rq *cfs_rq)
5191 {
5192 	struct task_group *tg = cfs_rq->tg;
5193 	int nr = min(tg_tasks(tg), tg_cpus(tg));
5194 	long tg_shares = READ_ONCE(tg->shares);
5195 	return __calc_smp_shares(cfs_rq, nr * tg_shares, nr * tg_shares);
5196 }
5197 
5198 /*
5199  * Func: fraction(tg->shares)
5200  *
5201  * This infamously results in tiny shares when you have many CPUs.
5202  */
5203 static long calc_smp_shares(struct cfs_rq *cfs_rq)
5204 {
5205 	struct task_group *tg = cfs_rq->tg;
5206 	long tg_shares = READ_ONCE(tg->shares);
5207 	return __calc_smp_shares(cfs_rq, tg_shares, tg_shares);
5208 }
5209 
5210 /*
5211  * Ignore this pesky SMP stuff, use (4).
5212  */
5213 static long calc_up_shares(struct cfs_rq *cfs_rq)
5214 {
5215 	struct task_group *tg = cfs_rq->tg;
5216 	return READ_ONCE(tg->shares);
5217 }
5218 
5219 DEFINE_STATIC_CALL(calc_group_shares, calc_concur_shares);
5220 
5221 void __sched_cgroup_mode_update(int mode)
5222 {
5223 	long (*func)(struct cfs_rq *);
5224 	switch (mode) {
5225 	case 0:
5226 		func = &calc_up_shares;
5227 		break;
5228 	case 1:
5229 		func = &calc_smp_shares;
5230 		break;
5231 	case 2:
5232 	default:
5233 		func = &calc_concur_shares;
5234 		break;
5235 	case 3:
5236 		func = &calc_max_shares;
5237 		break;
5238 	case 4:
5239 		func = &calc_tasks_shares;
5240 		break;
5241 	}
5242 	static_call_update(calc_group_shares, func);
5243 }
5244 
5245 /*
5246  * Recomputes the group entity based on the current state of its group
5247  * runqueue.
5248  */
5249 static void update_cfs_group(struct sched_entity *se)
5250 {
5251 	struct cfs_rq *gcfs_rq = group_cfs_rq(se);
5252 	long shares;
5253 
5254 	/*
5255 	 * When a group becomes empty, preserve its weight. This matters for
5256 	 * DELAY_DEQUEUE.
5257 	 */
5258 	if (!gcfs_rq || !gcfs_rq->load.weight)
5259 		return;
5260 
5261 	shares = static_call(calc_group_shares)(gcfs_rq);
5262 	reweight_entity(cfs_rq_of(se), se, shares);
5263 }
5264 
5265 #else /* !CONFIG_FAIR_GROUP_SCHED: */
5266 static inline void update_cfs_group(struct sched_entity *se)
5267 {
5268 }
5269 #endif /* !CONFIG_FAIR_GROUP_SCHED */
5270 
5271 static inline void cfs_rq_util_change(struct cfs_rq *cfs_rq, int flags)
5272 {
5273 	struct rq *rq = rq_of(cfs_rq);
5274 
5275 	if (&rq->cfs == cfs_rq) {
5276 		/*
5277 		 * There are a few boundary cases this might miss but it should
5278 		 * get called often enough that that should (hopefully) not be
5279 		 * a real problem.
5280 		 *
5281 		 * It will not get called when we go idle, because the idle
5282 		 * thread is a different class (!fair), nor will the utilization
5283 		 * number include things like RT tasks.
5284 		 *
5285 		 * As is, the util number is not freq-invariant (we'd have to
5286 		 * implement arch_scale_freq_capacity() for that).
5287 		 *
5288 		 * See cpu_util_cfs().
5289 		 */
5290 		cpufreq_update_util(rq, flags);
5291 	}
5292 }
5293 
5294 static inline bool load_avg_is_decayed(struct sched_avg *sa)
5295 {
5296 	if (sa->load_sum)
5297 		return false;
5298 
5299 	if (sa->util_sum)
5300 		return false;
5301 
5302 	if (sa->runnable_sum)
5303 		return false;
5304 
5305 	/*
5306 	 * _avg must be null when _sum are null because _avg = _sum / divider
5307 	 * Make sure that rounding and/or propagation of PELT values never
5308 	 * break this.
5309 	 */
5310 	WARN_ON_ONCE(sa->load_avg ||
5311 		      sa->util_avg ||
5312 		      sa->runnable_avg);
5313 
5314 	return true;
5315 }
5316 
5317 static inline u64 cfs_rq_last_update_time(struct cfs_rq *cfs_rq)
5318 {
5319 	return u64_u32_load_copy(cfs_rq->avg.last_update_time,
5320 				 cfs_rq->last_update_time_copy);
5321 }
5322 #ifdef CONFIG_FAIR_GROUP_SCHED
5323 /*
5324  * Because list_add_leaf_cfs_rq always places a child cfs_rq on the list
5325  * immediately before a parent cfs_rq, and cfs_rqs are removed from the list
5326  * bottom-up, we only have to test whether the cfs_rq before us on the list
5327  * is our child.
5328  * If cfs_rq is not on the list, test whether a child needs its to be added to
5329  * connect a branch to the tree  * (see list_add_leaf_cfs_rq() for details).
5330  */
5331 static inline bool child_cfs_rq_on_list(struct cfs_rq *cfs_rq)
5332 {
5333 	struct cfs_rq *prev_cfs_rq;
5334 	struct list_head *prev;
5335 	struct rq *rq = rq_of(cfs_rq);
5336 
5337 	if (cfs_rq->on_list) {
5338 		prev = cfs_rq->leaf_cfs_rq_list.prev;
5339 	} else {
5340 		prev = rq->tmp_alone_branch;
5341 	}
5342 
5343 	if (prev == &rq->leaf_cfs_rq_list)
5344 		return false;
5345 
5346 	prev_cfs_rq = container_of(prev, struct cfs_rq, leaf_cfs_rq_list);
5347 
5348 	return (prev_cfs_rq->tg->parent == cfs_rq->tg);
5349 }
5350 
5351 static inline bool cfs_rq_is_decayed(struct cfs_rq *cfs_rq)
5352 {
5353 	if (cfs_rq->load.weight)
5354 		return false;
5355 
5356 	if (!load_avg_is_decayed(&cfs_rq->avg))
5357 		return false;
5358 
5359 	if (child_cfs_rq_on_list(cfs_rq))
5360 		return false;
5361 
5362 	if (cfs_rq->tg_load_avg_contrib)
5363 		return false;
5364 
5365 	return true;
5366 }
5367 
5368 /**
5369  * update_tg_load_avg - update the tg's load avg
5370  * @cfs_rq: the cfs_rq whose avg changed
5371  *
5372  * This function 'ensures': tg->load_avg := \Sum tg->cfs_rq[]->avg.load.
5373  * However, because tg->load_avg is a global value there are performance
5374  * considerations.
5375  *
5376  * In order to avoid having to look at the other cfs_rq's, we use a
5377  * differential update where we store the last value we propagated. This in
5378  * turn allows skipping updates if the differential is 'small'.
5379  *
5380  * Updating tg's load_avg is necessary before update_cfs_group().
5381  */
5382 static inline void update_tg_load_avg(struct cfs_rq *cfs_rq)
5383 {
5384 	long dl, dr;
5385 	u64 now;
5386 
5387 	/*
5388 	 * No need to update load_avg for root_task_group as it is not used.
5389 	 */
5390 	if (cfs_rq->tg == &root_task_group)
5391 		return;
5392 
5393 	/* rq has been offline and doesn't contribute to the share anymore: */
5394 	if (!cpu_active(cpu_of(rq_of(cfs_rq))))
5395 		return;
5396 
5397 	/*
5398 	 * For migration heavy workloads, access to tg->load_avg can be
5399 	 * unbound. Limit the update rate to at most once per ms.
5400 	 */
5401 	now = rq_clock(rq_of(cfs_rq));
5402 	if (now - cfs_rq->last_update_tg_load_avg < NSEC_PER_MSEC)
5403 		return;
5404 
5405 	dl = cfs_rq->avg.load_avg - cfs_rq->tg_load_avg_contrib;
5406 	dr = cfs_rq->avg.runnable_avg - cfs_rq->tg_runnable_avg_contrib;
5407 	if (abs(dl) > cfs_rq->tg_load_avg_contrib / 64 ||
5408 	    abs(dr) > cfs_rq->tg_runnable_avg_contrib / 64) {
5409 		atomic_long_add(dl, &cfs_rq->tg->load_avg);
5410 		atomic_long_add(dr, &cfs_rq->tg->runnable_avg);
5411 		cfs_rq->tg_load_avg_contrib = cfs_rq->avg.load_avg;
5412 		cfs_rq->tg_runnable_avg_contrib = cfs_rq->avg.runnable_avg;
5413 		cfs_rq->last_update_tg_load_avg = now;
5414 	}
5415 }
5416 
5417 static inline void clear_tg_load_avg(struct cfs_rq *cfs_rq)
5418 {
5419 	long dl, dr;
5420 	u64 now;
5421 
5422 	/*
5423 	 * No need to update load_avg for root_task_group, as it is not used.
5424 	 */
5425 	if (cfs_rq->tg == &root_task_group)
5426 		return;
5427 
5428 	now = rq_clock(rq_of(cfs_rq));
5429 	dl = 0 - cfs_rq->tg_load_avg_contrib;
5430 	dr = 0 - cfs_rq->tg_runnable_avg_contrib;
5431 	atomic_long_add(dl, &cfs_rq->tg->load_avg);
5432 	atomic_long_add(dr, &cfs_rq->tg->runnable_avg);
5433 	cfs_rq->tg_load_avg_contrib = 0;
5434 	cfs_rq->tg_runnable_avg_contrib = 0;
5435 	cfs_rq->last_update_tg_load_avg = now;
5436 }
5437 
5438 /* CPU offline callback: */
5439 static void __maybe_unused clear_tg_offline_cfs_rqs(struct rq *rq)
5440 {
5441 	struct task_group *tg;
5442 
5443 	lockdep_assert_rq_held(rq);
5444 
5445 	/*
5446 	 * The rq clock has already been updated in
5447 	 * set_rq_offline(), so we should skip updating
5448 	 * the rq clock again in unthrottle_cfs_rq().
5449 	 */
5450 	rq_clock_start_loop_update(rq);
5451 
5452 	guard(rcu)();
5453 
5454 	list_for_each_entry_rcu(tg, &task_groups, list) {
5455 		struct cfs_rq *cfs_rq = tg_cfs_rq(tg, cpu_of(rq));
5456 
5457 		clear_tg_load_avg(cfs_rq);
5458 	}
5459 
5460 	rq_clock_stop_loop_update(rq);
5461 }
5462 
5463 /*
5464  * Called within set_task_rq() right before setting a task's CPU. The
5465  * caller only guarantees p->pi_lock is held; no other assumptions,
5466  * including the state of rq->lock, should be made.
5467  */
5468 void set_task_rq_fair(struct sched_entity *se,
5469 		      struct cfs_rq *prev, struct cfs_rq *next)
5470 {
5471 	u64 p_last_update_time;
5472 	u64 n_last_update_time;
5473 
5474 	if (!sched_feat(ATTACH_AGE_LOAD))
5475 		return;
5476 
5477 	/*
5478 	 * We are supposed to update the task to "current" time, then its up to
5479 	 * date and ready to go to new CPU/cfs_rq. But we have difficulty in
5480 	 * getting what current time is, so simply throw away the out-of-date
5481 	 * time. This will result in the wakee task is less decayed, but giving
5482 	 * the wakee more load sounds not bad.
5483 	 */
5484 	if (!(se->avg.last_update_time && prev))
5485 		return;
5486 
5487 	p_last_update_time = cfs_rq_last_update_time(prev);
5488 	n_last_update_time = cfs_rq_last_update_time(next);
5489 
5490 	__update_load_avg_blocked_se(p_last_update_time, se);
5491 	se->avg.last_update_time = n_last_update_time;
5492 }
5493 
5494 /*
5495  * When on migration a sched_entity joins/leaves the PELT hierarchy, we need to
5496  * propagate its contribution. The key to this propagation is the invariant
5497  * that for each group:
5498  *
5499  *   ge->avg == grq->avg						(1)
5500  *
5501  * _IFF_ we look at the pure running and runnable sums. Because they
5502  * represent the very same entity, just at different points in the hierarchy.
5503  *
5504  * Per the above update_tg_cfs_util() and update_tg_cfs_runnable() are trivial
5505  * and simply copies the running/runnable sum over (but still wrong, because
5506  * the group entity and group rq do not have their PELT windows aligned).
5507  *
5508  * However, update_tg_cfs_load() is more complex. So we have:
5509  *
5510  *   ge->avg.load_avg = ge->load.weight * ge->avg.runnable_avg		(2)
5511  *
5512  * And since, like util, the runnable part should be directly transferable,
5513  * the following would _appear_ to be the straight forward approach:
5514  *
5515  *   grq->avg.load_avg = grq->load.weight * grq->avg.runnable_avg	(3)
5516  *
5517  * And per (1) we have:
5518  *
5519  *   ge->avg.runnable_avg == grq->avg.runnable_avg
5520  *
5521  * Which gives:
5522  *
5523  *                      ge->load.weight * grq->avg.load_avg
5524  *   ge->avg.load_avg = -----------------------------------		(4)
5525  *                               grq->load.weight
5526  *
5527  * Except that is wrong!
5528  *
5529  * Because while for entities historical weight is not important and we
5530  * really only care about our future and therefore can consider a pure
5531  * runnable sum, runqueues can NOT do this.
5532  *
5533  * We specifically want runqueues to have a load_avg that includes
5534  * historical weights. Those represent the blocked load, the load we expect
5535  * to (shortly) return to us. This only works by keeping the weights as
5536  * integral part of the sum. We therefore cannot decompose as per (3).
5537  *
5538  * Another reason this doesn't work is that runnable isn't a 0-sum entity.
5539  * Imagine a rq with 2 tasks that each are runnable 2/3 of the time. Then the
5540  * rq itself is runnable anywhere between 2/3 and 1 depending on how the
5541  * runnable section of these tasks overlap (or not). If they were to perfectly
5542  * align the rq as a whole would be runnable 2/3 of the time. If however we
5543  * always have at least 1 runnable task, the rq as a whole is always runnable.
5544  *
5545  * So we'll have to approximate.. :/
5546  *
5547  * Given the constraint:
5548  *
5549  *   ge->avg.running_sum <= ge->avg.runnable_sum <= LOAD_AVG_MAX
5550  *
5551  * We can construct a rule that adds runnable to a rq by assuming minimal
5552  * overlap.
5553  *
5554  * On removal, we'll assume each task is equally runnable; which yields:
5555  *
5556  *   grq->avg.runnable_sum = grq->avg.load_sum / grq->load.weight
5557  *
5558  * XXX: only do this for the part of runnable > running ?
5559  *
5560  */
5561 static inline void
5562 update_tg_cfs_util(struct cfs_rq *cfs_rq, struct sched_entity *se, struct cfs_rq *gcfs_rq)
5563 {
5564 	long delta_sum, delta_avg = gcfs_rq->avg.util_avg - se->avg.util_avg;
5565 	u32 new_sum, divider;
5566 
5567 	/* Nothing to update */
5568 	if (!delta_avg)
5569 		return;
5570 
5571 	/*
5572 	 * cfs_rq->avg.period_contrib can be used for both cfs_rq and se.
5573 	 * See ___update_load_avg() for details.
5574 	 */
5575 	divider = get_pelt_divider(&cfs_rq->avg);
5576 
5577 	/* Set new sched_entity's utilization */
5578 	se->avg.util_avg = gcfs_rq->avg.util_avg;
5579 	new_sum = se->avg.util_avg * divider;
5580 	delta_sum = (long)new_sum - (long)se->avg.util_sum;
5581 	se->avg.util_sum = new_sum;
5582 
5583 	/* Update parent cfs_rq utilization */
5584 	__update_sa(&cfs_rq->avg, util, delta_avg, delta_sum);
5585 }
5586 
5587 static inline void
5588 update_tg_cfs_runnable(struct cfs_rq *cfs_rq, struct sched_entity *se, struct cfs_rq *gcfs_rq)
5589 {
5590 	long delta_sum, delta_avg = gcfs_rq->avg.runnable_avg - se->avg.runnable_avg;
5591 	u64 new_sum;
5592 	u32 divider;
5593 
5594 	/* Nothing to update */
5595 	if (!delta_avg)
5596 		return;
5597 
5598 	/*
5599 	 * cfs_rq->avg.period_contrib can be used for both cfs_rq and se.
5600 	 * See ___update_load_avg() for details.
5601 	 */
5602 	divider = get_pelt_divider(&cfs_rq->avg);
5603 
5604 	/* Set new sched_entity's runnable */
5605 	se->avg.runnable_avg = gcfs_rq->avg.runnable_avg;
5606 	new_sum = (u64)se->avg.runnable_avg * divider;
5607 	delta_sum = (long)new_sum - (long)se->avg.runnable_sum;
5608 	se->avg.runnable_sum = new_sum;
5609 
5610 	/* Update parent cfs_rq runnable */
5611 	__update_sa(&cfs_rq->avg, runnable, delta_avg, delta_sum);
5612 }
5613 
5614 static inline void
5615 update_tg_cfs_load(struct cfs_rq *cfs_rq, struct sched_entity *se, struct cfs_rq *gcfs_rq)
5616 {
5617 	long delta_avg, running_sum, runnable_sum = gcfs_rq->prop_runnable_sum;
5618 	unsigned long load_avg;
5619 	u64 load_sum = 0;
5620 	s64 delta_sum;
5621 	u32 divider;
5622 
5623 	if (!runnable_sum)
5624 		return;
5625 
5626 	gcfs_rq->prop_runnable_sum = 0;
5627 
5628 	/*
5629 	 * cfs_rq->avg.period_contrib can be used for both cfs_rq and se.
5630 	 * See ___update_load_avg() for details.
5631 	 */
5632 	divider = get_pelt_divider(&cfs_rq->avg);
5633 
5634 	if (runnable_sum >= 0) {
5635 		/*
5636 		 * Add runnable; clip at LOAD_AVG_MAX. Reflects that until
5637 		 * the CPU is saturated running == runnable.
5638 		 */
5639 		runnable_sum += se->avg.load_sum;
5640 		runnable_sum = min_t(long, runnable_sum, divider);
5641 	} else {
5642 		/*
5643 		 * Estimate the new unweighted runnable_sum of the gcfs_rq by
5644 		 * assuming all tasks are equally runnable.
5645 		 */
5646 		if (scale_load_down(gcfs_rq->load.weight)) {
5647 			load_sum = div_u64(gcfs_rq->avg.load_sum,
5648 				scale_load_down(gcfs_rq->load.weight));
5649 		}
5650 
5651 		/* But make sure to not inflate se's runnable */
5652 		runnable_sum = min(se->avg.load_sum, load_sum);
5653 	}
5654 
5655 	/*
5656 	 * runnable_sum can't be lower than running_sum
5657 	 * Rescale running sum to be in the same range as runnable sum
5658 	 * running_sum is in [0 : LOAD_AVG_MAX <<  SCHED_CAPACITY_SHIFT]
5659 	 * runnable_sum is in [0 : LOAD_AVG_MAX]
5660 	 */
5661 	running_sum = se->avg.util_sum >> SCHED_CAPACITY_SHIFT;
5662 	runnable_sum = max(runnable_sum, running_sum);
5663 
5664 	load_sum = se_weight(se) * runnable_sum;
5665 	load_avg = div_u64(load_sum, divider);
5666 
5667 	delta_avg = load_avg - se->avg.load_avg;
5668 	if (!delta_avg)
5669 		return;
5670 
5671 	delta_sum = load_sum - (s64)se_weight(se) * se->avg.load_sum;
5672 
5673 	se->avg.load_sum = runnable_sum;
5674 	se->avg.load_avg = load_avg;
5675 	__update_sa(&cfs_rq->avg, load, delta_avg, delta_sum);
5676 }
5677 
5678 static inline void add_tg_cfs_propagate(struct cfs_rq *cfs_rq, long runnable_sum)
5679 {
5680 	cfs_rq->propagate = 1;
5681 	cfs_rq->prop_runnable_sum += runnable_sum;
5682 }
5683 
5684 /* Update task and its cfs_rq load average */
5685 static inline int propagate_entity_load_avg(struct sched_entity *se)
5686 {
5687 	struct cfs_rq *cfs_rq, *gcfs_rq;
5688 
5689 	if (entity_is_task(se))
5690 		return 0;
5691 
5692 	gcfs_rq = group_cfs_rq(se);
5693 	if (!gcfs_rq->propagate)
5694 		return 0;
5695 
5696 	gcfs_rq->propagate = 0;
5697 
5698 	cfs_rq = cfs_rq_of(se);
5699 
5700 	add_tg_cfs_propagate(cfs_rq, gcfs_rq->prop_runnable_sum);
5701 
5702 	update_tg_cfs_util(cfs_rq, se, gcfs_rq);
5703 	update_tg_cfs_runnable(cfs_rq, se, gcfs_rq);
5704 	update_tg_cfs_load(cfs_rq, se, gcfs_rq);
5705 
5706 	trace_pelt_cfs_tp(cfs_rq);
5707 	trace_pelt_se_tp(se);
5708 
5709 	return 1;
5710 }
5711 
5712 /*
5713  * Check if we need to update the load and the utilization of a blocked
5714  * group_entity:
5715  */
5716 static inline bool skip_blocked_update(struct sched_entity *se)
5717 {
5718 	struct cfs_rq *gcfs_rq = group_cfs_rq(se);
5719 
5720 	/*
5721 	 * If sched_entity still have not zero load or utilization, we have to
5722 	 * decay it:
5723 	 */
5724 	if (se->avg.load_avg || se->avg.util_avg)
5725 		return false;
5726 
5727 	/*
5728 	 * If there is a pending propagation, we have to update the load and
5729 	 * the utilization of the sched_entity:
5730 	 */
5731 	if (gcfs_rq->propagate)
5732 		return false;
5733 
5734 	/*
5735 	 * Otherwise, the load and the utilization of the sched_entity is
5736 	 * already zero and there is no pending propagation, so it will be a
5737 	 * waste of time to try to decay it:
5738 	 */
5739 	return true;
5740 }
5741 
5742 #else /* !CONFIG_FAIR_GROUP_SCHED: */
5743 
5744 static inline void update_tg_load_avg(struct cfs_rq *cfs_rq) {}
5745 
5746 static inline void clear_tg_offline_cfs_rqs(struct rq *rq) {}
5747 
5748 static inline int propagate_entity_load_avg(struct sched_entity *se)
5749 {
5750 	return 0;
5751 }
5752 
5753 static inline void add_tg_cfs_propagate(struct cfs_rq *cfs_rq, long runnable_sum) {}
5754 
5755 #endif /* !CONFIG_FAIR_GROUP_SCHED */
5756 
5757 #ifdef CONFIG_NO_HZ_COMMON
5758 static inline void migrate_se_pelt_lag(struct sched_entity *se)
5759 {
5760 	u64 throttled = 0, now, lut;
5761 	struct cfs_rq *cfs_rq;
5762 	struct rq *rq;
5763 	bool is_idle;
5764 
5765 	if (load_avg_is_decayed(&se->avg))
5766 		return;
5767 
5768 	cfs_rq = cfs_rq_of(se);
5769 	rq = rq_of(cfs_rq);
5770 
5771 	rcu_read_lock();
5772 	is_idle = is_idle_task(rcu_dereference_all(rq->curr));
5773 	rcu_read_unlock();
5774 
5775 	/*
5776 	 * The lag estimation comes with a cost we don't want to pay all the
5777 	 * time. Hence, limiting to the case where the source CPU is idle and
5778 	 * we know we are at the greatest risk to have an outdated clock.
5779 	 */
5780 	if (!is_idle)
5781 		return;
5782 
5783 	/*
5784 	 * Estimated "now" is: last_update_time + cfs_idle_lag + rq_idle_lag, where:
5785 	 *
5786 	 *   last_update_time (the cfs_rq's last_update_time)
5787 	 *	= cfs_rq_clock_pelt()@cfs_rq_idle
5788 	 *      = rq_clock_pelt()@cfs_rq_idle
5789 	 *        - cfs->throttled_clock_pelt_time@cfs_rq_idle
5790 	 *
5791 	 *   cfs_idle_lag (delta between rq's update and cfs_rq's update)
5792 	 *      = rq_clock_pelt()@rq_idle - rq_clock_pelt()@cfs_rq_idle
5793 	 *
5794 	 *   rq_idle_lag (delta between now and rq's update)
5795 	 *      = sched_clock_cpu() - rq_clock()@rq_idle
5796 	 *
5797 	 * We can then write:
5798 	 *
5799 	 *    now = rq_clock_pelt()@rq_idle - cfs->throttled_clock_pelt_time +
5800 	 *          sched_clock_cpu() - rq_clock()@rq_idle
5801 	 * Where:
5802 	 *      rq_clock_pelt()@rq_idle is rq->clock_pelt_idle
5803 	 *      rq_clock()@rq_idle      is rq->clock_idle
5804 	 *      cfs->throttled_clock_pelt_time@cfs_rq_idle
5805 	 *                              is cfs_rq->throttled_pelt_idle
5806 	 */
5807 
5808 #ifdef CONFIG_CFS_BANDWIDTH
5809 	throttled = u64_u32_load(cfs_rq->throttled_pelt_idle);
5810 	/* The clock has been stopped for throttling */
5811 	if (throttled == U64_MAX)
5812 		return;
5813 #endif
5814 	now = u64_u32_load(rq->clock_pelt_idle);
5815 	/*
5816 	 * Paired with _update_idle_rq_clock_pelt(). It ensures at the worst case
5817 	 * is observed the old clock_pelt_idle value and the new clock_idle,
5818 	 * which lead to an underestimation. The opposite would lead to an
5819 	 * overestimation.
5820 	 */
5821 	smp_rmb();
5822 	lut = cfs_rq_last_update_time(cfs_rq);
5823 
5824 	now -= throttled;
5825 	if (now < lut)
5826 		/*
5827 		 * cfs_rq->avg.last_update_time is more recent than our
5828 		 * estimation, let's use it.
5829 		 */
5830 		now = lut;
5831 	else
5832 		now += sched_clock_cpu(cpu_of(rq)) - u64_u32_load(rq->clock_idle);
5833 
5834 	__update_load_avg_blocked_se(now, se);
5835 }
5836 #else /* !CONFIG_NO_HZ_COMMON: */
5837 static void migrate_se_pelt_lag(struct sched_entity *se) {}
5838 #endif /* !CONFIG_NO_HZ_COMMON */
5839 
5840 /**
5841  * update_cfs_rq_load_avg - update the cfs_rq's load/util averages
5842  * @now: current time, as per cfs_rq_clock_pelt()
5843  * @cfs_rq: cfs_rq to update
5844  *
5845  * The cfs_rq avg is the direct sum of all its entities (blocked and runnable)
5846  * avg. The immediate corollary is that all (fair) tasks must be attached.
5847  *
5848  * cfs_rq->avg is used for task_h_load() and update_cfs_group() for example.
5849  *
5850  * Return: true if the load decayed or we removed load.
5851  *
5852  * Since both these conditions indicate a changed cfs_rq->avg.load we should
5853  * call update_tg_load_avg() when this function returns true.
5854  */
5855 static inline int
5856 update_cfs_rq_load_avg(u64 now, struct cfs_rq *cfs_rq)
5857 {
5858 	unsigned long removed_load = 0, removed_util = 0, removed_runnable = 0;
5859 	struct sched_avg *sa = &cfs_rq->avg;
5860 	int decayed = 0;
5861 
5862 	if (cfs_rq->removed.nr) {
5863 		unsigned long r;
5864 		u32 divider = get_pelt_divider(&cfs_rq->avg);
5865 
5866 		raw_spin_lock(&cfs_rq->removed.lock);
5867 		swap(cfs_rq->removed.util_avg, removed_util);
5868 		swap(cfs_rq->removed.load_avg, removed_load);
5869 		swap(cfs_rq->removed.runnable_avg, removed_runnable);
5870 		cfs_rq->removed.nr = 0;
5871 		raw_spin_unlock(&cfs_rq->removed.lock);
5872 
5873 		r = removed_load;
5874 		__update_sa(sa, load, -r, -r*divider);
5875 
5876 		r = removed_util;
5877 		__update_sa(sa, util, -r, -r*divider);
5878 
5879 		r = removed_runnable;
5880 		__update_sa(sa, runnable, -r, -r*divider);
5881 
5882 		/*
5883 		 * removed_runnable is the unweighted version of removed_load so we
5884 		 * can use it to estimate removed_load_sum.
5885 		 */
5886 		add_tg_cfs_propagate(cfs_rq,
5887 			-(long)(removed_runnable * divider) >> SCHED_CAPACITY_SHIFT);
5888 
5889 		decayed = 1;
5890 	}
5891 
5892 	decayed |= __update_load_avg_cfs_rq(now, cfs_rq);
5893 	u64_u32_store_copy(sa->last_update_time,
5894 			   cfs_rq->last_update_time_copy,
5895 			   sa->last_update_time);
5896 	return decayed;
5897 }
5898 
5899 /**
5900  * attach_entity_load_avg - attach this entity to its cfs_rq load avg
5901  * @cfs_rq: cfs_rq to attach to
5902  * @se: sched_entity to attach
5903  *
5904  * Must call update_cfs_rq_load_avg() before this, since we rely on
5905  * cfs_rq->avg.last_update_time being current.
5906  */
5907 static void attach_entity_load_avg(struct cfs_rq *cfs_rq, struct sched_entity *se)
5908 {
5909 	/*
5910 	 * cfs_rq->avg.period_contrib can be used for both cfs_rq and se.
5911 	 * See ___update_load_avg() for details.
5912 	 */
5913 	u32 divider = get_pelt_divider(&cfs_rq->avg);
5914 
5915 	/*
5916 	 * When we attach the @se to the @cfs_rq, we must align the decay
5917 	 * window because without that, really weird and wonderful things can
5918 	 * happen.
5919 	 *
5920 	 * XXX illustrate
5921 	 */
5922 	se->avg.last_update_time = cfs_rq->avg.last_update_time;
5923 	se->avg.period_contrib = cfs_rq->avg.period_contrib;
5924 
5925 	/*
5926 	 * Hell(o) Nasty stuff.. we need to recompute _sum based on the new
5927 	 * period_contrib. This isn't strictly correct, but since we're
5928 	 * entirely outside of the PELT hierarchy, nobody cares if we truncate
5929 	 * _sum a little.
5930 	 */
5931 	se->avg.util_sum = se->avg.util_avg * divider;
5932 
5933 	se->avg.runnable_sum = se->avg.runnable_avg * divider;
5934 
5935 	se->avg.load_sum = se->avg.load_avg * divider;
5936 	if (se_weight(se) < se->avg.load_sum)
5937 		se->avg.load_sum = div_u64(se->avg.load_sum, se_weight(se));
5938 	else
5939 		se->avg.load_sum = 1;
5940 
5941 	enqueue_load_avg(cfs_rq, se);
5942 	cfs_rq->avg.util_avg += se->avg.util_avg;
5943 	cfs_rq->avg.util_sum += se->avg.util_sum;
5944 	cfs_rq->avg.runnable_avg += se->avg.runnable_avg;
5945 	cfs_rq->avg.runnable_sum += se->avg.runnable_sum;
5946 
5947 	add_tg_cfs_propagate(cfs_rq, se->avg.load_sum);
5948 
5949 	cfs_rq_util_change(cfs_rq, 0);
5950 
5951 	trace_pelt_cfs_tp(cfs_rq);
5952 }
5953 
5954 /**
5955  * detach_entity_load_avg - detach this entity from its cfs_rq load avg
5956  * @cfs_rq: cfs_rq to detach from
5957  * @se: sched_entity to detach
5958  *
5959  * Must call update_cfs_rq_load_avg() before this, since we rely on
5960  * cfs_rq->avg.last_update_time being current.
5961  */
5962 static void detach_entity_load_avg(struct cfs_rq *cfs_rq, struct sched_entity *se)
5963 {
5964 	dequeue_load_avg(cfs_rq, se);
5965 	__update_sa(&cfs_rq->avg, util, -se->avg.util_avg, -se->avg.util_sum);
5966 	__update_sa(&cfs_rq->avg, runnable, -se->avg.runnable_avg, -se->avg.runnable_sum);
5967 
5968 	add_tg_cfs_propagate(cfs_rq, -se->avg.load_sum);
5969 
5970 	cfs_rq_util_change(cfs_rq, 0);
5971 
5972 	trace_pelt_cfs_tp(cfs_rq);
5973 }
5974 
5975 #define UTIL_EST_MARGIN (SCHED_CAPACITY_SCALE / 100)
5976 
5977 static inline void util_est_update(struct sched_entity *se)
5978 {
5979 	unsigned int ewma, dequeued, last_ewma_diff;
5980 
5981 	if (!sched_feat(UTIL_EST))
5982 		return;
5983 
5984 	/* Get current estimate of utilization */
5985 	ewma = READ_ONCE(se->avg.util_est);
5986 
5987 	/*
5988 	 * If the PELT values haven't changed since enqueue time,
5989 	 * skip the util_est update.
5990 	 */
5991 	if (ewma & UTIL_AVG_UNCHANGED)
5992 		return;
5993 
5994 	/* Get utilization at dequeue */
5995 	dequeued = READ_ONCE(se->avg.util_avg);
5996 
5997 	/*
5998 	 * Reset EWMA on utilization increases, the moving average is used only
5999 	 * to smooth utilization decreases.
6000 	 */
6001 	if (ewma <= dequeued) {
6002 		ewma = dequeued;
6003 		goto done;
6004 	}
6005 
6006 	/*
6007 	 * Skip update of task's estimated utilization when its members are
6008 	 * already ~1% close to its last activation value.
6009 	 */
6010 	last_ewma_diff = ewma - dequeued;
6011 	if (last_ewma_diff < UTIL_EST_MARGIN)
6012 		goto done;
6013 
6014 	/*
6015 	 * To avoid underestimate of task utilization, skip updates of EWMA if
6016 	 * we cannot grant that thread got all CPU time it wanted.
6017 	 */
6018 	if ((dequeued + UTIL_EST_MARGIN) < READ_ONCE(se->avg.runnable_avg))
6019 		goto done;
6020 
6021 	/*
6022 	 * Update Task's estimated utilization
6023 	 *
6024 	 * When *p completes an activation we can consolidate another sample
6025 	 * of the task size. This is done by using this value to update the
6026 	 * Exponential Weighted Moving Average (EWMA):
6027 	 *
6028 	 *  ewma(t) = w *  task_util(p) + (1-w) * ewma(t-1)
6029 	 *          = w *  task_util(p) +         ewma(t-1)  - w * ewma(t-1)
6030 	 *          = w * (task_util(p) -         ewma(t-1)) +     ewma(t-1)
6031 	 *          = w * (      -last_ewma_diff           ) +     ewma(t-1)
6032 	 *          = w * (-last_ewma_diff +  ewma(t-1) / w)
6033 	 *
6034 	 * Where 'w' is the weight of new samples, which is configured to be
6035 	 * 0.25, thus making w=1/4 ( >>= UTIL_EST_WEIGHT_SHIFT)
6036 	 */
6037 	ewma <<= UTIL_EST_WEIGHT_SHIFT;
6038 	ewma  -= last_ewma_diff;
6039 	ewma >>= UTIL_EST_WEIGHT_SHIFT;
6040 done:
6041 	ewma |= UTIL_AVG_UNCHANGED;
6042 	WRITE_ONCE(se->avg.util_est, ewma);
6043 
6044 	trace_sched_util_est_se_tp(se);
6045 }
6046 
6047 /*
6048  * Optional action to be done while updating the load average
6049  */
6050 #define UPDATE_TG	0x01
6051 #define SKIP_AGE_LOAD	0x02
6052 #define DO_ATTACH	0x04
6053 #define DO_DETACH	0x08
6054 #define UPDATE_UTIL_EST	0x10
6055 
6056 /* Update task and its cfs_rq load average */
6057 static inline void update_load_avg(struct cfs_rq *cfs_rq, struct sched_entity *se, int flags)
6058 {
6059 	u64 now = cfs_rq_clock_pelt(cfs_rq);
6060 	int decayed;
6061 
6062 	/*
6063 	 * Track task load average for carrying it to new CPU after migrated, and
6064 	 * track group sched_entity load average for task_h_load calculation in migration
6065 	 */
6066 	if (se->avg.last_update_time && !(flags & SKIP_AGE_LOAD))
6067 		__update_load_avg_se(now, cfs_rq, se);
6068 
6069 	decayed  = update_cfs_rq_load_avg(now, cfs_rq);
6070 	decayed |= propagate_entity_load_avg(se);
6071 
6072 	if (!se->avg.last_update_time && (flags & DO_ATTACH)) {
6073 
6074 		/*
6075 		 * DO_ATTACH means we're here from enqueue_entity().
6076 		 * !last_update_time means we've passed through
6077 		 * migrate_task_rq_fair() indicating we migrated.
6078 		 *
6079 		 * IOW we're enqueueing a task on a new CPU.
6080 		 */
6081 		attach_entity_load_avg(cfs_rq, se);
6082 		update_tg_load_avg(cfs_rq);
6083 
6084 	} else if (flags & DO_DETACH) {
6085 		/*
6086 		 * DO_DETACH means we're here from dequeue_entity()
6087 		 * and we are migrating task out of the CPU.
6088 		 */
6089 		detach_entity_load_avg(cfs_rq, se);
6090 		update_tg_load_avg(cfs_rq);
6091 	} else if (decayed) {
6092 		cfs_rq_util_change(cfs_rq, 0);
6093 
6094 		if (flags & UPDATE_TG)
6095 			update_tg_load_avg(cfs_rq);
6096 	}
6097 
6098 	if (flags & UPDATE_UTIL_EST)
6099 		util_est_update(se);
6100 }
6101 
6102 /*
6103  * Synchronize entity load avg of dequeued entity without locking
6104  * the previous rq.
6105  */
6106 static void sync_entity_load_avg(struct sched_entity *se)
6107 {
6108 	struct cfs_rq *cfs_rq = cfs_rq_of(se);
6109 	u64 last_update_time;
6110 
6111 	last_update_time = cfs_rq_last_update_time(cfs_rq);
6112 	__update_load_avg_blocked_se(last_update_time, se);
6113 }
6114 
6115 /*
6116  * Task first catches up with cfs_rq, and then subtract
6117  * itself from the cfs_rq (task must be off the queue now).
6118  */
6119 static void remove_entity_load_avg(struct sched_entity *se)
6120 {
6121 	struct cfs_rq *cfs_rq = cfs_rq_of(se);
6122 	unsigned long flags;
6123 
6124 	/*
6125 	 * tasks cannot exit without having gone through wake_up_new_task() ->
6126 	 * enqueue_task_fair() which will have added things to the cfs_rq,
6127 	 * so we can remove unconditionally.
6128 	 */
6129 
6130 	sync_entity_load_avg(se);
6131 
6132 	raw_spin_lock_irqsave(&cfs_rq->removed.lock, flags);
6133 	++cfs_rq->removed.nr;
6134 	cfs_rq->removed.util_avg	+= se->avg.util_avg;
6135 	cfs_rq->removed.load_avg	+= se->avg.load_avg;
6136 	cfs_rq->removed.runnable_avg	+= se->avg.runnable_avg;
6137 	raw_spin_unlock_irqrestore(&cfs_rq->removed.lock, flags);
6138 }
6139 
6140 static inline unsigned long cfs_rq_runnable_avg(struct cfs_rq *cfs_rq)
6141 {
6142 	return cfs_rq->avg.runnable_avg;
6143 }
6144 
6145 static inline unsigned long cfs_rq_load_avg(struct cfs_rq *cfs_rq)
6146 {
6147 	return cfs_rq->avg.load_avg;
6148 }
6149 
6150 static int sched_balance_newidle(struct rq *this_rq, struct rq_flags *rf)
6151 	__must_hold(__rq_lockp(this_rq));
6152 
6153 static inline unsigned long task_util(struct task_struct *p)
6154 {
6155 	return READ_ONCE(p->se.avg.util_avg);
6156 }
6157 
6158 static inline unsigned long _task_util_est(struct task_struct *p)
6159 {
6160 	return READ_ONCE(p->se.avg.util_est) & ~UTIL_AVG_UNCHANGED;
6161 }
6162 
6163 static inline unsigned long task_util_est(struct task_struct *p)
6164 {
6165 	return max(task_util(p), _task_util_est(p));
6166 }
6167 
6168 static inline void util_est_enqueue(struct cfs_rq *cfs_rq,
6169 				    struct task_struct *p)
6170 {
6171 	unsigned int enqueued;
6172 
6173 	if (!sched_feat(UTIL_EST))
6174 		return;
6175 
6176 	/* Update root cfs_rq's estimated utilization */
6177 	enqueued  = cfs_rq->avg.util_est;
6178 	enqueued += _task_util_est(p);
6179 	WRITE_ONCE(cfs_rq->avg.util_est, enqueued);
6180 
6181 	trace_sched_util_est_cfs_tp(cfs_rq);
6182 }
6183 
6184 static inline void util_est_dequeue(struct cfs_rq *cfs_rq,
6185 				    struct task_struct *p)
6186 {
6187 	unsigned int enqueued;
6188 
6189 	if (!sched_feat(UTIL_EST))
6190 		return;
6191 
6192 	/* Update root cfs_rq's estimated utilization */
6193 	enqueued  = cfs_rq->avg.util_est;
6194 	enqueued -= min_t(unsigned int, enqueued, _task_util_est(p));
6195 	WRITE_ONCE(cfs_rq->avg.util_est, enqueued);
6196 
6197 	trace_sched_util_est_cfs_tp(cfs_rq);
6198 }
6199 
6200 static inline unsigned long get_actual_cpu_capacity(int cpu)
6201 {
6202 	unsigned long capacity = arch_scale_cpu_capacity(cpu);
6203 
6204 	capacity -= max(hw_load_avg(cpu_rq(cpu)), cpufreq_get_pressure(cpu));
6205 
6206 	return capacity;
6207 }
6208 
6209 static inline int util_fits_cpu(unsigned long util,
6210 				unsigned long uclamp_min,
6211 				unsigned long uclamp_max,
6212 				int cpu)
6213 {
6214 	unsigned long capacity = capacity_of(cpu);
6215 	unsigned long capacity_orig;
6216 	bool fits, uclamp_max_fits;
6217 
6218 	/*
6219 	 * Check if the real util fits without any uclamp boost/cap applied.
6220 	 */
6221 	fits = fits_capacity(util, capacity);
6222 
6223 	if (!uclamp_is_used())
6224 		return fits;
6225 
6226 	/*
6227 	 * We must use arch_scale_cpu_capacity() for comparing against uclamp_min and
6228 	 * uclamp_max. We only care about capacity pressure (by using
6229 	 * capacity_of()) for comparing against the real util.
6230 	 *
6231 	 * If a task is boosted to 1024 for example, we don't want a tiny
6232 	 * pressure to skew the check whether it fits a CPU or not.
6233 	 *
6234 	 * Similarly if a task is capped to arch_scale_cpu_capacity(little_cpu), it
6235 	 * should fit a little cpu even if there's some pressure.
6236 	 *
6237 	 * Only exception is for HW or cpufreq pressure since it has a direct impact
6238 	 * on available OPP of the system.
6239 	 *
6240 	 * We honour it for uclamp_min only as a drop in performance level
6241 	 * could result in not getting the requested minimum performance level.
6242 	 *
6243 	 * For uclamp_max, we can tolerate a drop in performance level as the
6244 	 * goal is to cap the task. So it's okay if it's getting less.
6245 	 */
6246 	capacity_orig = arch_scale_cpu_capacity(cpu);
6247 
6248 	/*
6249 	 * We want to force a task to fit a cpu as implied by uclamp_max.
6250 	 * But we do have some corner cases to cater for..
6251 	 *
6252 	 *
6253 	 *                                 C=z
6254 	 *   |                             ___
6255 	 *   |                  C=y       |   |
6256 	 *   |_ _ _ _ _ _ _ _ _ ___ _ _ _ | _ | _ _ _ _ _  uclamp_max
6257 	 *   |      C=x        |   |      |   |
6258 	 *   |      ___        |   |      |   |
6259 	 *   |     |   |       |   |      |   |    (util somewhere in this region)
6260 	 *   |     |   |       |   |      |   |
6261 	 *   |     |   |       |   |      |   |
6262 	 *   +----------------------------------------
6263 	 *         CPU0        CPU1       CPU2
6264 	 *
6265 	 *   In the above example if a task is capped to a specific performance
6266 	 *   point, y, then when:
6267 	 *
6268 	 *   * util = 80% of x then it does not fit on CPU0 and should migrate
6269 	 *     to CPU1
6270 	 *   * util = 80% of y then it is forced to fit on CPU1 to honour
6271 	 *     uclamp_max request.
6272 	 *
6273 	 *   which is what we're enforcing here. A task always fits if
6274 	 *   uclamp_max <= capacity_orig. But when uclamp_max > capacity_orig,
6275 	 *   the normal upmigration rules should withhold still.
6276 	 *
6277 	 *   Only exception is when we are on max capacity, then we need to be
6278 	 *   careful not to block overutilized state. This is so because:
6279 	 *
6280 	 *     1. There's no concept of capping at max_capacity! We can't go
6281 	 *        beyond this performance level anyway.
6282 	 *     2. The system is being saturated when we're operating near
6283 	 *        max capacity, it doesn't make sense to block overutilized.
6284 	 */
6285 	uclamp_max_fits = (capacity_orig == SCHED_CAPACITY_SCALE) && (uclamp_max == SCHED_CAPACITY_SCALE);
6286 	uclamp_max_fits = !uclamp_max_fits && (uclamp_max <= capacity_orig);
6287 	fits = fits || uclamp_max_fits;
6288 
6289 	/*
6290 	 *
6291 	 *                                 C=z
6292 	 *   |                             ___       (region a, capped, util >= uclamp_max)
6293 	 *   |                  C=y       |   |
6294 	 *   |_ _ _ _ _ _ _ _ _ ___ _ _ _ | _ | _ _ _ _ _ uclamp_max
6295 	 *   |      C=x        |   |      |   |
6296 	 *   |      ___        |   |      |   |      (region b, uclamp_min <= util <= uclamp_max)
6297 	 *   |_ _ _|_ _|_ _ _ _| _ | _ _ _| _ | _ _ _ _ _ uclamp_min
6298 	 *   |     |   |       |   |      |   |
6299 	 *   |     |   |       |   |      |   |      (region c, boosted, util < uclamp_min)
6300 	 *   +----------------------------------------
6301 	 *         CPU0        CPU1       CPU2
6302 	 *
6303 	 * a) If util > uclamp_max, then we're capped, we don't care about
6304 	 *    actual fitness value here. We only care if uclamp_max fits
6305 	 *    capacity without taking margin/pressure into account.
6306 	 *    See comment above.
6307 	 *
6308 	 * b) If uclamp_min <= util <= uclamp_max, then the normal
6309 	 *    fits_capacity() rules apply. Except we need to ensure that we
6310 	 *    enforce we remain within uclamp_max, see comment above.
6311 	 *
6312 	 * c) If util < uclamp_min, then we are boosted. Same as (b) but we
6313 	 *    need to take into account the boosted value fits the CPU without
6314 	 *    taking margin/pressure into account.
6315 	 *
6316 	 * Cases (a) and (b) are handled in the 'fits' variable already. We
6317 	 * just need to consider an extra check for case (c) after ensuring we
6318 	 * handle the case uclamp_min > uclamp_max.
6319 	 */
6320 	uclamp_min = min(uclamp_min, uclamp_max);
6321 	if (fits && (util < uclamp_min) &&
6322 	    (uclamp_min > get_actual_cpu_capacity(cpu)))
6323 		return -1;
6324 
6325 	return fits;
6326 }
6327 
6328 static inline int task_fits_cpu(struct task_struct *p, int cpu)
6329 {
6330 	unsigned long uclamp_min = uclamp_eff_value(p, UCLAMP_MIN);
6331 	unsigned long uclamp_max = uclamp_eff_value(p, UCLAMP_MAX);
6332 	unsigned long util = task_util_est(p);
6333 	/*
6334 	 * Return true only if the cpu fully fits the task requirements, which
6335 	 * include the utilization but also the performance hints.
6336 	 */
6337 	return (util_fits_cpu(util, uclamp_min, uclamp_max, cpu) > 0);
6338 }
6339 
6340 static inline void update_misfit_status(struct task_struct *p, struct rq *rq)
6341 {
6342 	int cpu = cpu_of(rq);
6343 
6344 	if (!sched_asym_cpucap_active())
6345 		return;
6346 
6347 	/*
6348 	 * Affinity allows us to go somewhere higher?  Or are we on biggest
6349 	 * available CPU already? Or do we fit into this CPU ?
6350 	 */
6351 	if (!p || (p->nr_cpus_allowed == 1) ||
6352 	    (arch_scale_cpu_capacity(cpu) == p->max_allowed_capacity) ||
6353 	    task_fits_cpu(p, cpu)) {
6354 
6355 		rq->misfit_task_load = 0;
6356 		return;
6357 	}
6358 
6359 	/*
6360 	 * Make sure that misfit_task_load will not be null even if
6361 	 * task_h_load() returns 0.
6362 	 */
6363 	rq->misfit_task_load = max_t(unsigned long, task_h_load(p), 1);
6364 }
6365 
6366 void __setparam_fair(struct task_struct *p, const struct sched_attr *attr)
6367 {
6368 	struct sched_entity *se = &p->se;
6369 
6370 	p->static_prio = NICE_TO_PRIO(attr->sched_nice);
6371 	if (attr->sched_runtime) {
6372 		se->custom_slice = 1;
6373 		se->slice = clamp_t(u64, attr->sched_runtime,
6374 				      NSEC_PER_MSEC/10,   /* HZ=1000 * 10 */
6375 				      NSEC_PER_MSEC*100); /* HZ=100  / 10 */
6376 	} else {
6377 		se->custom_slice = 0;
6378 		se->slice = sysctl_sched_base_slice;
6379 	}
6380 }
6381 
6382 static void
6383 place_entity(struct cfs_rq *cfs_rq, struct sched_entity *se, int flags)
6384 {
6385 	u64 vslice, vruntime = avg_vruntime(cfs_rq);
6386 	unsigned int nr_queued = cfs_rq->h_nr_queued;
6387 	bool update_zero = false;
6388 	s64 lag = 0;
6389 
6390 	if (!se->custom_slice)
6391 		se->slice = sysctl_sched_base_slice;
6392 	vslice = calc_delta_fair(se->slice, se);
6393 
6394 	if (flags & ENQUEUE_QUEUED)
6395 		nr_queued -= 1;
6396 
6397 	/*
6398 	 * Due to how V is constructed as the weighted average of entities,
6399 	 * adding tasks with positive lag, or removing tasks with negative lag
6400 	 * will move 'time' backwards, this can screw around with the lag of
6401 	 * other tasks.
6402 	 *
6403 	 * EEVDF: placement strategy #1 / #2
6404 	 */
6405 	if (sched_feat(PLACE_LAG) && nr_queued && se->vlag) {
6406 		struct sched_entity *curr = cfs_rq->curr;
6407 		long load, weight;
6408 
6409 		lag = se->vlag;
6410 
6411 		/*
6412 		 * If we want to place a task and preserve lag, we have to
6413 		 * consider the effect of the new entity on the weighted
6414 		 * average and compensate for this, otherwise lag can quickly
6415 		 * evaporate.
6416 		 *
6417 		 * Lag is defined as:
6418 		 *
6419 		 *   lag_i = S - s_i = w_i * (V - v_i)
6420 		 *
6421 		 * To avoid the 'w_i' term all over the place, we only track
6422 		 * the virtual lag:
6423 		 *
6424 		 *   vl_i = V - v_i <=> v_i = V - vl_i
6425 		 *
6426 		 * And we take V to be the weighted average of all v:
6427 		 *
6428 		 *   V = (\Sum w_j*v_j) / W
6429 		 *
6430 		 * Where W is: \Sum w_j
6431 		 *
6432 		 * Then, the weighted average after adding an entity with lag
6433 		 * vl_i is given by:
6434 		 *
6435 		 *   V' = (\Sum w_j*v_j + w_i*v_i) / (W + w_i)
6436 		 *      = (W*V + w_i*(V - vl_i)) / (W + w_i)
6437 		 *      = (W*V + w_i*V - w_i*vl_i) / (W + w_i)
6438 		 *      = (V*(W + w_i) - w_i*vl_i) / (W + w_i)
6439 		 *      = V - w_i*vl_i / (W + w_i)
6440 		 *
6441 		 * And the actual lag after adding an entity with vl_i is:
6442 		 *
6443 		 *   vl'_i = V' - v_i
6444 		 *         = V - w_i*vl_i / (W + w_i) - (V - vl_i)
6445 		 *         = vl_i - w_i*vl_i / (W + w_i)
6446 		 *
6447 		 * Which is strictly less than vl_i. So in order to preserve lag
6448 		 * we should inflate the lag before placement such that the
6449 		 * effective lag after placement comes out right.
6450 		 *
6451 		 * As such, invert the above relation for vl'_i to get the vl_i
6452 		 * we need to use such that the lag after placement is the lag
6453 		 * we computed before dequeue.
6454 		 *
6455 		 *   vl'_i = vl_i - w_i*vl_i / (W + w_i)
6456 		 *         = ((W + w_i)*vl_i - w_i*vl_i) / (W + w_i)
6457 		 *
6458 		 *   (W + w_i)*vl'_i = (W + w_i)*vl_i - w_i*vl_i
6459 		 *                   = W*vl_i
6460 		 *
6461 		 *   vl_i = (W + w_i)*vl'_i / W
6462 		 */
6463 		load = cfs_rq->sum_weight;
6464 		if (curr && curr->on_rq)
6465 			load += avg_vruntime_weight(cfs_rq, curr->h_load.weight);
6466 
6467 		weight = avg_vruntime_weight(cfs_rq, se->h_load.weight);
6468 		lag *= load + weight;
6469 		if (WARN_ON_ONCE(!load))
6470 			load = 1;
6471 		lag = div64_long(lag, load);
6472 
6473 		/*
6474 		 * A heavy entity (relative to the tree) will pull the
6475 		 * avg_vruntime close to its vruntime position on enqueue. But
6476 		 * the zero_vruntime point is only updated at the next
6477 		 * update_deadline()/place_entity()/update_entity_lag().
6478 		 *
6479 		 * Specifically (see the comment near avg_vruntime_weight()):
6480 		 *
6481 		 *   sum_w_vruntime = \Sum (v_i - v0) * w_i
6482 		 *
6483 		 * Note that if v0 is near a light entity, both terms will be
6484 		 * small for the light entity, while in that case both terms
6485 		 * are large for the heavy entity, leading to risk of
6486 		 * overflow.
6487 		 *
6488 		 * OTOH if v0 is near the heavy entity, then the difference is
6489 		 * larger for the light entity, but the factor is small, while
6490 		 * for the heavy entity the difference is small but the factor
6491 		 * is large. Avoiding the multiplication overflow.
6492 		 */
6493 		if (weight > load)
6494 			update_zero = true;
6495 	}
6496 
6497 	se->vruntime = vruntime - lag;
6498 
6499 	if (update_zero)
6500 		update_zero_vruntime(cfs_rq, -lag);
6501 
6502 	if (sched_feat(PLACE_REL_DEADLINE) && se->rel_deadline) {
6503 		se->deadline += se->vruntime;
6504 		se->rel_deadline = 0;
6505 		return;
6506 	}
6507 
6508 	/*
6509 	 * When joining the competition; the existing tasks will be,
6510 	 * on average, halfway through their slice, as such start tasks
6511 	 * off with half a slice to ease into the competition.
6512 	 */
6513 	if (sched_feat(PLACE_DEADLINE_INITIAL) && (flags & ENQUEUE_INITIAL))
6514 		vslice /= 2;
6515 
6516 	/*
6517 	 * EEVDF: vd_i = ve_i + r_i/w_i
6518 	 */
6519 	se->deadline = se->vruntime + vslice;
6520 }
6521 
6522 static void check_enqueue_throttle(struct cfs_rq *cfs_rq);
6523 static inline int cfs_rq_throttled(struct cfs_rq *cfs_rq);
6524 
6525 static void
6526 enqueue_entity(struct cfs_rq *cfs_rq, struct sched_entity *se, int flags)
6527 {
6528 	/*
6529 	 * When enqueuing a sched_entity, we must:
6530 	 *   - Update loads to have both entity and cfs_rq synced with now.
6531 	 *   - For group_entity, update its runnable_weight to reflect the new
6532 	 *     h_nr_runnable of its group cfs_rq.
6533 	 *   - For group_entity, update its weight to reflect the new share of
6534 	 *     its group cfs_rq
6535 	 *   - Add its new weight to cfs_rq->load.weight
6536 	 */
6537 	update_load_avg(cfs_rq, se, UPDATE_TG | DO_ATTACH);
6538 	se_update_runnable(se);
6539 	/*
6540 	 * XXX update_load_avg() above will have attached us to the pelt sum;
6541 	 * but update_cfs_group() here will re-adjust the weight and have to
6542 	 * undo/redo all that. Seems wasteful.
6543 	 */
6544 	update_cfs_group(se);
6545 
6546 	account_entity_enqueue(cfs_rq, se);
6547 
6548 	/* Entity has migrated, no longer consider this task hot */
6549 	if (flags & ENQUEUE_MIGRATED)
6550 		se->exec_start = 0;
6551 
6552 	check_schedstat_required();
6553 	update_stats_enqueue_fair(cfs_rq, se, flags);
6554 	se->on_rq = 1;
6555 
6556 	if (cfs_rq->nr_queued == 1) {
6557 		check_enqueue_throttle(cfs_rq);
6558 		list_add_leaf_cfs_rq(cfs_rq);
6559 #ifdef CONFIG_CFS_BANDWIDTH
6560 		if (cfs_rq->pelt_clock_throttled) {
6561 			struct rq *rq = rq_of(cfs_rq);
6562 
6563 			cfs_rq->throttled_clock_pelt_time += rq_clock_pelt(rq) -
6564 				cfs_rq->throttled_clock_pelt;
6565 			cfs_rq->pelt_clock_throttled = 0;
6566 		}
6567 #endif
6568 	}
6569 }
6570 
6571 static void set_next_buddy(struct cfs_rq *cfs_rq, struct sched_entity *se)
6572 {
6573 	if (WARN_ON_ONCE(!se->on_rq || se->sched_delayed))
6574 		return;
6575 	if (se_is_idle(se))
6576 		return;
6577 	cfs_rq->next = se;
6578 }
6579 
6580 static void clear_buddies(struct cfs_rq *cfs_rq, struct sched_entity *se)
6581 {
6582 	if (cfs_rq->next == se)
6583 		cfs_rq->next = NULL;
6584 }
6585 
6586 static __always_inline void return_cfs_rq_runtime(struct cfs_rq *cfs_rq);
6587 
6588 static void set_delayed(struct sched_entity *se)
6589 {
6590 	/*
6591 	 * Delayed se of cfs_rq have no tasks queued on them.
6592 	 * Do not adjust h_nr_runnable since __dequeue_task()
6593 	 * will account it for blocked tasks.
6594 	 *
6595 	 * This check can be removed because when flat pick
6596 	 * patches get merged as only task can get delayed,
6597 	 * same for clear_delayed().
6598 	 */
6599 	if (!entity_is_task(se)) {
6600 		se->sched_delayed = 1;
6601 		return;
6602 	}
6603 
6604 	/*
6605 	 * Drop a task leaving the runnable set.
6606 	 * Needs to be called before sched_delayed is set.
6607 	 * clear_delayed() mirrors this after clearing the flag.
6608 	 */
6609 	pref_llc_running_dec(rq_of(cfs_rq_of(se)), task_of(se));
6610 	se->sched_delayed = 1;
6611 
6612 	for_each_sched_entity(se) {
6613 		struct cfs_rq *cfs_rq = cfs_rq_of(se);
6614 
6615 		cfs_rq->h_nr_runnable--;
6616 	}
6617 }
6618 
6619 static void clear_delayed(struct sched_entity *se)
6620 {
6621 	se->sched_delayed = 0;
6622 
6623 	/*
6624 	 * Delayed se of cfs_rq have no tasks queued on them.
6625 	 * Do not adjust h_nr_runnable since a dequeue has
6626 	 * already accounted for it or an enqueue of a task
6627 	 * below it will account for it in enqueue_task_fair().
6628 	 */
6629 	if (!entity_is_task(se))
6630 		return;
6631 
6632 	/*
6633 	 * Re-add on wake, after sched_delayed is cleared. On a final delayed
6634 	 * dequeue account_llc_dequeue() already cleared pref_llc_queued, so
6635 	 * this does nothing.
6636 	 */
6637 	pref_llc_running_inc(rq_of(cfs_rq_of(se)), task_of(se));
6638 
6639 	for_each_sched_entity(se) {
6640 		struct cfs_rq *cfs_rq = cfs_rq_of(se);
6641 
6642 		cfs_rq->h_nr_runnable++;
6643 	}
6644 }
6645 
6646 static void
6647 dequeue_entity(struct cfs_rq *cfs_rq, struct sched_entity *se, int flags)
6648 {
6649 	int action = UPDATE_TG;
6650 
6651 	if (entity_is_task(se)) {
6652 		if (task_on_rq_migrating(task_of(se)))
6653 			action |= DO_DETACH;
6654 
6655 		if ((flags & DEQUEUE_SLEEP) && !(flags & DEQUEUE_DELAYED))
6656 			action |= UPDATE_UTIL_EST;
6657 	}
6658 
6659 	/*
6660 	 * When dequeuing a sched_entity, we must:
6661 	 *   - Update loads to have both entity and cfs_rq synced with now.
6662 	 *   - For group_entity, update its runnable_weight to reflect the new
6663 	 *     h_nr_runnable of its group cfs_rq.
6664 	 *   - Subtract its previous weight from cfs_rq->load.weight.
6665 	 *   - For group entity, update its weight to reflect the new share
6666 	 *     of its group cfs_rq.
6667 	 */
6668 	update_load_avg(cfs_rq, se, action);
6669 	se_update_runnable(se);
6670 
6671 	update_stats_dequeue_fair(cfs_rq, se, flags);
6672 
6673 	se->on_rq = 0;
6674 	account_entity_dequeue(cfs_rq, se);
6675 
6676 	/* return excess runtime on last dequeue */
6677 	return_cfs_rq_runtime(cfs_rq);
6678 
6679 	update_cfs_group(se);
6680 
6681 	if (cfs_rq->nr_queued == 0) {
6682 		update_idle_cfs_rq_clock_pelt(cfs_rq);
6683 #ifdef CONFIG_CFS_BANDWIDTH
6684 		if (throttled_hierarchy(cfs_rq)) {
6685 			struct rq *rq = rq_of(cfs_rq);
6686 
6687 			list_del_leaf_cfs_rq(cfs_rq);
6688 			cfs_rq->throttled_clock_pelt = rq_clock_pelt(rq);
6689 			cfs_rq->pelt_clock_throttled = 1;
6690 		}
6691 #endif
6692 	}
6693 }
6694 
6695 static void
6696 set_next_entity(struct cfs_rq *cfs_rq, struct sched_entity *se)
6697 {
6698 	/* 'current' is not kept within the tree. */
6699 	if (se->on_rq) {
6700 		/*
6701 		 * Any task has to be enqueued before it get to execute on
6702 		 * a CPU. So account for the time it spent waiting on the
6703 		 * runqueue.
6704 		 */
6705 		update_stats_wait_end_fair(cfs_rq, se);
6706 		update_load_avg(cfs_rq, se, UPDATE_TG);
6707 	}
6708 
6709 	update_stats_curr_start(cfs_rq, se);
6710 	WARN_ON_ONCE(cfs_rq->h_curr);
6711 	cfs_rq->h_curr = se;
6712 
6713 	/*
6714 	 * Track our maximum slice length, if the CPU's load is at
6715 	 * least twice that of our own weight (i.e. don't track it
6716 	 * when there are only lesser-weight tasks around):
6717 	 */
6718 	if (schedstat_enabled() &&
6719 	    rq_of(cfs_rq)->cfs.load.weight >= 2*se->load.weight) {
6720 		struct sched_statistics *stats;
6721 
6722 		stats = __schedstats_from_se(se);
6723 		__schedstat_set(stats->slice_max,
6724 				max((u64)stats->slice_max,
6725 				    se->sum_exec_runtime - se->prev_sum_exec_runtime));
6726 	}
6727 
6728 	se->prev_sum_exec_runtime = se->sum_exec_runtime;
6729 }
6730 
6731 static bool __dequeue_task(struct rq *rq, struct task_struct *p, int flags);
6732 
6733 static struct sched_entity *
6734 pick_next_entity(struct rq *rq, bool protect)
6735 {
6736 	struct cfs_rq *cfs_rq = &rq->cfs;
6737 	struct sched_entity *se;
6738 
6739 	se = pick_eevdf(cfs_rq, protect);
6740 	if (se->sched_delayed) {
6741 		__dequeue_task(rq, task_of(se), DEQUEUE_SLEEP | DEQUEUE_DELAYED);
6742 		/*
6743 		 * Must not reference @se again, see __block_task().
6744 		 */
6745 		return NULL;
6746 	}
6747 	return se;
6748 }
6749 
6750 static void put_prev_entity(struct cfs_rq *cfs_rq, struct sched_entity *prev)
6751 {
6752 	/*
6753 	 * If still on the runqueue then deactivate_task()
6754 	 * was not called and update_curr() has to be done:
6755 	 */
6756 	if (prev->on_rq)
6757 		update_curr(cfs_rq);
6758 
6759 	if (prev->on_rq) {
6760 		update_stats_wait_start_fair(cfs_rq, prev);
6761 		/* in !on_rq case, update occurred at dequeue */
6762 		update_load_avg(cfs_rq, prev, 0);
6763 	}
6764 	WARN_ON_ONCE(cfs_rq->h_curr != prev);
6765 	cfs_rq->h_curr = NULL;
6766 }
6767 
6768 static void
6769 entity_tick(struct cfs_rq *cfs_rq, struct sched_entity *curr, int queued)
6770 {
6771 	/*
6772 	 * Update run-time statistics of the 'current'.
6773 	 */
6774 	update_curr(cfs_rq);
6775 
6776 	/*
6777 	 * Ensure that runnable average is periodically updated.
6778 	 */
6779 	update_load_avg(cfs_rq, curr, UPDATE_TG);
6780 	update_cfs_group(curr);
6781 
6782 #ifdef CONFIG_SCHED_HRTICK
6783 	/*
6784 	 * queued ticks are scheduled to match the slice, so don't bother
6785 	 * validating it and just reschedule.
6786 	 */
6787 	if (queued) {
6788 		resched_curr(rq_of(cfs_rq));
6789 		return;
6790 	}
6791 #endif
6792 }
6793 
6794 
6795 /**************************************************
6796  * CFS bandwidth control machinery
6797  */
6798 
6799 #ifdef CONFIG_CFS_BANDWIDTH
6800 
6801 #ifdef CONFIG_JUMP_LABEL
6802 static struct static_key __cfs_bandwidth_used;
6803 
6804 static inline bool cfs_bandwidth_used(void)
6805 {
6806 	return static_key_false(&__cfs_bandwidth_used);
6807 }
6808 
6809 void cfs_bandwidth_usage_inc(void)
6810 {
6811 	static_key_slow_inc_cpuslocked(&__cfs_bandwidth_used);
6812 }
6813 
6814 void cfs_bandwidth_usage_dec(void)
6815 {
6816 	static_key_slow_dec_cpuslocked(&__cfs_bandwidth_used);
6817 }
6818 #else /* !CONFIG_JUMP_LABEL: */
6819 static bool cfs_bandwidth_used(void)
6820 {
6821 	return true;
6822 }
6823 
6824 void cfs_bandwidth_usage_inc(void) {}
6825 void cfs_bandwidth_usage_dec(void) {}
6826 #endif /* !CONFIG_JUMP_LABEL */
6827 
6828 static inline u64 sched_cfs_bandwidth_slice(void)
6829 {
6830 	return (u64)sysctl_sched_cfs_bandwidth_slice * NSEC_PER_USEC;
6831 }
6832 
6833 /*
6834  * Replenish runtime according to assigned quota. We use sched_clock_cpu
6835  * directly instead of rq->clock to avoid adding additional synchronization
6836  * around rq->lock.
6837  *
6838  * requires cfs_b->lock
6839  */
6840 void __refill_cfs_bandwidth_runtime(struct cfs_bandwidth *cfs_b)
6841 {
6842 	s64 runtime;
6843 
6844 	if (unlikely(cfs_b->quota == RUNTIME_INF))
6845 		return;
6846 
6847 	cfs_b->runtime += cfs_b->quota;
6848 	runtime = cfs_b->runtime_snap - cfs_b->runtime;
6849 	if (runtime > 0) {
6850 		cfs_b->burst_time += runtime;
6851 		cfs_b->nr_burst++;
6852 	}
6853 
6854 	cfs_b->runtime = min(cfs_b->runtime, cfs_b->quota + cfs_b->burst);
6855 	cfs_b->runtime_snap = cfs_b->runtime;
6856 }
6857 
6858 static inline struct cfs_bandwidth *tg_cfs_bandwidth(struct task_group *tg)
6859 {
6860 	return &tg->cfs_bandwidth;
6861 }
6862 
6863 /* returns 0 on failure to allocate runtime */
6864 static int __assign_cfs_rq_runtime(struct cfs_bandwidth *cfs_b,
6865 				   struct cfs_rq *cfs_rq, u64 target_runtime)
6866 {
6867 	u64 min_amount, amount = 0;
6868 
6869 	lockdep_assert_held(&cfs_b->lock);
6870 
6871 	/* note: this is a positive sum as runtime_remaining <= 0 */
6872 	min_amount = target_runtime - cfs_rq->runtime_remaining;
6873 
6874 	if (cfs_b->quota == RUNTIME_INF)
6875 		amount = min_amount;
6876 	else {
6877 		start_cfs_bandwidth(cfs_b);
6878 
6879 		if (cfs_b->runtime > 0) {
6880 			amount = min(cfs_b->runtime, min_amount);
6881 			cfs_b->runtime -= amount;
6882 			cfs_b->idle = 0;
6883 		}
6884 	}
6885 
6886 	cfs_rq->runtime_remaining += amount;
6887 
6888 	return cfs_rq->runtime_remaining > 0;
6889 }
6890 
6891 static bool throttle_cfs_rq(struct cfs_rq *cfs_rq);
6892 
6893 static bool __account_cfs_rq_runtime(struct cfs_rq *cfs_rq, u64 delta_exec)
6894 {
6895 	/* dock delta_exec before expiring quota (as it could span periods) */
6896 	cfs_rq->runtime_remaining -= delta_exec;
6897 
6898 	if (likely(cfs_rq->runtime_remaining > 0))
6899 		return false;
6900 
6901 	if (cfs_rq->throttled)
6902 		return true;
6903 	/*
6904 	 * throttle_cfs_rq() will try to extend the runtime first
6905 	 * before throttling the hierarchy.
6906 	 */
6907 	return throttle_cfs_rq(cfs_rq);
6908 }
6909 
6910 static __always_inline
6911 bool account_cfs_rq_runtime(struct cfs_rq *cfs_rq, u64 delta_exec)
6912 {
6913 	if (!cfs_bandwidth_used() || !cfs_rq->runtime_enabled)
6914 		return false;
6915 
6916 	return __account_cfs_rq_runtime(cfs_rq, delta_exec);
6917 }
6918 
6919 static inline int cfs_rq_throttled(struct cfs_rq *cfs_rq)
6920 {
6921 	return cfs_bandwidth_used() && cfs_rq->throttled;
6922 }
6923 
6924 static inline bool cfs_rq_pelt_clock_throttled(struct cfs_rq *cfs_rq)
6925 {
6926 	return cfs_bandwidth_used() && cfs_rq->pelt_clock_throttled;
6927 }
6928 
6929 /* check whether cfs_rq, or any parent, is throttled */
6930 static inline int throttled_hierarchy(struct cfs_rq *cfs_rq)
6931 {
6932 	return cfs_bandwidth_used() && cfs_rq->throttle_count;
6933 }
6934 
6935 static inline int lb_throttled_hierarchy(struct task_struct *p, int dst_cpu)
6936 {
6937 	return throttled_hierarchy(tg_cfs_rq(task_group(p), dst_cpu));
6938 }
6939 
6940 static inline bool task_is_throttled(struct task_struct *p)
6941 {
6942 	return cfs_bandwidth_used() && p->throttled;
6943 }
6944 
6945 static bool dequeue_task_fair(struct rq *rq, struct task_struct *p, int flags);
6946 static void throttle_cfs_rq_work(struct callback_head *work)
6947 {
6948 	struct task_struct *p = container_of(work, struct task_struct, sched_throttle_work);
6949 	struct sched_entity *se;
6950 	struct cfs_rq *cfs_rq;
6951 	struct rq *rq;
6952 
6953 	WARN_ON_ONCE(p != current);
6954 	p->sched_throttle_work.next = &p->sched_throttle_work;
6955 
6956 	/*
6957 	 * If task is exiting, then there won't be a return to userspace, so we
6958 	 * don't have to bother with any of this.
6959 	 */
6960 	if ((p->flags & PF_EXITING))
6961 		return;
6962 
6963 	scoped_guard(task_rq_lock, p) {
6964 		se = &p->se;
6965 		cfs_rq = cfs_rq_of(se);
6966 
6967 		/* Raced, forget */
6968 		if (p->sched_class != &fair_sched_class)
6969 			return;
6970 
6971 		/*
6972 		 * If not in limbo, then either replenish has happened or this
6973 		 * task got migrated out of the throttled cfs_rq, move along.
6974 		 */
6975 		if (!cfs_rq->throttle_count)
6976 			return;
6977 		rq = scope.rq;
6978 		update_rq_clock(rq);
6979 		WARN_ON_ONCE(p->throttled || !list_empty(&p->throttle_node));
6980 		dequeue_task_fair(rq, p, DEQUEUE_SLEEP | DEQUEUE_THROTTLE);
6981 		list_add(&p->throttle_node, &cfs_rq->throttled_limbo_list);
6982 		/*
6983 		 * Must not set throttled before dequeue or dequeue will
6984 		 * mistakenly regard this task as an already throttled one.
6985 		 */
6986 		p->throttled = true;
6987 		resched_curr(rq);
6988 	}
6989 }
6990 
6991 void init_cfs_throttle_work(struct task_struct *p)
6992 {
6993 	init_task_work(&p->sched_throttle_work, throttle_cfs_rq_work);
6994 	/* Protect against double add, see throttle_cfs_rq() and throttle_cfs_rq_work() */
6995 	p->sched_throttle_work.next = &p->sched_throttle_work;
6996 	INIT_LIST_HEAD(&p->throttle_node);
6997 }
6998 
6999 /*
7000  * Task is throttled and someone wants to dequeue it again:
7001  * it could be sched/core when core needs to do things like
7002  * task affinity change, task group change, task sched class
7003  * change etc. and in these cases, DEQUEUE_SLEEP is not set;
7004  * or the task is blocked after throttled due to freezer etc.
7005  * and in these cases, DEQUEUE_SLEEP is set.
7006  */
7007 static void detach_task_cfs_rq(struct task_struct *p);
7008 static void dequeue_throttled_task(struct task_struct *p, int flags)
7009 {
7010 	WARN_ON_ONCE(p->se.on_rq);
7011 	list_del_init(&p->throttle_node);
7012 
7013 	/* task blocked after throttled */
7014 	if (flags & DEQUEUE_SLEEP) {
7015 		p->throttled = false;
7016 		return;
7017 	}
7018 
7019 	/*
7020 	 * task is migrating off its old cfs_rq, detach
7021 	 * the task's load from its old cfs_rq.
7022 	 */
7023 	if (task_on_rq_migrating(p))
7024 		detach_task_cfs_rq(p);
7025 }
7026 
7027 static bool enqueue_throttled_task(struct task_struct *p)
7028 {
7029 	struct cfs_rq *cfs_rq = cfs_rq_of(&p->se);
7030 
7031 	/* @p should have gone through dequeue_throttled_task() first */
7032 	WARN_ON_ONCE(!list_empty(&p->throttle_node));
7033 
7034 	/*
7035 	 * If the throttled task @p is enqueued to a throttled cfs_rq,
7036 	 * take the fast path by directly putting the task on the
7037 	 * target cfs_rq's limbo list.
7038 	 *
7039 	 * Do not do that when @p is current because the following race can
7040 	 * cause @p's group_node to be incorectly re-insterted in its rq's
7041 	 * cfs_tasks list, despite being throttled:
7042 	 *
7043 	 *     cpuX                       cpuY
7044 	 *   p ret2user
7045 	 *  throttle_cfs_rq_work()  sched_move_task(p)
7046 	 *  LOCK task_rq_lock
7047 	 *  dequeue_task_fair(p)
7048 	 *  UNLOCK task_rq_lock
7049 	 *                          LOCK task_rq_lock
7050 	 *                          task_current_donor(p) == true
7051 	 *                          task_on_rq_queued(p) == true
7052 	 *                          dequeue_task(p)
7053 	 *                          put_prev_task(p)
7054 	 *                          sched_change_group()
7055 	 *                          enqueue_task(p) -> p's new cfs_rq
7056 	 *                                             is throttled, go
7057 	 *                                             fast path and skip
7058 	 *                                             actual enqueue
7059 	 *                          set_next_task(p)
7060 	 *                    list_move(&se->group_node, &rq->cfs_tasks); // bug
7061 	 *  schedule()
7062 	 *
7063 	 * In the above race case, @p current cfs_rq is in the same rq as
7064 	 * its previous cfs_rq because sched_move_task() only moves a task
7065 	 * to a different group from the same rq, so we can use its current
7066 	 * cfs_rq to derive rq and test if the task is current.
7067 	 */
7068 	if (throttled_hierarchy(cfs_rq) &&
7069 	    !task_current_donor(rq_of(cfs_rq), p)) {
7070 		list_add(&p->throttle_node, &cfs_rq->throttled_limbo_list);
7071 		return true;
7072 	}
7073 
7074 	/* we can't take the fast path, do an actual enqueue*/
7075 	p->throttled = false;
7076 	return false;
7077 }
7078 
7079 static void enqueue_task_fair(struct rq *rq, struct task_struct *p, int flags);
7080 static int tg_unthrottle_up(struct task_group *tg, void *data)
7081 {
7082 	struct rq *rq = data;
7083 	struct cfs_rq *cfs_rq = tg_cfs_rq(tg, cpu_of(rq));
7084 	struct task_struct *p, *tmp;
7085 	LIST_HEAD(throttled_tasks);
7086 
7087 	/*
7088 	 * If cfs_rq->curr is set, the cfs_rq might not have caught up
7089 	 * since the last clock update. Do it now before we begin
7090 	 * queueing task onto it to save the need for unnecessarily
7091 	 * unthrottle the hierarchy for this cfs_rq to be throttled
7092 	 * right back again.
7093 	 */
7094 	update_curr(cfs_rq);
7095 
7096 	if (--cfs_rq->throttle_count)
7097 		return 0;
7098 
7099 	if (cfs_rq->pelt_clock_throttled) {
7100 		cfs_rq->throttled_clock_pelt_time += rq_clock_pelt(rq) -
7101 					     cfs_rq->throttled_clock_pelt;
7102 		cfs_rq->pelt_clock_throttled = 0;
7103 	}
7104 
7105 	if (cfs_rq->throttled_clock_self) {
7106 		u64 delta = rq_clock(rq) - cfs_rq->throttled_clock_self;
7107 
7108 		cfs_rq->throttled_clock_self = 0;
7109 
7110 		if (WARN_ON_ONCE((s64)delta < 0))
7111 			delta = 0;
7112 
7113 		cfs_rq->throttled_clock_self_time += delta;
7114 	}
7115 
7116 	/*
7117 	 * Move the tasks to a local list since an update_curr() during
7118 	 * enqueue_task_fair() can throttle a higher cfs_rq, and it can
7119 	 * see the "throttled_limbo_list" being non-empty in
7120 	 * tg_throttle_down() if throttle_count turned 0 above.
7121 	 */
7122 	list_splice_init(&cfs_rq->throttled_limbo_list, &throttled_tasks);
7123 
7124 	/* Re-enqueue the tasks that have been throttled at this level. */
7125 	list_for_each_entry_safe(p, tmp, &throttled_tasks, throttle_node) {
7126 		/*
7127 		 * Back to being throttled! Break out and put the remaining
7128 		 * tasks back onto the limbo_list to prevent running them
7129 		 * unnecessarily.
7130 		 */
7131 		if (cfs_rq->throttle_count)
7132 			break;
7133 
7134 		list_del_init(&p->throttle_node);
7135 		p->throttled = false;
7136 		enqueue_task_fair(rq, p, ENQUEUE_WAKEUP);
7137 	}
7138 
7139 	list_splice(&throttled_tasks, &cfs_rq->throttled_limbo_list);
7140 
7141 	/* Add cfs_rq with load or one or more already running entities to the list */
7142 	if (!cfs_rq_is_decayed(cfs_rq))
7143 		list_add_leaf_cfs_rq(cfs_rq);
7144 
7145 	return 0;
7146 }
7147 
7148 static inline bool task_has_throttle_work(struct task_struct *p)
7149 {
7150 	return p->sched_throttle_work.next != &p->sched_throttle_work;
7151 }
7152 
7153 static inline void task_throttle_setup_work(struct task_struct *p)
7154 {
7155 	if (task_has_throttle_work(p))
7156 		return;
7157 
7158 	/*
7159 	 * Kthreads and exiting tasks don't return to userspace, so adding the
7160 	 * work is pointless
7161 	 */
7162 	if ((p->flags & (PF_EXITING | PF_KTHREAD)))
7163 		return;
7164 
7165 	task_work_add(p, &p->sched_throttle_work, TWA_RESUME);
7166 }
7167 
7168 static void record_throttle_clock(struct cfs_rq *cfs_rq)
7169 {
7170 	struct rq *rq = rq_of(cfs_rq);
7171 
7172 	if (cfs_rq_throttled(cfs_rq) && !cfs_rq->throttled_clock)
7173 		cfs_rq->throttled_clock = rq_clock(rq);
7174 
7175 	if (!cfs_rq->throttled_clock_self)
7176 		cfs_rq->throttled_clock_self = rq_clock(rq);
7177 }
7178 
7179 static int tg_throttle_down(struct task_group *tg, void *data)
7180 {
7181 	struct rq *rq = data;
7182 	struct cfs_rq *cfs_rq = tg_cfs_rq(tg, cpu_of(rq));
7183 
7184 	if (cfs_rq->throttle_count++)
7185 		return 0;
7186 
7187 	/*
7188 	 * For cfs_rqs that still have entities enqueued, PELT clock
7189 	 * stop happens at dequeue time when all entities are dequeued.
7190 	 */
7191 	if (!cfs_rq->nr_queued) {
7192 		list_del_leaf_cfs_rq(cfs_rq);
7193 		cfs_rq->throttled_clock_pelt = rq_clock_pelt(rq);
7194 		cfs_rq->pelt_clock_throttled = 1;
7195 	}
7196 
7197 	WARN_ON_ONCE(cfs_rq->throttled_clock_self);
7198 	WARN_ON_ONCE(!list_empty(&cfs_rq->throttled_limbo_list));
7199 	return 0;
7200 }
7201 
7202 static bool throttle_cfs_rq(struct cfs_rq *cfs_rq)
7203 {
7204 	struct cfs_bandwidth *cfs_b = tg_cfs_bandwidth(cfs_rq->tg);
7205 	struct sched_entity *curr = cfs_rq->h_curr;
7206 	struct rq *rq = rq_of(cfs_rq);
7207 
7208 	scoped_guard(raw_spinlock, &cfs_b->lock) {
7209 		u64 target_runtime = 1;
7210 
7211 		/*
7212 		 * If cfs_rq->h_curr is still runnable, we are here from an
7213 		 * update_curr(). Request sysctl_sched_cfs_bandwidth_slice
7214 		 * worth of bandwidth to continue running.
7215 		 *
7216 		 * If the curr is not runnable, just request enough bandwidth
7217 		 * to be runnable next time the pick selects this cfs_rq.
7218 		 */
7219 		if (curr && curr->on_rq)
7220 			target_runtime = sched_cfs_bandwidth_slice();
7221 
7222 		/*
7223 		 * Check if We have raced with bandwidth becoming available. If
7224 		 * we actually throttled the timer might not unthrottle us for
7225 		 * an entire period. We additionally needed to make sure that
7226 		 * any subsequent check_cfs_rq_runtime calls agree not to
7227 		 * throttle us, as we may commit to do cfs put_prev+pick_next,
7228 		 * so we ask for 1ns of runtime rather than just check cfs_b.
7229 		 *
7230 		 * This will start the period timer if necessary.
7231 		 */
7232 		if (__assign_cfs_rq_runtime(cfs_b, cfs_rq, target_runtime))
7233 			return false;
7234 
7235 		/*
7236 		 * No bandwidth available; Add ourselves on the list to be
7237 		 * unthrottled later.
7238 		 */
7239 		list_add_tail_rcu(&cfs_rq->throttled_list,
7240 				  &cfs_b->throttled_cfs_rq);
7241 	}
7242 
7243 	/* freeze hierarchy runnable averages while throttled */
7244 	scoped_guard(rcu)
7245 		walk_tg_tree_from(cfs_rq->tg, tg_throttle_down, tg_nop, (void *)rq);
7246 
7247 	/*
7248 	 * Note: distribution will already see us throttled via the
7249 	 * throttled-list.  rq->lock protects completion.
7250 	 */
7251 	cfs_rq->throttled = 1;
7252 	WARN_ON_ONCE(cfs_rq->throttled_clock);
7253 
7254 	/*
7255 	 * If current hierarchy was throttled, add throttle work to the
7256 	 * current donor. In case of proxy-execution, the execution
7257 	 * context cannot exit to the userspace while holding a mutex
7258 	 * and the rule of throttle deferral to only throttle the
7259 	 * throttled context at exit to userspace is still preserved.
7260 	 */
7261 	if (curr && curr->on_rq)
7262 		task_throttle_setup_work(rq->donor);
7263 
7264 	return true;
7265 }
7266 
7267 void unthrottle_cfs_rq(struct cfs_rq *cfs_rq)
7268 {
7269 	struct rq *rq = rq_of(cfs_rq);
7270 	struct cfs_bandwidth *cfs_b = tg_cfs_bandwidth(cfs_rq->tg);
7271 	struct sched_entity *se = cfs_rq_se(cfs_rq);
7272 
7273 	/*
7274 	 * It's possible we are called with runtime_remaining < 0 due to things
7275 	 * like async unthrottled us with a positive runtime_remaining but other
7276 	 * still running entities consumed those runtime before we reached here.
7277 	 *
7278 	 * We can't unthrottle this cfs_rq without any runtime remaining because
7279 	 * any enqueue in tg_unthrottle_up() will immediately trigger a throttle,
7280 	 * which is not supposed to happen on unthrottle path.
7281 	 *
7282 	 * Catch up on the remaining runtime since last clock update before
7283 	 * checking runtime remaining.
7284 	 */
7285 	update_curr(cfs_rq);
7286 	if (cfs_rq->runtime_enabled && cfs_rq->runtime_remaining <= 0)
7287 		return;
7288 
7289 	cfs_rq->throttled = 0;
7290 
7291 	scoped_guard(raw_spinlock, &cfs_b->lock) {
7292 		list_del_rcu(&cfs_rq->throttled_list);
7293 
7294 		if (!cfs_rq->throttled_clock)
7295 			break;
7296 
7297 		cfs_b->throttled_time += rq_clock(rq) - cfs_rq->throttled_clock;
7298 		cfs_rq->throttled_clock = 0;
7299 	}
7300 
7301 	/* update hierarchical throttle state */
7302 	walk_tg_tree_from(cfs_rq->tg, tg_nop, tg_unthrottle_up, (void *)rq);
7303 
7304 	if (!cfs_rq->load.weight) {
7305 		if (!cfs_rq->on_list)
7306 			return;
7307 		/*
7308 		 * Nothing to run but something to decay (on_list)?
7309 		 * Complete the branch.
7310 		 */
7311 		for_each_sched_entity(se) {
7312 			if (list_add_leaf_cfs_rq(cfs_rq_of(se)))
7313 				break;
7314 		}
7315 	}
7316 
7317 	assert_list_leaf_cfs_rq(rq);
7318 
7319 	/* Determine whether we need to wake up potentially idle CPU: */
7320 	if (rq->curr == rq->idle && rq->cfs.h_nr_queued)
7321 		resched_curr(rq);
7322 }
7323 
7324 static void __cfsb_csd_unthrottle(void *arg)
7325 {
7326 	struct cfs_rq *cursor, *tmp;
7327 	struct rq *rq = arg;
7328 
7329 	guard(rq_lock)(rq);
7330 
7331 	/*
7332 	 * Iterating over the list can trigger several call to
7333 	 * update_rq_clock() in unthrottle_cfs_rq().
7334 	 * Do it once and skip the potential next ones.
7335 	 */
7336 	update_rq_clock(rq);
7337 	rq_clock_start_loop_update(rq);
7338 
7339 	/*
7340 	 * Since we hold rq lock we're safe from concurrent manipulation of
7341 	 * the CSD list. However, this RCU critical section annotates the
7342 	 * fact that we pair with sched_free_group_rcu(), so that we cannot
7343 	 * race with group being freed in the window between removing it
7344 	 * from the list and advancing to the next entry in the list.
7345 	 */
7346 	guard(rcu)();
7347 
7348 	list_for_each_entry_safe(cursor, tmp, &rq->cfsb_csd_list,
7349 				 throttled_csd_list) {
7350 		list_del_init(&cursor->throttled_csd_list);
7351 
7352 		if (cfs_rq_throttled(cursor))
7353 			unthrottle_cfs_rq(cursor);
7354 	}
7355 
7356 	rq_clock_stop_loop_update(rq);
7357 }
7358 
7359 static inline void __unthrottle_cfs_rq_async(struct cfs_rq *cfs_rq)
7360 {
7361 	struct rq *rq = rq_of(cfs_rq);
7362 	bool first;
7363 
7364 	if (rq == this_rq()) {
7365 		update_rq_clock(rq);
7366 		unthrottle_cfs_rq(cfs_rq);
7367 		return;
7368 	}
7369 
7370 	/* Already enqueued */
7371 	if (WARN_ON_ONCE(!list_empty(&cfs_rq->throttled_csd_list)))
7372 		return;
7373 
7374 	first = list_empty(&rq->cfsb_csd_list);
7375 	list_add_tail(&cfs_rq->throttled_csd_list, &rq->cfsb_csd_list);
7376 	if (first)
7377 		smp_call_function_single_async(cpu_of(rq), &rq->cfsb_csd);
7378 }
7379 
7380 static void unthrottle_cfs_rq_async(struct cfs_rq *cfs_rq)
7381 {
7382 	lockdep_assert_rq_held(rq_of(cfs_rq));
7383 
7384 	if (WARN_ON_ONCE(!cfs_rq_throttled(cfs_rq) ||
7385 	    cfs_rq->runtime_remaining <= 0))
7386 		return;
7387 
7388 	__unthrottle_cfs_rq_async(cfs_rq);
7389 }
7390 
7391 static bool distribute_cfs_runtime(struct cfs_bandwidth *cfs_b)
7392 {
7393 	bool throttled = false, unthrottle_local = false;
7394 	int this_cpu = smp_processor_id();
7395 	u64 runtime, remaining = 1;
7396 	struct cfs_rq *cfs_rq;
7397 	struct rq *rq;
7398 
7399 	guard(rcu)();
7400 
7401 	list_for_each_entry_rcu(cfs_rq, &cfs_b->throttled_cfs_rq,
7402 				throttled_list) {
7403 		rq = rq_of(cfs_rq);
7404 
7405 		if (!remaining) {
7406 			throttled = true;
7407 			break;
7408 		}
7409 
7410 		guard(rq_lock_irqsave)(rq);
7411 
7412 		if (!cfs_rq_throttled(cfs_rq))
7413 			continue;
7414 
7415 		/* Already queued for async unthrottle */
7416 		if (!list_empty(&cfs_rq->throttled_csd_list))
7417 			continue;
7418 
7419 		if (cfs_rq->h_curr) {
7420 			update_rq_clock(rq);
7421 			update_curr(cfs_rq);
7422 		}
7423 
7424 		/* By the above checks, this should never be true */
7425 		WARN_ON_ONCE(cfs_rq->runtime_remaining > 0);
7426 
7427 		scoped_guard(raw_spinlock, &cfs_b->lock) {
7428 			runtime = -cfs_rq->runtime_remaining + 1;
7429 			if (runtime > cfs_b->runtime)
7430 				runtime = cfs_b->runtime;
7431 			cfs_b->runtime -= runtime;
7432 			remaining = cfs_b->runtime;
7433 		}
7434 
7435 		cfs_rq->runtime_remaining += runtime;
7436 
7437 		/*
7438 		 * Ran out of bandwidth during distribution!
7439 		 * Indicate throttled entities and break early.
7440 		 */
7441 		if (cfs_rq->runtime_remaining <= 0) {
7442 			throttled = true;
7443 			break;
7444 		}
7445 
7446 		/* we check whether we're throttled above */
7447 		if (cpu_of(rq) != this_cpu) {
7448 			unthrottle_cfs_rq_async(cfs_rq);
7449 			continue;
7450 		}
7451 
7452 		/*
7453 		 * Allow a parallel async unthrottle to unthrottle
7454 		 * this cfs_rq too via __cfsb_csd_unthrottle().
7455 		 * If we are first, do it ourselves at the end and
7456 		 * save on an IPI from remote CPUs.
7457 		 */
7458 		unthrottle_local = list_empty(&rq->cfsb_csd_list);
7459 		list_add_tail(&cfs_rq->throttled_csd_list, &rq->cfsb_csd_list);
7460 	}
7461 
7462 	if (unthrottle_local) {
7463 		/*
7464 		 * Protect against an IPI that is also trying to flush
7465 		 * the unthrottled cfs_rq(s) from this CPU's csd_list.
7466 		 */
7467 		scoped_guard(irqsave)
7468 			__cfsb_csd_unthrottle(cpu_rq(this_cpu));
7469 	}
7470 
7471 	return throttled;
7472 }
7473 
7474 /*
7475  * Responsible for refilling a task_group's bandwidth and unthrottling its
7476  * cfs_rqs as appropriate. If there has been no activity within the last
7477  * period the timer is deactivated until scheduling resumes; cfs_b->idle is
7478  * used to track this state.
7479  */
7480 static int do_sched_cfs_period_timer(struct cfs_bandwidth *cfs_b, int overrun, unsigned long flags)
7481 	__must_hold(&cfs_b->lock)
7482 {
7483 	int throttled;
7484 
7485 	/* no need to continue the timer with no bandwidth constraint */
7486 	if (cfs_b->quota == RUNTIME_INF)
7487 		goto out_deactivate;
7488 
7489 	throttled = !list_empty(&cfs_b->throttled_cfs_rq);
7490 	cfs_b->nr_periods += overrun;
7491 
7492 	/* Refill extra burst quota even if cfs_b->idle */
7493 	__refill_cfs_bandwidth_runtime(cfs_b);
7494 
7495 	/*
7496 	 * idle depends on !throttled (for the case of a large deficit), and if
7497 	 * we're going inactive then everything else can be deferred
7498 	 */
7499 	if (cfs_b->idle && !throttled)
7500 		goto out_deactivate;
7501 
7502 	if (!throttled) {
7503 		/* mark as potentially idle for the upcoming period */
7504 		cfs_b->idle = 1;
7505 		return 0;
7506 	}
7507 
7508 	/* account preceding periods in which throttling occurred */
7509 	cfs_b->nr_throttled += overrun;
7510 
7511 	/*
7512 	 * This check is repeated as we release cfs_b->lock while we unthrottle.
7513 	 */
7514 	while (throttled && cfs_b->runtime > 0) {
7515 		raw_spin_unlock_irqrestore(&cfs_b->lock, flags);
7516 		/* we can't nest cfs_b->lock while distributing bandwidth */
7517 		throttled = distribute_cfs_runtime(cfs_b);
7518 		raw_spin_lock_irqsave(&cfs_b->lock, flags);
7519 	}
7520 
7521 	/*
7522 	 * While we are ensured activity in the period following an
7523 	 * unthrottle, this also covers the case in which the new bandwidth is
7524 	 * insufficient to cover the existing bandwidth deficit.  (Forcing the
7525 	 * timer to remain active while there are any throttled entities.)
7526 	 */
7527 	cfs_b->idle = 0;
7528 
7529 	return 0;
7530 
7531 out_deactivate:
7532 	return 1;
7533 }
7534 
7535 /* a cfs_rq won't donate quota below this amount */
7536 static const u64 min_cfs_rq_runtime = 1 * NSEC_PER_MSEC;
7537 /* minimum remaining period time to redistribute slack quota */
7538 static const u64 min_bandwidth_expiration = 2 * NSEC_PER_MSEC;
7539 /* how long we wait to gather additional slack before distributing */
7540 static const u64 cfs_bandwidth_slack_period = 5 * NSEC_PER_MSEC;
7541 
7542 /*
7543  * Are we near the end of the current quota period?
7544  *
7545  * Requires cfs_b->lock for hrtimer_expires_remaining to be safe against the
7546  * hrtimer base being cleared by hrtimer_start. In the case of
7547  * migrate_hrtimers, base is never cleared, so we are fine.
7548  */
7549 static int runtime_refresh_within(struct cfs_bandwidth *cfs_b, u64 min_expire)
7550 {
7551 	struct hrtimer *refresh_timer = &cfs_b->period_timer;
7552 	s64 remaining;
7553 
7554 	/* if the call-back is running a quota refresh is already occurring */
7555 	if (hrtimer_callback_running(refresh_timer))
7556 		return 1;
7557 
7558 	/* is a quota refresh about to occur? */
7559 	remaining = ktime_to_ns(hrtimer_expires_remaining(refresh_timer));
7560 	if (remaining < (s64)min_expire)
7561 		return 1;
7562 
7563 	return 0;
7564 }
7565 
7566 static void start_cfs_slack_bandwidth(struct cfs_bandwidth *cfs_b)
7567 {
7568 	u64 min_left = cfs_bandwidth_slack_period + min_bandwidth_expiration;
7569 
7570 	/* if there's a quota refresh soon don't bother with slack */
7571 	if (runtime_refresh_within(cfs_b, min_left))
7572 		return;
7573 
7574 	/* don't push forwards an existing deferred unthrottle */
7575 	if (cfs_b->slack_started)
7576 		return;
7577 	cfs_b->slack_started = true;
7578 
7579 	hrtimer_start(&cfs_b->slack_timer,
7580 			ns_to_ktime(cfs_bandwidth_slack_period),
7581 			HRTIMER_MODE_REL);
7582 }
7583 
7584 /* we know any runtime found here is valid as update_curr() precedes return */
7585 static void __return_cfs_rq_runtime(struct cfs_rq *cfs_rq)
7586 {
7587 	struct cfs_bandwidth *cfs_b = tg_cfs_bandwidth(cfs_rq->tg);
7588 	s64 slack_runtime = cfs_rq->runtime_remaining - min_cfs_rq_runtime;
7589 
7590 	if (slack_runtime <= 0)
7591 		return;
7592 
7593 	guard(raw_spinlock)(&cfs_b->lock);
7594 
7595 	if (cfs_b->quota != RUNTIME_INF) {
7596 		cfs_b->runtime += slack_runtime;
7597 
7598 		/* we are under rq->lock, defer unthrottling using a timer */
7599 		if (cfs_b->runtime > sched_cfs_bandwidth_slice() &&
7600 		    !list_empty(&cfs_b->throttled_cfs_rq))
7601 			start_cfs_slack_bandwidth(cfs_b);
7602 	}
7603 
7604 	/* even if it's not valid for return we don't want to try again */
7605 	cfs_rq->runtime_remaining -= slack_runtime;
7606 }
7607 
7608 static __always_inline void return_cfs_rq_runtime(struct cfs_rq *cfs_rq)
7609 {
7610 	if (!cfs_bandwidth_used())
7611 		return;
7612 
7613 	if (!cfs_rq->runtime_enabled || cfs_rq->nr_queued)
7614 		return;
7615 
7616 	__return_cfs_rq_runtime(cfs_rq);
7617 }
7618 
7619 /*
7620  * This is done with a timer (instead of inline with bandwidth return) since
7621  * it's necessary to juggle rq->locks to unthrottle their respective cfs_rqs.
7622  */
7623 static void do_sched_cfs_slack_timer(struct cfs_bandwidth *cfs_b)
7624 {
7625 	/* confirm we're still not at a refresh boundary */
7626 	scoped_guard(raw_spinlock_irqsave, &cfs_b->lock) {
7627 		u64 runtime = 0, slice = sched_cfs_bandwidth_slice();
7628 
7629 		cfs_b->slack_started = false;
7630 
7631 		if (runtime_refresh_within(cfs_b, min_bandwidth_expiration))
7632 			return;
7633 
7634 		if (cfs_b->quota != RUNTIME_INF && cfs_b->runtime > slice)
7635 			runtime = cfs_b->runtime;
7636 
7637 		if (!runtime)
7638 			return;
7639 	}
7640 
7641 	distribute_cfs_runtime(cfs_b);
7642 }
7643 
7644 /*
7645  * When a group wakes up we want to make sure that its quota is not already
7646  * expired/exceeded, otherwise it may be allowed to steal additional ticks of
7647  * runtime as update_curr() throttling can not trigger until it's on-rq.
7648  */
7649 static void check_enqueue_throttle(struct cfs_rq *cfs_rq)
7650 {
7651 	if (!cfs_bandwidth_used())
7652 		return;
7653 
7654 	/* an active group must be handled by the update_curr() path */
7655 	if (!cfs_rq->runtime_enabled || cfs_rq->h_curr)
7656 		return;
7657 
7658 	/* ensure the group is not already throttled */
7659 	if (cfs_rq_throttled(cfs_rq))
7660 		return;
7661 
7662 	/* update runtime allocation */
7663 	account_cfs_rq_runtime(cfs_rq, 0);
7664 }
7665 
7666 static void sync_throttle(struct task_group *tg, int cpu)
7667 {
7668 	struct cfs_rq *pcfs_rq, *cfs_rq;
7669 
7670 	if (!cfs_bandwidth_used())
7671 		return;
7672 
7673 	if (!tg->parent)
7674 		return;
7675 
7676 	cfs_rq = tg_cfs_rq(tg, cpu);
7677 	pcfs_rq = tg_cfs_rq(tg->parent, cpu);
7678 
7679 	cfs_rq->throttle_count = pcfs_rq->throttle_count;
7680 	cfs_rq->throttled_clock_pelt = rq_clock_pelt(cpu_rq(cpu));
7681 
7682 	/*
7683 	 * It is not enough to sync the "pelt_clock_throttled" indicator
7684 	 * with the parent cfs_rq when the hierarchy is not queued.
7685 	 * Always join a throttled hierarchy with PELT clock throttled
7686 	 * and leaf it to the first enqueue, or distribution to
7687 	 * unthrottle the PELT clock.
7688 	 */
7689 	if (cfs_rq->throttle_count)
7690 		cfs_rq->pelt_clock_throttled = 1;
7691 }
7692 
7693 static enum hrtimer_restart sched_cfs_slack_timer(struct hrtimer *timer)
7694 {
7695 	struct cfs_bandwidth *cfs_b =
7696 		container_of(timer, struct cfs_bandwidth, slack_timer);
7697 
7698 	do_sched_cfs_slack_timer(cfs_b);
7699 
7700 	return HRTIMER_NORESTART;
7701 }
7702 
7703 static enum hrtimer_restart sched_cfs_period_timer(struct hrtimer *timer)
7704 {
7705 	struct cfs_bandwidth *cfs_b =
7706 		container_of(timer, struct cfs_bandwidth, period_timer);
7707 	int overrun;
7708 	int idle = 0;
7709 	int count = 0;
7710 
7711 	CLASS(raw_spinlock_irqsave, cfsb_guard)(&cfs_b->lock);
7712 
7713 	for (;;) {
7714 		overrun = hrtimer_forward_now(timer, cfs_b->period);
7715 		if (!overrun)
7716 			break;
7717 
7718 		idle = do_sched_cfs_period_timer(cfs_b, overrun, cfsb_guard.flags);
7719 
7720 		if (++count > 3) {
7721 			u64 new, old = ktime_to_ns(cfs_b->period);
7722 
7723 			/*
7724 			 * Grow period by a factor of 2 to avoid losing precision.
7725 			 * Precision loss in the quota/period ratio can cause __cfs_schedulable
7726 			 * to fail.
7727 			 */
7728 			new = old * 2;
7729 			if (new < max_bw_quota_period_us * NSEC_PER_USEC) {
7730 				cfs_b->period = ns_to_ktime(new);
7731 				cfs_b->quota *= 2;
7732 				cfs_b->burst *= 2;
7733 
7734 				pr_warn_ratelimited(
7735 	"cfs_period_timer[cpu%d]: period too short, scaling up (new cfs_period_us = %lld, cfs_quota_us = %lld)\n",
7736 					smp_processor_id(),
7737 					div_u64(new, NSEC_PER_USEC),
7738 					div_u64(cfs_b->quota, NSEC_PER_USEC));
7739 			} else {
7740 				pr_warn_ratelimited(
7741 	"cfs_period_timer[cpu%d]: period too short, but cannot scale up without losing precision (cfs_period_us = %lld, cfs_quota_us = %lld)\n",
7742 					smp_processor_id(),
7743 					div_u64(old, NSEC_PER_USEC),
7744 					div_u64(cfs_b->quota, NSEC_PER_USEC));
7745 			}
7746 
7747 			/* reset count so we don't come right back in here */
7748 			count = 0;
7749 		}
7750 	}
7751 
7752 	if (idle) {
7753 		cfs_b->period_active = 0;
7754 		return HRTIMER_NORESTART;
7755 	}
7756 
7757 	return HRTIMER_RESTART;
7758 }
7759 
7760 void init_cfs_bandwidth(struct cfs_bandwidth *cfs_b, struct cfs_bandwidth *parent)
7761 {
7762 	raw_spin_lock_init(&cfs_b->lock);
7763 	cfs_b->runtime = 0;
7764 	cfs_b->quota = RUNTIME_INF;
7765 	cfs_b->period = us_to_ktime(default_bw_period_us());
7766 	cfs_b->burst = 0;
7767 	cfs_b->hierarchical_quota = parent ? parent->hierarchical_quota : RUNTIME_INF;
7768 
7769 	INIT_LIST_HEAD(&cfs_b->throttled_cfs_rq);
7770 	hrtimer_setup(&cfs_b->period_timer, sched_cfs_period_timer, CLOCK_MONOTONIC,
7771 		      HRTIMER_MODE_ABS_PINNED);
7772 
7773 	/* Add a random offset so that timers interleave */
7774 	hrtimer_set_expires(&cfs_b->period_timer,
7775 			    get_random_u32_below(cfs_b->period));
7776 	hrtimer_setup(&cfs_b->slack_timer, sched_cfs_slack_timer, CLOCK_MONOTONIC,
7777 		      HRTIMER_MODE_REL);
7778 	cfs_b->slack_started = false;
7779 }
7780 
7781 static void init_cfs_rq_runtime(struct cfs_rq *cfs_rq)
7782 {
7783 	cfs_rq->runtime_enabled = 0;
7784 	INIT_LIST_HEAD(&cfs_rq->throttled_list);
7785 	INIT_LIST_HEAD(&cfs_rq->throttled_csd_list);
7786 	INIT_LIST_HEAD(&cfs_rq->throttled_limbo_list);
7787 }
7788 
7789 void start_cfs_bandwidth(struct cfs_bandwidth *cfs_b)
7790 {
7791 	lockdep_assert_held(&cfs_b->lock);
7792 
7793 	if (cfs_b->period_active)
7794 		return;
7795 
7796 	cfs_b->period_active = 1;
7797 	hrtimer_forward_now(&cfs_b->period_timer, cfs_b->period);
7798 	hrtimer_start_expires(&cfs_b->period_timer, HRTIMER_MODE_ABS_PINNED);
7799 }
7800 
7801 static void destroy_cfs_bandwidth(struct cfs_bandwidth *cfs_b)
7802 {
7803 	int __maybe_unused i;
7804 
7805 	/* init_cfs_bandwidth() was not called */
7806 	if (!cfs_b->throttled_cfs_rq.next)
7807 		return;
7808 
7809 	hrtimer_cancel(&cfs_b->period_timer);
7810 	hrtimer_cancel(&cfs_b->slack_timer);
7811 
7812 	/*
7813 	 * It is possible that we still have some cfs_rq's pending on a CSD
7814 	 * list, though this race is very rare. In order for this to occur, we
7815 	 * must have raced with the last task leaving the group while there
7816 	 * exist throttled cfs_rq(s), and the period_timer must have queued the
7817 	 * CSD item but the remote cpu has not yet processed it. To handle this,
7818 	 * we can simply flush all pending CSD work inline here. We're
7819 	 * guaranteed at this point that no additional cfs_rq of this group can
7820 	 * join a CSD list.
7821 	 */
7822 	for_each_possible_cpu(i) {
7823 		struct rq *rq = cpu_rq(i);
7824 
7825 		if (list_empty(&rq->cfsb_csd_list))
7826 			continue;
7827 
7828 		scoped_guard(irqsave)
7829 			__cfsb_csd_unthrottle(rq);
7830 	}
7831 }
7832 
7833 /*
7834  * Both these CPU hotplug callbacks race against unregister_fair_sched_group()
7835  *
7836  * The race is harmless, since modifying bandwidth settings of unhooked group
7837  * bits doesn't do much.
7838  */
7839 
7840 /* cpu online callback */
7841 static void __maybe_unused update_runtime_enabled(struct rq *rq)
7842 {
7843 	struct task_group *tg;
7844 
7845 	lockdep_assert_rq_held(rq);
7846 
7847 	guard(rcu)();
7848 
7849 	list_for_each_entry_rcu(tg, &task_groups, list) {
7850 		struct cfs_bandwidth *cfs_b = &tg->cfs_bandwidth;
7851 		struct cfs_rq *cfs_rq = tg_cfs_rq(tg, cpu_of(rq));
7852 
7853 		scoped_guard(raw_spinlock, &cfs_b->lock)
7854 			cfs_rq->runtime_enabled = cfs_b->quota != RUNTIME_INF;
7855 	}
7856 }
7857 
7858 /* cpu offline callback */
7859 static void __maybe_unused unthrottle_offline_cfs_rqs(struct rq *rq)
7860 {
7861 	struct task_group *tg;
7862 
7863 	lockdep_assert_rq_held(rq);
7864 
7865 	// Do not unthrottle for an active CPU
7866 	if (cpumask_test_cpu(cpu_of(rq), cpu_active_mask))
7867 		return;
7868 
7869 	/*
7870 	 * The rq clock has already been updated in the
7871 	 * set_rq_offline(), so we should skip updating
7872 	 * the rq clock again in unthrottle_cfs_rq().
7873 	 */
7874 	rq_clock_start_loop_update(rq);
7875 
7876 	guard(rcu)();
7877 
7878 	list_for_each_entry_rcu(tg, &task_groups, list) {
7879 		struct cfs_rq *cfs_rq = tg_cfs_rq(tg, cpu_of(rq));
7880 
7881 		if (!cfs_rq->runtime_enabled)
7882 			continue;
7883 
7884 		/*
7885 		 * Offline rq is schedulable till CPU is completely disabled
7886 		 * in take_cpu_down(), so we prevent new cfs throttling here.
7887 		 */
7888 		cfs_rq->runtime_enabled = 0;
7889 
7890 		if (!cfs_rq_throttled(cfs_rq))
7891 			continue;
7892 
7893 		/*
7894 		 * clock_task is not advancing so we just need to make sure
7895 		 * there's some valid quota amount
7896 		 */
7897 		cfs_rq->runtime_remaining = 1;
7898 		unthrottle_cfs_rq(cfs_rq);
7899 	}
7900 
7901 	rq_clock_stop_loop_update(rq);
7902 }
7903 
7904 bool cfs_task_bw_constrained(struct task_struct *p)
7905 {
7906 	struct cfs_rq *cfs_rq = task_cfs_rq(p);
7907 
7908 	if (!cfs_bandwidth_used())
7909 		return false;
7910 
7911 	if (cfs_rq->runtime_enabled ||
7912 	    tg_cfs_bandwidth(cfs_rq->tg)->hierarchical_quota != RUNTIME_INF)
7913 		return true;
7914 
7915 	return false;
7916 }
7917 
7918 #ifdef CONFIG_NO_HZ_FULL
7919 /* called from pick_next_task_fair() */
7920 static void sched_fair_update_stop_tick(struct rq *rq, struct task_struct *p)
7921 {
7922 	int cpu = cpu_of(rq);
7923 
7924 	if (!cfs_bandwidth_used())
7925 		return;
7926 
7927 	if (!tick_nohz_full_cpu(cpu))
7928 		return;
7929 
7930 	if (rq->nr_running != 1)
7931 		return;
7932 
7933 	/*
7934 	 *  We know there is only one task runnable and we've just picked it. The
7935 	 *  normal enqueue path will have cleared TICK_DEP_BIT_SCHED if we will
7936 	 *  be otherwise able to stop the tick. Just need to check if we are using
7937 	 *  bandwidth control.
7938 	 */
7939 	if (cfs_task_bw_constrained(p))
7940 		tick_nohz_dep_set_cpu(cpu, TICK_DEP_BIT_SCHED);
7941 }
7942 #endif /* CONFIG_NO_HZ_FULL */
7943 
7944 #else /* !CONFIG_CFS_BANDWIDTH: */
7945 
7946 static bool account_cfs_rq_runtime(struct cfs_rq *cfs_rq, u64 delta_exec) { return false; }
7947 static void check_enqueue_throttle(struct cfs_rq *cfs_rq) {}
7948 static inline void sync_throttle(struct task_group *tg, int cpu) {}
7949 static __always_inline void return_cfs_rq_runtime(struct cfs_rq *cfs_rq) {}
7950 static void task_throttle_setup_work(struct task_struct *p) {}
7951 static bool task_is_throttled(struct task_struct *p) { return false; }
7952 static void dequeue_throttled_task(struct task_struct *p, int flags) {}
7953 static bool enqueue_throttled_task(struct task_struct *p) { return false; }
7954 static void record_throttle_clock(struct cfs_rq *cfs_rq) {}
7955 
7956 static inline int cfs_rq_throttled(struct cfs_rq *cfs_rq)
7957 {
7958 	return 0;
7959 }
7960 
7961 static inline bool cfs_rq_pelt_clock_throttled(struct cfs_rq *cfs_rq)
7962 {
7963 	return false;
7964 }
7965 
7966 static inline int throttled_hierarchy(struct cfs_rq *cfs_rq)
7967 {
7968 	return 0;
7969 }
7970 
7971 static inline int lb_throttled_hierarchy(struct task_struct *p, int dst_cpu)
7972 {
7973 	return 0;
7974 }
7975 
7976 #ifdef CONFIG_FAIR_GROUP_SCHED
7977 void init_cfs_bandwidth(struct cfs_bandwidth *cfs_b, struct cfs_bandwidth *parent) {}
7978 static void init_cfs_rq_runtime(struct cfs_rq *cfs_rq) {}
7979 #endif
7980 
7981 static inline struct cfs_bandwidth *tg_cfs_bandwidth(struct task_group *tg)
7982 {
7983 	return NULL;
7984 }
7985 static inline void destroy_cfs_bandwidth(struct cfs_bandwidth *cfs_b) {}
7986 static inline void update_runtime_enabled(struct rq *rq) {}
7987 static inline void unthrottle_offline_cfs_rqs(struct rq *rq) {}
7988 #ifdef CONFIG_CGROUP_SCHED
7989 bool cfs_task_bw_constrained(struct task_struct *p)
7990 {
7991 	return false;
7992 }
7993 #endif
7994 #endif /* !CONFIG_CFS_BANDWIDTH */
7995 
7996 #if !defined(CONFIG_CFS_BANDWIDTH) || !defined(CONFIG_NO_HZ_FULL)
7997 static inline void sched_fair_update_stop_tick(struct rq *rq, struct task_struct *p) {}
7998 #endif
7999 
8000 /**************************************************
8001  * CFS operations on tasks:
8002  */
8003 
8004 #ifdef CONFIG_SCHED_HRTICK
8005 static void hrtick_start_fair(struct rq *rq, struct task_struct *p)
8006 {
8007 	struct sched_entity *se = &p->se;
8008 	unsigned long scale = 1024;
8009 	unsigned long util = 0;
8010 	u64 vdelta;
8011 	u64 delta;
8012 
8013 	WARN_ON_ONCE(task_rq(p) != rq);
8014 
8015 	if (rq->cfs.h_nr_queued <= 1)
8016 		return;
8017 
8018 	/*
8019 	 * Compute time until virtual deadline
8020 	 */
8021 	vdelta = se->deadline - se->vruntime;
8022 	if ((s64)vdelta < 0) {
8023 		if (task_current_donor(rq, p))
8024 			resched_curr(rq);
8025 		return;
8026 	}
8027 	delta = (se->h_load.weight * vdelta) / NICE_0_LOAD;
8028 
8029 	/*
8030 	 * Correct for instantaneous load of other classes.
8031 	 */
8032 	util += cpu_util_irq(rq);
8033 	if (util && util < 1024) {
8034 		scale *= 1024;
8035 		scale /= (1024 - util);
8036 	}
8037 
8038 	hrtick_start(rq, (scale * delta) / 1024);
8039 }
8040 
8041 /*
8042  * Called on enqueue to start the hrtick when h_nr_queued becomes more than 1.
8043  */
8044 static void hrtick_update(struct rq *rq)
8045 {
8046 	struct task_struct *donor = rq->donor;
8047 
8048 	if (!hrtick_enabled_fair(rq) || donor->sched_class != &fair_sched_class)
8049 		return;
8050 
8051 	if (hrtick_active(rq))
8052 		return;
8053 
8054 	hrtick_start_fair(rq, donor);
8055 }
8056 #else /* !CONFIG_SCHED_HRTICK: */
8057 static inline void
8058 hrtick_start_fair(struct rq *rq, struct task_struct *p)
8059 {
8060 }
8061 
8062 static inline void hrtick_update(struct rq *rq)
8063 {
8064 }
8065 #endif /* !CONFIG_SCHED_HRTICK */
8066 
8067 static inline bool cpu_overutilized(int cpu)
8068 {
8069 	unsigned long rq_util_max;
8070 
8071 	if (!sched_energy_enabled())
8072 		return false;
8073 
8074 	rq_util_max = uclamp_rq_get(cpu_rq(cpu), UCLAMP_MAX);
8075 
8076 	/* Return true only if the utilization doesn't fit CPU's capacity */
8077 	return !util_fits_cpu(cpu_util_cfs(cpu), 0, rq_util_max, cpu);
8078 }
8079 
8080 /*
8081  * overutilized value make sense only if EAS is enabled
8082  */
8083 static inline bool is_rd_overutilized(struct root_domain *rd)
8084 {
8085 	return !sched_energy_enabled() || READ_ONCE(rd->overutilized);
8086 }
8087 
8088 static inline void set_rd_overutilized(struct root_domain *rd, bool flag)
8089 {
8090 	if (!sched_energy_enabled())
8091 		return;
8092 
8093 	WRITE_ONCE(rd->overutilized, flag);
8094 	trace_sched_overutilized_tp(rd, flag);
8095 }
8096 
8097 static inline void check_update_overutilized_status(struct rq *rq)
8098 {
8099 	/*
8100 	 * overutilized field is used for load balancing decisions only
8101 	 * if energy aware scheduler is being used
8102 	 */
8103 
8104 	if (!is_rd_overutilized(rq->rd) && cpu_overutilized(rq->cpu))
8105 		set_rd_overutilized(rq->rd, 1);
8106 }
8107 
8108 /* Runqueue only has SCHED_IDLE tasks enqueued */
8109 static int sched_idle_rq(struct rq *rq)
8110 {
8111 	return unlikely(rq->nr_running == rq->cfs.h_nr_idle &&
8112 			rq->nr_running);
8113 }
8114 
8115 static int choose_sched_idle_rq(struct rq *rq, struct task_struct *p)
8116 {
8117 	return sched_idle_rq(rq) && !task_has_idle_policy(p);
8118 }
8119 
8120 static int choose_idle_cpu(int cpu, struct task_struct *p)
8121 {
8122 	return available_idle_cpu(cpu) ||
8123 	       choose_sched_idle_rq(cpu_rq(cpu), p);
8124 }
8125 
8126 static void
8127 requeue_delayed_entity(struct cfs_rq *cfs_rq, struct sched_entity *se)
8128 {
8129 	/*
8130 	 * se->sched_delayed should imply: se->on_rq == 1.
8131 	 * Because a delayed entity is one that is still on
8132 	 * the runqueue competing until elegibility.
8133 	 */
8134 	WARN_ON_ONCE(!se->sched_delayed);
8135 	WARN_ON_ONCE(!se->on_rq);
8136 
8137 	if (update_entity_lag(cfs_rq, se)) {
8138 		cfs_rq->h_nr_queued--;
8139 		if (se != cfs_rq->curr)
8140 			__dequeue_entity(cfs_rq, se);
8141 		place_entity(cfs_rq, se, 0);
8142 		if (se != cfs_rq->curr)
8143 			__enqueue_entity(cfs_rq, se);
8144 		cfs_rq->h_nr_queued++;
8145 	}
8146 
8147 	update_load_avg(cfs_rq, se, 0);
8148 	clear_delayed(se);
8149 }
8150 
8151 static unsigned long enqueue_hierarchy(struct task_struct *p, int flags)
8152 {
8153 	unsigned long weight = NICE_0_LOAD;
8154 	int task_new = !(flags & ENQUEUE_WAKEUP);
8155 	struct sched_entity *se = &p->se;
8156 	int h_nr_idle = task_has_idle_policy(p);
8157 	int h_nr_runnable = 1;
8158 
8159 	if (task_new && se->sched_delayed)
8160 		h_nr_runnable = 0;
8161 
8162 	for_each_sched_entity(se) {
8163 		struct cfs_rq *cfs_rq = cfs_rq_of(se);
8164 
8165 		update_curr(cfs_rq);
8166 
8167 		if (!se->on_rq) {
8168 			enqueue_entity(cfs_rq, se, flags);
8169 		} else {
8170 			update_load_avg(cfs_rq, se, UPDATE_TG);
8171 			se_update_runnable(se);
8172 			update_cfs_group(se);
8173 		}
8174 
8175 		cfs_rq->h_nr_runnable += h_nr_runnable;
8176 		cfs_rq->h_nr_queued++;
8177 		cfs_rq->h_nr_idle += h_nr_idle;
8178 
8179 		if (cfs_rq_is_idle(cfs_rq))
8180 			h_nr_idle = 1;
8181 
8182 		weight = __calc_prop_weight(cfs_rq, se, weight);
8183 
8184 		flags = ENQUEUE_WAKEUP;
8185 	}
8186 
8187 	return weight;
8188 }
8189 
8190 /* Update curr's vruntime before placing entity or updating lag */
8191 static inline void update_curr_eevdf(struct cfs_rq *cfs_rq)
8192 {
8193 	if (!cfs_rq->curr)
8194 		return;
8195 
8196 	update_curr(cfs_rq_of(cfs_rq->curr));
8197 }
8198 
8199 /*
8200  * The enqueue_task method is called before nr_running is
8201  * increased. Here we update the fair scheduling stats and
8202  * then put the task into the rbtree:
8203  */
8204 static void
8205 enqueue_task_fair(struct rq *rq, struct task_struct *p, int flags)
8206 {
8207 	int rq_h_nr_queued = rq->cfs.h_nr_queued;
8208 	int task_new = !(flags & ENQUEUE_WAKEUP);
8209 	struct sched_entity *se = &p->se;
8210 	struct cfs_rq *cfs_rq = &rq->cfs;
8211 	unsigned long weight;
8212 	bool curr;
8213 
8214 	if (task_is_throttled(p) && enqueue_throttled_task(p))
8215 		return;
8216 
8217 	/*
8218 	 * The code below (indirectly) updates schedutil which looks at
8219 	 * the cfs_rq utilization to select a frequency.
8220 	 * Let's add the task's estimated utilization to the cfs_rq's
8221 	 * estimated utilization, before we update schedutil.
8222 	 */
8223 	if (!p->se.sched_delayed || (flags & ENQUEUE_DELAYED))
8224 		util_est_enqueue(cfs_rq, p);
8225 
8226 	update_curr_eevdf(cfs_rq);
8227 
8228 	if (flags & ENQUEUE_DELAYED) {
8229 		requeue_delayed_entity(cfs_rq, se);
8230 		return;
8231 	}
8232 
8233 	/*
8234 	 * If in_iowait is set, the code below may not trigger any cpufreq
8235 	 * utilization updates, so do it here explicitly with the IOWAIT flag
8236 	 * passed.
8237 	 */
8238 	if (p->in_iowait)
8239 		cpufreq_update_util(rq, SCHED_CPUFREQ_IOWAIT);
8240 
8241 	/*
8242 	 * XXX comment on the curr thing
8243 	 */
8244 	curr = (cfs_rq->curr == se);
8245 	if (curr)
8246 		place_entity(cfs_rq, se, flags);
8247 
8248 	if (se->on_rq && se->sched_delayed)
8249 		requeue_delayed_entity(cfs_rq, se);
8250 
8251 	weight = enqueue_hierarchy(p, flags);
8252 
8253 	if (!curr) {
8254 		reweight_eevdf(cfs_rq, se, weight, false);
8255 		place_entity(cfs_rq, se, flags | ENQUEUE_QUEUED);
8256 		__enqueue_entity(cfs_rq, se);
8257 	}
8258 
8259 	if (!rq_h_nr_queued && rq->cfs.h_nr_queued)
8260 		dl_server_start(&rq->fair_server);
8261 
8262 	/* At this point se is NULL and we are at root level*/
8263 	add_nr_running(rq, 1);
8264 
8265 	/*
8266 	 * Since new tasks are assigned an initial util_avg equal to
8267 	 * half of the spare capacity of their CPU, tiny tasks have the
8268 	 * ability to cross the overutilized threshold, which will
8269 	 * result in the load balancer ruining all the task placement
8270 	 * done by EAS. As a way to mitigate that effect, do not account
8271 	 * for the first enqueue operation of new tasks during the
8272 	 * overutilized flag detection.
8273 	 *
8274 	 * A better way of solving this problem would be to wait for
8275 	 * the PELT signals of tasks to converge before taking them
8276 	 * into account, but that is not straightforward to implement,
8277 	 * and the following generally works well enough in practice.
8278 	 */
8279 	if (!task_new)
8280 		check_update_overutilized_status(rq);
8281 
8282 	assert_list_leaf_cfs_rq(rq);
8283 
8284 	hrtick_update(rq);
8285 }
8286 
8287 static void dequeue_hierarchy(struct task_struct *p, int flags)
8288 {
8289 	struct sched_entity *se = &p->se;
8290 	bool task_sleep = flags & DEQUEUE_SLEEP;
8291 	bool task_delayed = flags & DEQUEUE_DELAYED;
8292 	bool task_throttled = flags & DEQUEUE_THROTTLE;
8293 	int h_nr_runnable = 0;
8294 	int h_nr_idle = task_has_idle_policy(p);
8295 	bool dequeue = true;
8296 
8297 	if (task_sleep || task_delayed || !se->sched_delayed)
8298 		h_nr_runnable = 1;
8299 
8300 	for_each_sched_entity(se) {
8301 		struct cfs_rq *cfs_rq = cfs_rq_of(se);
8302 
8303 		update_curr(cfs_rq);
8304 
8305 		if (dequeue) {
8306 			dequeue_entity(cfs_rq, se, flags);
8307 			/* Don't dequeue parent if it has other entities besides us */
8308 			if (cfs_rq->load.weight)
8309 				dequeue = false;
8310 		} else {
8311 			update_load_avg(cfs_rq, se, UPDATE_TG);
8312 			se_update_runnable(se);
8313 			update_cfs_group(se);
8314 		}
8315 
8316 		cfs_rq->h_nr_runnable -= h_nr_runnable;
8317 		cfs_rq->h_nr_queued--;
8318 		cfs_rq->h_nr_idle -= h_nr_idle;
8319 
8320 		if (cfs_rq_is_idle(cfs_rq))
8321 			h_nr_idle = 1;
8322 
8323 		if (throttled_hierarchy(cfs_rq) && task_throttled)
8324 			record_throttle_clock(cfs_rq);
8325 
8326 		flags |= DEQUEUE_SLEEP;
8327 		flags &= ~(DEQUEUE_DELAYED | DEQUEUE_SPECIAL);
8328 	}
8329 }
8330 
8331 /*
8332  * The part of dequeue_task_fair() that is needed to dequeue delayed tasks.
8333  *
8334  * Returns:
8335  *   true  - dequeued
8336  *   false - delayed
8337  */
8338 static bool __dequeue_task(struct rq *rq, struct task_struct *p, int flags)
8339 {
8340 	struct sched_entity *se = &p->se;
8341 	struct cfs_rq *cfs_rq = &rq->cfs;
8342 	bool was_sched_idle = sched_idle_rq(rq);
8343 	bool task_sleep = flags & DEQUEUE_SLEEP;
8344 	bool task_delayed = flags & DEQUEUE_DELAYED;
8345 
8346 	clear_buddies(cfs_rq, se);
8347 
8348 	update_curr_eevdf(cfs_rq);
8349 	update_entity_lag(cfs_rq, se);
8350 
8351 	if (flags & DEQUEUE_DELAYED) {
8352 		WARN_ON_ONCE(!se->sched_delayed);
8353 	} else {
8354 		bool delay = task_sleep;
8355 		/*
8356 		 * DELAY_DEQUEUE relies on spurious wakeups, special task
8357 		 * states must not suffer spurious wakeups, excempt them.
8358 		 */
8359 		if (flags & (DEQUEUE_SPECIAL | DEQUEUE_THROTTLE))
8360 			delay = false;
8361 
8362 		WARN_ON_ONCE(delay && se->sched_delayed);
8363 
8364 		if (sched_feat(DELAY_DEQUEUE) && delay &&
8365 		    !entity_eligible(cfs_rq, se)) {
8366 			update_load_avg(cfs_rq_of(se), se, UPDATE_UTIL_EST);
8367 			set_delayed(se);
8368 			return false;
8369 		}
8370 	}
8371 
8372 	dequeue_hierarchy(p, flags);
8373 
8374 	if (sched_feat(PLACE_REL_DEADLINE) && !task_sleep) {
8375 		se->deadline -= se->vruntime;
8376 		se->rel_deadline = 1;
8377 	}
8378 	if (se != cfs_rq->curr)
8379 		__dequeue_entity(cfs_rq, se);
8380 
8381 	sub_nr_running(rq, 1);
8382 
8383 	/* balance early to pull high priority tasks */
8384 	if (unlikely(!was_sched_idle && sched_idle_rq(rq)))
8385 		rq->next_balance = jiffies;
8386 
8387 	if (task_delayed) {
8388 		clear_delayed(se);
8389 
8390 		WARN_ON_ONCE(!task_sleep);
8391 		WARN_ON_ONCE(p->on_rq != 1);
8392 
8393 		/*
8394 		 * Fix-up what block_task() skipped.
8395 		 *
8396 		 * Must be last, @p might not be valid after this.
8397 		 */
8398 		__block_task(rq, p);
8399 	}
8400 
8401 	return true;
8402 }
8403 
8404 /*
8405  * The dequeue_task method is called before nr_running is
8406  * decreased. We remove the task from the rbtree and
8407  * update the fair scheduling stats:
8408  */
8409 static bool dequeue_task_fair(struct rq *rq, struct task_struct *p, int flags)
8410 {
8411 	if (task_is_throttled(p)) {
8412 		dequeue_throttled_task(p, flags);
8413 		return true;
8414 	}
8415 
8416 	if (!p->se.sched_delayed)
8417 		util_est_dequeue(&rq->cfs, p);
8418 
8419 	if (!__dequeue_task(rq, p, flags))
8420 		return false;
8421 
8422 	/*
8423 	 * Must not reference @p after __dequeue_task(DEQUEUE_DELAYED).
8424 	 */
8425 	return true;
8426 }
8427 
8428 static inline unsigned int cfs_h_nr_delayed(struct rq *rq)
8429 {
8430 	return (rq->cfs.h_nr_queued - rq->cfs.h_nr_runnable);
8431 }
8432 
8433 /* Working cpumask for: sched_balance_rq(), sched_balance_newidle(). */
8434 static DEFINE_PER_CPU(cpumask_var_t, load_balance_mask);
8435 static DEFINE_PER_CPU(cpumask_var_t, select_rq_mask);
8436 static DEFINE_PER_CPU(cpumask_var_t, should_we_balance_tmpmask);
8437 
8438 #ifdef CONFIG_NO_HZ_COMMON
8439 
8440 static struct {
8441 	cpumask_var_t idle_cpus_mask;
8442 	int has_blocked_load;		/* Idle CPUS has blocked load */
8443 	int needs_update;		/* Newly idle CPUs need their next_balance collated */
8444 	unsigned long next_balance;     /* in jiffy units */
8445 	unsigned long next_blocked;	/* Next update of blocked load in jiffies */
8446 } nohz ____cacheline_aligned;
8447 
8448 #endif /* CONFIG_NO_HZ_COMMON */
8449 
8450 static unsigned long cpu_load(struct rq *rq)
8451 {
8452 	return cfs_rq_load_avg(&rq->cfs);
8453 }
8454 
8455 /*
8456  * cpu_load_without - compute CPU load without any contributions from *p
8457  * @cpu: the CPU which load is requested
8458  * @p: the task which load should be discounted
8459  *
8460  * The load of a CPU is defined by the load of tasks currently enqueued on that
8461  * CPU as well as tasks which are currently sleeping after an execution on that
8462  * CPU.
8463  *
8464  * This method returns the load of the specified CPU by discounting the load of
8465  * the specified task, whenever the task is currently contributing to the CPU
8466  * load.
8467  */
8468 static unsigned long cpu_load_without(struct rq *rq, struct task_struct *p)
8469 {
8470 	struct cfs_rq *cfs_rq;
8471 	unsigned int load;
8472 
8473 	/* Task has no contribution or is new */
8474 	if (cpu_of(rq) != task_cpu(p) || !READ_ONCE(p->se.avg.last_update_time))
8475 		return cpu_load(rq);
8476 
8477 	cfs_rq = &rq->cfs;
8478 	load = READ_ONCE(cfs_rq->avg.load_avg);
8479 
8480 	/* Discount task's util from CPU's util */
8481 	lsub_positive(&load, task_h_load(p));
8482 
8483 	return load;
8484 }
8485 
8486 static unsigned long cpu_runnable(struct rq *rq)
8487 {
8488 	return cfs_rq_runnable_avg(&rq->cfs);
8489 }
8490 
8491 static unsigned long cpu_runnable_without(struct rq *rq, struct task_struct *p)
8492 {
8493 	struct cfs_rq *cfs_rq;
8494 	unsigned int runnable;
8495 
8496 	/* Task has no contribution or is new */
8497 	if (cpu_of(rq) != task_cpu(p) || !READ_ONCE(p->se.avg.last_update_time))
8498 		return cpu_runnable(rq);
8499 
8500 	cfs_rq = &rq->cfs;
8501 	runnable = READ_ONCE(cfs_rq->avg.runnable_avg);
8502 
8503 	/* Discount task's runnable from CPU's runnable */
8504 	lsub_positive(&runnable, p->se.avg.runnable_avg);
8505 
8506 	return runnable;
8507 }
8508 
8509 static unsigned long capacity_of(int cpu)
8510 {
8511 	return cpu_rq(cpu)->cpu_capacity;
8512 }
8513 
8514 static void record_wakee(struct task_struct *p)
8515 {
8516 	/*
8517 	 * Only decay a single time; tasks that have less then 1 wakeup per
8518 	 * jiffy will not have built up many flips.
8519 	 */
8520 	if (time_after(jiffies, current->wakee_flip_decay_ts + HZ)) {
8521 		current->wakee_flips >>= 1;
8522 		current->wakee_flip_decay_ts = jiffies;
8523 	}
8524 
8525 	if (current->last_wakee != p) {
8526 		current->last_wakee = p;
8527 		current->wakee_flips++;
8528 	}
8529 }
8530 
8531 /*
8532  * Detect M:N waker/wakee relationships via a switching-frequency heuristic.
8533  *
8534  * A waker of many should wake a different task than the one last awakened
8535  * at a frequency roughly N times higher than one of its wakees.
8536  *
8537  * In order to determine whether we should let the load spread vs consolidating
8538  * to shared cache, we look for a minimum 'flip' frequency of llc_size in one
8539  * partner, and a factor of lls_size higher frequency in the other.
8540  *
8541  * With both conditions met, we can be relatively sure that the relationship is
8542  * non-monogamous, with partner count exceeding socket size.
8543  *
8544  * Waker/wakee being client/server, worker/dispatcher, interrupt source or
8545  * whatever is irrelevant, spread criteria is apparent partner count exceeds
8546  * socket size.
8547  */
8548 static int wake_wide(struct task_struct *p)
8549 {
8550 	unsigned int master = current->wakee_flips;
8551 	unsigned int slave = p->wakee_flips;
8552 	int factor = __this_cpu_read(sd_llc_size);
8553 
8554 	if (master < slave)
8555 		swap(master, slave);
8556 	if (slave < factor || master < slave * factor)
8557 		return 0;
8558 	return 1;
8559 }
8560 
8561 /*
8562  * The purpose of wake_affine() is to quickly determine on which CPU we can run
8563  * soonest. For the purpose of speed we only consider the waking and previous
8564  * CPU.
8565  *
8566  * wake_affine_idle() - only considers 'now', it check if the waking CPU is
8567  *			cache-affine and is (or	will be) idle.
8568  *
8569  * wake_affine_weight() - considers the weight to reflect the average
8570  *			  scheduling latency of the CPUs. This seems to work
8571  *			  for the overloaded case.
8572  */
8573 static int
8574 wake_affine_idle(int this_cpu, int prev_cpu, int sync)
8575 {
8576 	/*
8577 	 * If this_cpu is idle, it implies the wakeup is from interrupt
8578 	 * context. Only allow the move if cache is shared. Otherwise an
8579 	 * interrupt intensive workload could force all tasks onto one
8580 	 * node depending on the IO topology or IRQ affinity settings.
8581 	 *
8582 	 * If the prev_cpu is idle and cache affine then avoid a migration.
8583 	 * There is no guarantee that the cache hot data from an interrupt
8584 	 * is more important than cache hot data on the prev_cpu and from
8585 	 * a cpufreq perspective, it's better to have higher utilisation
8586 	 * on one CPU.
8587 	 */
8588 	if (available_idle_cpu(this_cpu) && cpus_share_cache(this_cpu, prev_cpu))
8589 		return available_idle_cpu(prev_cpu) ? prev_cpu : this_cpu;
8590 
8591 	if (sync) {
8592 		struct rq *rq = cpu_rq(this_cpu);
8593 
8594 		if ((rq->nr_running - cfs_h_nr_delayed(rq)) == 1)
8595 			return this_cpu;
8596 	}
8597 
8598 	if (available_idle_cpu(prev_cpu))
8599 		return prev_cpu;
8600 
8601 	return nr_cpumask_bits;
8602 }
8603 
8604 static int
8605 wake_affine_weight(struct sched_domain *sd, struct task_struct *p,
8606 		   int this_cpu, int prev_cpu, int sync)
8607 {
8608 	s64 this_eff_load, prev_eff_load;
8609 	unsigned long task_load;
8610 
8611 	this_eff_load = cpu_load(cpu_rq(this_cpu));
8612 
8613 	if (sync) {
8614 		unsigned long current_load = task_h_load(current);
8615 
8616 		if (current_load > this_eff_load)
8617 			return this_cpu;
8618 
8619 		this_eff_load -= current_load;
8620 	}
8621 
8622 	task_load = task_h_load(p);
8623 
8624 	this_eff_load += task_load;
8625 	if (sched_feat(WA_BIAS))
8626 		this_eff_load *= 100;
8627 	this_eff_load *= capacity_of(prev_cpu);
8628 
8629 	prev_eff_load = cpu_load(cpu_rq(prev_cpu));
8630 	prev_eff_load -= task_load;
8631 	if (sched_feat(WA_BIAS))
8632 		prev_eff_load *= 100 + (sd->imbalance_pct - 100) / 2;
8633 	prev_eff_load *= capacity_of(this_cpu);
8634 
8635 	/*
8636 	 * If sync, adjust the weight of prev_eff_load such that if
8637 	 * prev_eff == this_eff that select_idle_sibling() will consider
8638 	 * stacking the wakee on top of the waker if no other CPU is
8639 	 * idle.
8640 	 */
8641 	if (sync)
8642 		prev_eff_load += 1;
8643 
8644 	return this_eff_load < prev_eff_load ? this_cpu : nr_cpumask_bits;
8645 }
8646 
8647 static int wake_affine(struct sched_domain *sd, struct task_struct *p,
8648 		       int this_cpu, int prev_cpu, int sync)
8649 {
8650 	int target = nr_cpumask_bits;
8651 
8652 	if (sched_feat(WA_IDLE))
8653 		target = wake_affine_idle(this_cpu, prev_cpu, sync);
8654 
8655 	if (sched_feat(WA_WEIGHT) && target == nr_cpumask_bits)
8656 		target = wake_affine_weight(sd, p, this_cpu, prev_cpu, sync);
8657 
8658 	schedstat_inc(p->stats.nr_wakeups_affine_attempts);
8659 	if (target != this_cpu)
8660 		return prev_cpu;
8661 
8662 	schedstat_inc(sd->ttwu_move_affine);
8663 	schedstat_inc(p->stats.nr_wakeups_affine);
8664 	return target;
8665 }
8666 
8667 static struct sched_group *
8668 sched_balance_find_dst_group(struct sched_domain *sd, struct task_struct *p, int this_cpu);
8669 
8670 /*
8671  * sched_balance_find_dst_group_cpu - find the idlest CPU among the CPUs in the group.
8672  */
8673 static int
8674 sched_balance_find_dst_group_cpu(struct sched_group *group, struct task_struct *p, int this_cpu)
8675 {
8676 	unsigned long load, min_load = ULONG_MAX;
8677 	unsigned int min_exit_latency = UINT_MAX;
8678 	u64 latest_idle_timestamp = 0;
8679 	int least_loaded_cpu = this_cpu;
8680 	int shallowest_idle_cpu = -1;
8681 	int i;
8682 
8683 	/* Check if we have any choice: */
8684 	if (group->group_weight == 1)
8685 		return cpumask_first(sched_group_span(group));
8686 
8687 	/* Traverse only the allowed CPUs */
8688 	for_each_cpu_and(i, sched_group_span(group), p->cpus_ptr) {
8689 		struct rq *rq = cpu_rq(i);
8690 
8691 		if (!sched_core_cookie_match(rq, p))
8692 			continue;
8693 
8694 		if (choose_sched_idle_rq(rq, p))
8695 			return i;
8696 
8697 		if (available_idle_cpu(i)) {
8698 			struct cpuidle_state *idle = idle_get_state(rq);
8699 			if (idle && idle->exit_latency < min_exit_latency) {
8700 				/*
8701 				 * We give priority to a CPU whose idle state
8702 				 * has the smallest exit latency irrespective
8703 				 * of any idle timestamp.
8704 				 */
8705 				min_exit_latency = idle->exit_latency;
8706 				latest_idle_timestamp = rq->idle_stamp;
8707 				shallowest_idle_cpu = i;
8708 			} else if ((!idle || idle->exit_latency == min_exit_latency) &&
8709 				   rq->idle_stamp > latest_idle_timestamp) {
8710 				/*
8711 				 * If equal or no active idle state, then
8712 				 * the most recently idled CPU might have
8713 				 * a warmer cache.
8714 				 */
8715 				latest_idle_timestamp = rq->idle_stamp;
8716 				shallowest_idle_cpu = i;
8717 			}
8718 		} else if (shallowest_idle_cpu == -1) {
8719 			load = cpu_load(cpu_rq(i));
8720 			if (load < min_load) {
8721 				min_load = load;
8722 				least_loaded_cpu = i;
8723 			}
8724 		}
8725 	}
8726 
8727 	return shallowest_idle_cpu != -1 ? shallowest_idle_cpu : least_loaded_cpu;
8728 }
8729 
8730 static inline int sched_balance_find_dst_cpu(struct sched_domain *sd, struct task_struct *p,
8731 				  int cpu, int prev_cpu, int sd_flag)
8732 {
8733 	int new_cpu = cpu;
8734 
8735 	if (!cpumask_intersects(sched_domain_span(sd), p->cpus_ptr))
8736 		return prev_cpu;
8737 
8738 	/*
8739 	 * We need task's util for cpu_util_without, sync it up to
8740 	 * prev_cpu's last_update_time.
8741 	 */
8742 	if (!(sd_flag & SD_BALANCE_FORK))
8743 		sync_entity_load_avg(&p->se);
8744 
8745 	while (sd) {
8746 		struct sched_group *group;
8747 		struct sched_domain *tmp;
8748 		int weight;
8749 
8750 		if (!(sd->flags & sd_flag)) {
8751 			sd = sd->child;
8752 			continue;
8753 		}
8754 
8755 		group = sched_balance_find_dst_group(sd, p, cpu);
8756 		if (!group) {
8757 			sd = sd->child;
8758 			continue;
8759 		}
8760 
8761 		new_cpu = sched_balance_find_dst_group_cpu(group, p, cpu);
8762 		if (new_cpu == cpu) {
8763 			/* Now try balancing at a lower domain level of 'cpu': */
8764 			sd = sd->child;
8765 			continue;
8766 		}
8767 
8768 		/* Now try balancing at a lower domain level of 'new_cpu': */
8769 		cpu = new_cpu;
8770 		weight = sd->span_weight;
8771 		sd = NULL;
8772 		for_each_domain(cpu, tmp) {
8773 			if (weight <= tmp->span_weight)
8774 				break;
8775 			if (tmp->flags & sd_flag)
8776 				sd = tmp;
8777 		}
8778 	}
8779 
8780 	return new_cpu;
8781 }
8782 
8783 static inline int __select_idle_cpu(int cpu, struct task_struct *p)
8784 {
8785 	if (choose_idle_cpu(cpu, p) && sched_cpu_cookie_match(cpu_rq(cpu), p))
8786 		return cpu;
8787 
8788 	return -1;
8789 }
8790 
8791 DEFINE_STATIC_KEY_FALSE(sched_smt_present);
8792 EXPORT_SYMBOL_GPL(sched_smt_present);
8793 
8794 static inline void set_idle_cores(int cpu, int val)
8795 {
8796 	struct sched_domain_shared *sds;
8797 
8798 	sds = rcu_dereference_all(per_cpu(sd_balance_shared, cpu));
8799 	if (sds)
8800 		WRITE_ONCE(sds->has_idle_cores, val);
8801 }
8802 
8803 static inline bool test_idle_cores(int cpu)
8804 {
8805 	struct sched_domain_shared *sds;
8806 
8807 	sds = rcu_dereference_all(per_cpu(sd_balance_shared, cpu));
8808 	if (sds)
8809 		return READ_ONCE(sds->has_idle_cores);
8810 
8811 	return false;
8812 }
8813 
8814 /*
8815  * Scans the local SMT mask to see if the entire core is idle, and records this
8816  * information in sd_balance_shared->has_idle_cores.
8817  *
8818  * Since SMT siblings share all cache levels, inspecting this limited remote
8819  * state should be fairly cheap.
8820  */
8821 void __update_idle_core(struct rq *rq)
8822 {
8823 	int core = cpu_of(rq);
8824 	int cpu;
8825 
8826 	rcu_read_lock();
8827 	if (test_idle_cores(core))
8828 		goto unlock;
8829 
8830 	for_each_cpu(cpu, cpu_smt_mask(core)) {
8831 		if (cpu == core)
8832 			continue;
8833 
8834 		if (!available_idle_cpu(cpu))
8835 			goto unlock;
8836 	}
8837 
8838 	set_idle_cores(core, 1);
8839 unlock:
8840 	rcu_read_unlock();
8841 }
8842 
8843 /*
8844  * Scan the entire LLC domain for idle cores; this dynamically switches off if
8845  * there are no idle cores left in the system; tracked through
8846  * sd_balance_shared->has_idle_cores and enabled through update_idle_core()
8847  * above.
8848  */
8849 static int select_idle_core(struct task_struct *p, int core, struct cpumask *cpus, int *idle_cpu)
8850 {
8851 	bool idle = true;
8852 	int cpu;
8853 
8854 	for_each_cpu(cpu, cpu_smt_mask(core)) {
8855 		if (!available_idle_cpu(cpu)) {
8856 			idle = false;
8857 			if (*idle_cpu == -1) {
8858 				if (choose_sched_idle_rq(cpu_rq(cpu), p) &&
8859 				    cpumask_test_cpu(cpu, cpus)) {
8860 					*idle_cpu = cpu;
8861 					break;
8862 				}
8863 				continue;
8864 			}
8865 			break;
8866 		}
8867 		if (*idle_cpu == -1 && cpumask_test_cpu(cpu, cpus))
8868 			*idle_cpu = cpu;
8869 	}
8870 
8871 	if (idle)
8872 		return core;
8873 
8874 	cpumask_andnot(cpus, cpus, cpu_smt_mask(core));
8875 	return -1;
8876 }
8877 
8878 /*
8879  * Scan the local SMT mask for idle CPUs.
8880  */
8881 static int select_idle_smt(struct task_struct *p, struct sched_domain *sd, int target)
8882 {
8883 	int cpu;
8884 
8885 	for_each_cpu_and(cpu, cpu_smt_mask(target), p->cpus_ptr) {
8886 		if (cpu == target)
8887 			continue;
8888 		/*
8889 		 * Check if the CPU is in the LLC scheduling domain of @target.
8890 		 * Due to isolcpus, there is no guarantee that all the siblings are in the domain.
8891 		 */
8892 		if (!cpumask_test_cpu(cpu, sched_domain_span(sd)))
8893 			continue;
8894 		if (choose_idle_cpu(cpu, p))
8895 			return cpu;
8896 	}
8897 
8898 	return -1;
8899 }
8900 
8901 /*
8902  * Scan the LLC domain for idle CPUs; this is dynamically regulated by
8903  * comparing the average scan cost (tracked in sd->avg_scan_cost) against the
8904  * average idle time for this rq (as found in rq->avg_idle).
8905  */
8906 static int select_idle_cpu(struct task_struct *p, struct sched_domain *sd, bool has_idle_core, int target)
8907 {
8908 	struct cpumask *cpus = this_cpu_cpumask_var_ptr(select_rq_mask);
8909 	int i, cpu, idle_cpu = -1, nr = INT_MAX;
8910 
8911 	if (sched_feat(SIS_UTIL) && sd->shared) {
8912 		/*
8913 		 * Increment because !--nr is the condition to stop scan.
8914 		 *
8915 		 * Since "sd" is "sd_llc" for target CPU dereferenced in the
8916 		 * caller, it is safe to directly dereference "sd->shared".
8917 		 * Topology bits always ensure it assigned for "sd_llc" abd it
8918 		 * cannot disappear as long as we have a RCU protected
8919 		 * reference to one the associated "sd" here.
8920 		 */
8921 		nr = READ_ONCE(sd->shared->nr_idle_scan) + 1;
8922 		/* overloaded LLC is unlikely to have idle cpu/core */
8923 		if (nr == 1)
8924 			return -1;
8925 	}
8926 
8927 	if (!cpumask_and(cpus, sched_domain_span(sd), p->cpus_ptr))
8928 		return -1;
8929 
8930 	if (static_branch_unlikely(&sched_cluster_active)) {
8931 		struct sched_group *sg = sd->groups;
8932 
8933 		if (sg->flags & SD_CLUSTER) {
8934 			for_each_cpu_wrap(cpu, sched_group_span(sg), target + 1) {
8935 				if (!cpumask_test_cpu(cpu, cpus))
8936 					continue;
8937 
8938 				if (has_idle_core) {
8939 					i = select_idle_core(p, cpu, cpus, &idle_cpu);
8940 					if ((unsigned int)i < nr_cpumask_bits)
8941 						return i;
8942 				} else {
8943 					if (--nr <= 0)
8944 						return -1;
8945 					idle_cpu = __select_idle_cpu(cpu, p);
8946 					if ((unsigned int)idle_cpu < nr_cpumask_bits)
8947 						return idle_cpu;
8948 				}
8949 			}
8950 			cpumask_andnot(cpus, cpus, sched_group_span(sg));
8951 		}
8952 	}
8953 
8954 	for_each_cpu_wrap(cpu, cpus, target + 1) {
8955 		if (has_idle_core) {
8956 			i = select_idle_core(p, cpu, cpus, &idle_cpu);
8957 			if ((unsigned int)i < nr_cpumask_bits)
8958 				return i;
8959 
8960 		} else {
8961 			if (--nr <= 0)
8962 				return -1;
8963 			idle_cpu = __select_idle_cpu(cpu, p);
8964 			if ((unsigned int)idle_cpu < nr_cpumask_bits)
8965 				break;
8966 		}
8967 	}
8968 
8969 	if (has_idle_core)
8970 		set_idle_cores(target, false);
8971 
8972 	return idle_cpu;
8973 }
8974 
8975 /*
8976  * Idle-capacity scan converts util_fits_cpu() outcomes into preference ranks,
8977  * where lower values indicate a better fit - see select_idle_capacity().
8978  *
8979  * A CPU that both fits the task and sits on a fully-idle SMT core is returned
8980  * immediately and is never assigned one of these ranks. On !SMT every CPU is
8981  * its own "core", so the early return covers all fits-and-idle cases and the
8982  * core-tier ranks below become unreachable.
8983  *
8984  *   Rank                            Val  Tier    Meaning
8985  *   ------------------------------  ---  ------  ---------------------------
8986  *   ASYM_IDLE_UCLAMP_MISFIT         -4   core    Idle core; capacity fits
8987  *                                                util but uclamp_min misses.
8988  *   ASYM_IDLE_COMPLETE_MISFIT       -3   core    Idle core; capacity does
8989  *                                                not fit. Still beats every
8990  *                                                thread-tier rank: a busy
8991  *                                                sibling cuts effective
8992  *                                                capacity more than a
8993  *                                                misfit hurts a quiet core.
8994  *   ASYM_IDLE_THREAD_FITS           -2   thread  Busy SMT sibling; capacity
8995  *                                                fits util + uclamp.
8996  *   ASYM_IDLE_THREAD_UCLAMP_MISFIT  -1   thread  Busy SMT sibling; capacity
8997  *                                                fits but uclamp_min misses
8998  *                                                (native util_fits_cpu()
8999  *                                                return value).
9000  *   ASYM_IDLE_THREAD_MISFIT          0   thread  Busy SMT sibling; capacity
9001  *                                                does not fit.
9002  *
9003  * ASYM_IDLE_CORE_BIAS (-3) is an offset, not a state. On an idle core,
9004  * fits += ASYM_IDLE_CORE_BIAS rebases thread-tier ranks into the core tier:
9005  *
9006  *   ASYM_IDLE_THREAD_UCLAMP_MISFIT (-1) + BIAS -> ASYM_IDLE_UCLAMP_MISFIT   (-4)
9007  *   ASYM_IDLE_THREAD_MISFIT         (0) + BIAS -> ASYM_IDLE_COMPLETE_MISFIT (-3)
9008  *
9009  * ASYM_IDLE_THREAD_FITS (-2) is never rebased because a fully-fitting idle-core
9010  * candidate early-returns from select_idle_capacity().
9011  */
9012 enum asym_fits_state {
9013 	ASYM_IDLE_UCLAMP_MISFIT = -4,
9014 	ASYM_IDLE_COMPLETE_MISFIT,
9015 	ASYM_IDLE_THREAD_FITS,
9016 	ASYM_IDLE_THREAD_UCLAMP_MISFIT,
9017 	ASYM_IDLE_THREAD_MISFIT,
9018 
9019 	/* util_fits_cpu() bias for idle core */
9020 	ASYM_IDLE_CORE_BIAS = -3,
9021 };
9022 
9023 /*
9024  * Scan the asym_capacity domain for idle CPUs; pick the first idle one on which
9025  * the task fits. If no CPU is big enough, but there are idle ones, try to
9026  * maximize capacity.
9027  */
9028 static int
9029 select_idle_capacity(struct task_struct *p, struct sched_domain *sd, int target)
9030 {
9031 	/*
9032 	 * On !SMT systems, has_idle_core is always false and preferred_core
9033 	 * is always true (CPU == core), so the SMT preference logic below
9034 	 * collapses to the plain capacity scan.
9035 	 */
9036 	bool has_idle_core = sched_smt_active() && test_idle_cores(target);
9037 	unsigned long task_util, util_min, util_max, best_cap = 0;
9038 	int fits, best_fits = ASYM_IDLE_THREAD_MISFIT;
9039 	int cpu, best_cpu = -1;
9040 	struct cpumask *cpus;
9041 	int nr = INT_MAX;
9042 
9043 	cpus = this_cpu_cpumask_var_ptr(select_rq_mask);
9044 	cpumask_and(cpus, sched_domain_span(sd), p->cpus_ptr);
9045 
9046 	task_util = task_util_est(p);
9047 	util_min = uclamp_eff_value(p, UCLAMP_MIN);
9048 	util_max = uclamp_eff_value(p, UCLAMP_MAX);
9049 
9050 	if (sched_feat(SIS_UTIL) && sd->shared) {
9051 		/*
9052 		 * Same nr_idle_scan hint as select_idle_cpu(), nr only limits
9053 		 * the scan when not preferring an idle core.
9054 		 */
9055 		nr = READ_ONCE(sd->shared->nr_idle_scan) + 1;
9056 		/* overloaded domain is unlikely to have idle cpu/core */
9057 		if (nr == 1)
9058 			return -1;
9059 	}
9060 
9061 	for_each_cpu_wrap(cpu, cpus, target) {
9062 		bool preferred_core = !has_idle_core || is_core_idle(cpu);
9063 		unsigned long cpu_cap = capacity_of(cpu);
9064 
9065 		/*
9066 		 * Stop when the nr_idle_scan is exhausted (mirrors
9067 		 * select_idle_cpu() logic).
9068 		 */
9069 		if (!has_idle_core && --nr <= 0)
9070 			return best_cpu;
9071 
9072 		if (!choose_idle_cpu(cpu, p))
9073 			continue;
9074 
9075 		fits = util_fits_cpu(task_util, util_min, util_max, cpu);
9076 
9077 		/*
9078 		 * Perfect fit: capacity satisfies util + uclamp and the CPU
9079 		 * sits on a fully-idle SMT core, this is a !SMT system, or
9080 		 * there is no idle core to find.
9081 		 * Short-circuit the rank-based selection and return
9082 		 * immediately.
9083 		 */
9084 		if (fits > 0 && preferred_core)
9085 			return cpu;
9086 		/*
9087 		 * Only the min performance hint (i.e. uclamp_min) doesn't fit.
9088 		 * Look for the CPU with best capacity.
9089 		 */
9090 		else if (fits < 0)
9091 			cpu_cap = get_actual_cpu_capacity(cpu);
9092 		/*
9093 		 * fits > 0 implies we are not on a preferred core, but the util
9094 		 * fits CPU capacity. Set fits to ASYM_IDLE_THREAD_FITS
9095 		 * so the effective range becomes
9096 		 * [ASYM_IDLE_THREAD_FITS, ASYM_IDLE_THREAD_MISFIT], where:
9097 		 *    ASYM_IDLE_THREAD_MISFIT - does not fit
9098 		 *    ASYM_IDLE_THREAD_UCLAMP_MISFIT - fits with the exception of UCLAMP_MIN
9099 		 *    ASYM_IDLE_THREAD_FITS - fits with the exception of preferred_core
9100 		 */
9101 		else if (fits > 0)
9102 			fits = ASYM_IDLE_THREAD_FITS;
9103 
9104 		/*
9105 		 * If we are on a preferred core, translate the range of fits
9106 		 * of [ASYM_IDLE_THREAD_UCLAMP_MISFIT, ASYM_IDLE_THREAD_MISFIT] to
9107 		 * [ASYM_IDLE_UCLAMP_MISFIT, ASYM_IDLE_COMPLETE_MISFIT].
9108 		 * This ensures that an idle core is always given priority over
9109 		 * (partially) busy core.
9110 		 *
9111 		 * A fully fitting idle core would have returned early and hence
9112 		 * fits > 0 for preferred_core need not be dealt with.
9113 		 */
9114 		if (preferred_core)
9115 			fits += ASYM_IDLE_CORE_BIAS;
9116 
9117 		/*
9118 		 * First, select CPU which fits better (lower is more preferred).
9119 		 * Then, select the one with best capacity at same level.
9120 		 */
9121 		if ((fits < best_fits) ||
9122 		    ((fits == best_fits) && (cpu_cap > best_cap))) {
9123 			best_cap = cpu_cap;
9124 			best_cpu = cpu;
9125 			best_fits = fits;
9126 		}
9127 	}
9128 
9129 	/*
9130 	 * A value in the [ASYM_IDLE_UCLAMP_MISFIT, ASYM_IDLE_COMPLETE_MISFIT]
9131 	 * range means the chosen CPU is in a fully idle SMT core. Values above
9132 	 * ASYM_IDLE_COMPLETE_MISFIT mean we never ranked such a CPU best.
9133 	 *
9134 	 * The asym-capacity wakeup path returns from select_idle_sibling()
9135 	 * after this function and never runs select_idle_cpu(), so the usual
9136 	 * select_idle_cpu() tail that clears idle cores must live here when the
9137 	 * idle-core preference did not win.
9138 	 */
9139 	if (has_idle_core && best_fits > ASYM_IDLE_COMPLETE_MISFIT)
9140 		set_idle_cores(target, false);
9141 
9142 	return best_cpu;
9143 }
9144 
9145 static inline bool asym_fits_cpu(unsigned long util,
9146 				 unsigned long util_min,
9147 				 unsigned long util_max,
9148 				 int cpu)
9149 {
9150 	if (sched_asym_cpucap_active()) {
9151 		/*
9152 		 * Return true only if the cpu fully fits the task requirements
9153 		 * which include the utilization and the performance hints.
9154 		 *
9155 		 * When SMT is active, also require that the core has no busy
9156 		 * siblings.
9157 		 *
9158 		 * Note: gating on is_core_idle() also makes the early-bailout
9159 		 * candidates in select_idle_sibling() (target, prev,
9160 		 * recent_used_cpu) idle-core-aware on ASYM+SMT, which the
9161 		 * NO_ASYM path does not do.
9162 		 */
9163 		return (!sched_smt_active() || is_core_idle(cpu)) &&
9164 		       (util_fits_cpu(util, util_min, util_max, cpu) > 0);
9165 	}
9166 
9167 	return true;
9168 }
9169 
9170 /*
9171  * Try and locate an idle core/thread in the LLC cache domain.
9172  */
9173 static int select_idle_sibling(struct task_struct *p, int prev, int target)
9174 {
9175 	bool has_idle_core = false;
9176 	struct sched_domain *sd;
9177 	unsigned long task_util, util_min, util_max;
9178 	int i, recent_used_cpu, prev_aff = -1;
9179 
9180 	/*
9181 	 * On asymmetric system, update task utilization because we will check
9182 	 * that the task fits with CPU's capacity.
9183 	 */
9184 	if (sched_asym_cpucap_active()) {
9185 		sync_entity_load_avg(&p->se);
9186 		task_util = task_util_est(p);
9187 		util_min = uclamp_eff_value(p, UCLAMP_MIN);
9188 		util_max = uclamp_eff_value(p, UCLAMP_MAX);
9189 	}
9190 
9191 	/*
9192 	 * per-cpu select_rq_mask usage
9193 	 */
9194 	lockdep_assert_irqs_disabled();
9195 
9196 	if (choose_idle_cpu(target, p) &&
9197 	    asym_fits_cpu(task_util, util_min, util_max, target))
9198 		return target;
9199 
9200 	/*
9201 	 * If the previous CPU is cache affine and idle, don't be stupid:
9202 	 */
9203 	if (prev != target && cpus_share_cache(prev, target) &&
9204 	    choose_idle_cpu(prev, p) &&
9205 	    asym_fits_cpu(task_util, util_min, util_max, prev)) {
9206 
9207 		if (!static_branch_unlikely(&sched_cluster_active) ||
9208 		    cpus_share_resources(prev, target))
9209 			return prev;
9210 
9211 		prev_aff = prev;
9212 	}
9213 
9214 	/*
9215 	 * Allow a per-cpu kthread to stack with the wakee if the
9216 	 * kworker thread and the tasks previous CPUs are the same.
9217 	 * The assumption is that the wakee queued work for the
9218 	 * per-cpu kthread that is now complete and the wakeup is
9219 	 * essentially a sync wakeup. An obvious example of this
9220 	 * pattern is IO completions.
9221 	 */
9222 	if (is_per_cpu_kthread(current) &&
9223 	    in_task() &&
9224 	    prev == smp_processor_id() &&
9225 	    this_rq()->nr_running <= 1 &&
9226 	    asym_fits_cpu(task_util, util_min, util_max, prev)) {
9227 		return prev;
9228 	}
9229 
9230 	/* Check a recently used CPU as a potential idle candidate: */
9231 	recent_used_cpu = p->recent_used_cpu;
9232 	p->recent_used_cpu = prev;
9233 	if (recent_used_cpu != prev &&
9234 	    recent_used_cpu != target &&
9235 	    cpus_share_cache(recent_used_cpu, target) &&
9236 	    choose_idle_cpu(recent_used_cpu, p) &&
9237 	    cpumask_test_cpu(recent_used_cpu, p->cpus_ptr) &&
9238 	    asym_fits_cpu(task_util, util_min, util_max, recent_used_cpu)) {
9239 
9240 		if (!static_branch_unlikely(&sched_cluster_active) ||
9241 		    cpus_share_resources(recent_used_cpu, target))
9242 			return recent_used_cpu;
9243 
9244 	} else {
9245 		recent_used_cpu = -1;
9246 	}
9247 
9248 	/*
9249 	 * For asymmetric CPU capacity systems, our domain of interest is
9250 	 * sd_asym_cpucapacity rather than sd_llc.
9251 	 */
9252 	if (sched_asym_cpucap_active()) {
9253 		sd = rcu_dereference_all(per_cpu(sd_asym_cpucapacity, target));
9254 		/*
9255 		 * On an asymmetric CPU capacity system where an exclusive
9256 		 * cpuset defines a symmetric island (i.e. one unique
9257 		 * capacity_orig value through the cpuset), the key will be set
9258 		 * but the CPUs within that cpuset will not have a domain with
9259 		 * SD_ASYM_CPUCAPACITY. These should follow the usual symmetric
9260 		 * capacity path.
9261 		 */
9262 		if (sd) {
9263 			i = select_idle_capacity(p, sd, target);
9264 			return ((unsigned)i < nr_cpumask_bits) ? i : target;
9265 		}
9266 	}
9267 
9268 	sd = rcu_dereference_all(per_cpu(sd_llc, target));
9269 	if (!sd)
9270 		return target;
9271 
9272 	if (sched_smt_active()) {
9273 		has_idle_core = test_idle_cores(target);
9274 
9275 		if (!has_idle_core && cpus_share_cache(prev, target)) {
9276 			i = select_idle_smt(p, sd, prev);
9277 			if ((unsigned int)i < nr_cpumask_bits)
9278 				return i;
9279 		}
9280 	}
9281 
9282 	i = select_idle_cpu(p, sd, has_idle_core, target);
9283 	if ((unsigned)i < nr_cpumask_bits)
9284 		return i;
9285 
9286 	/*
9287 	 * For cluster machines which have lower sharing cache like L2 or
9288 	 * LLC Tag, we tend to find an idle CPU in the target's cluster
9289 	 * first. But prev_cpu or recent_used_cpu may also be a good candidate,
9290 	 * use them if possible when no idle CPU found in select_idle_cpu().
9291 	 */
9292 	if ((unsigned int)prev_aff < nr_cpumask_bits)
9293 		return prev_aff;
9294 	if ((unsigned int)recent_used_cpu < nr_cpumask_bits)
9295 		return recent_used_cpu;
9296 
9297 	return target;
9298 }
9299 
9300 /**
9301  * cpu_util() - Estimates the amount of CPU capacity used by CFS tasks.
9302  * @cpu: the CPU to get the utilization for
9303  * @p: task for which the CPU utilization should be predicted or NULL
9304  * @dst_cpu: CPU @p migrates to, -1 if @p moves from @cpu or @p == NULL
9305  * @boost: 1 to enable boosting, otherwise 0
9306  *
9307  * The unit of the return value must be the same as the one of CPU capacity
9308  * so that CPU utilization can be compared with CPU capacity.
9309  *
9310  * CPU utilization is the sum of running time of runnable tasks plus the
9311  * recent utilization of currently non-runnable tasks on that CPU.
9312  * It represents the amount of CPU capacity currently used by CFS tasks in
9313  * the range [0..max CPU capacity] with max CPU capacity being the CPU
9314  * capacity at f_max.
9315  *
9316  * The estimated CPU utilization is defined as the maximum between CPU
9317  * utilization and sum of the estimated utilization of the currently
9318  * runnable tasks on that CPU. It preserves a utilization "snapshot" of
9319  * previously-executed tasks, which helps better deduce how busy a CPU will
9320  * be when a long-sleeping task wakes up. The contribution to CPU utilization
9321  * of such a task would be significantly decayed at this point of time.
9322  *
9323  * Boosted CPU utilization is defined as max(CPU runnable, CPU utilization).
9324  * CPU contention for CFS tasks can be detected by CPU runnable > CPU
9325  * utilization. Boosting is implemented in cpu_util() so that internal
9326  * users (e.g. EAS) can use it next to external users (e.g. schedutil),
9327  * latter via cpu_util_cfs_boost().
9328  *
9329  * CPU utilization can be higher than the current CPU capacity
9330  * (f_curr/f_max * max CPU capacity) or even the max CPU capacity because
9331  * of rounding errors as well as task migrations or wakeups of new tasks.
9332  * CPU utilization has to be capped to fit into the [0..max CPU capacity]
9333  * range. Otherwise a group of CPUs (CPU0 util = 121% + CPU1 util = 80%)
9334  * could be seen as over-utilized even though CPU1 has 20% of spare CPU
9335  * capacity. CPU utilization is allowed to overshoot current CPU capacity
9336  * though since this is useful for predicting the CPU capacity required
9337  * after task migrations (scheduler-driven DVFS).
9338  *
9339  * Return: (Boosted) (estimated) utilization for the specified CPU.
9340  */
9341 static unsigned long
9342 cpu_util(int cpu, struct task_struct *p, int dst_cpu, int boost)
9343 {
9344 	bool add_task = p && task_cpu(p) != cpu && dst_cpu == cpu;
9345 	bool sub_task = p && task_cpu(p) == cpu && dst_cpu != cpu;
9346 	struct cfs_rq *cfs_rq = &cpu_rq(cpu)->cfs;
9347 	unsigned long util = READ_ONCE(cfs_rq->avg.util_avg);
9348 	unsigned long runnable;
9349 
9350 	/*
9351 	 * If @dst_cpu is -1 or @p migrates from @cpu to @dst_cpu remove its
9352 	 * contribution. If @p migrates from another CPU to @cpu add its
9353 	 * contribution. In all the other cases @cpu is not impacted by the
9354 	 * migration so its util_avg is already correct.
9355 	 */
9356 	if (add_task)
9357 		util += task_util(p);
9358 	else if (sub_task)
9359 		lsub_positive(&util, task_util(p));
9360 
9361 	if (boost) {
9362 		runnable = READ_ONCE(cfs_rq->avg.runnable_avg);
9363 		if (add_task)
9364 			runnable += READ_ONCE(p->se.avg.runnable_avg);
9365 		else if (sub_task)
9366 			lsub_positive(&runnable,
9367 				      READ_ONCE(p->se.avg.runnable_avg));
9368 		util = max(util, runnable);
9369 	}
9370 
9371 	if (sched_feat(UTIL_EST)) {
9372 		unsigned long util_est;
9373 
9374 		util_est = READ_ONCE(cfs_rq->avg.util_est);
9375 
9376 		/*
9377 		 * During wake-up @p isn't enqueued yet and doesn't contribute
9378 		 * to any cpu_rq(cpu)->cfs.avg.util_est.
9379 		 * If @dst_cpu == @cpu add it to "simulate" cpu_util after @p
9380 		 * has been enqueued.
9381 		 *
9382 		 * During exec (@dst_cpu = -1) @p is enqueued and does
9383 		 * contribute to cpu_rq(cpu)->cfs.util_est.
9384 		 * Remove it to "simulate" cpu_util without @p's contribution.
9385 		 *
9386 		 * Despite the task_on_rq_queued(@p) check there is still a
9387 		 * small window for a possible race when an exec
9388 		 * select_task_rq_fair() races with LB's detach_task().
9389 		 *
9390 		 *   detach_task()
9391 		 *     deactivate_task()
9392 		 *       p->on_rq = TASK_ON_RQ_MIGRATING;
9393 		 *       -------------------------------- A
9394 		 *       dequeue_task()                    \
9395 		 *         dequeue_task_fair()              + Race Time
9396 		 *           util_est_dequeue()            /
9397 		 *       -------------------------------- B
9398 		 *
9399 		 * The additional check "current == p" is required to further
9400 		 * reduce the race window.
9401 		 */
9402 		if (dst_cpu == cpu)
9403 			util_est += _task_util_est(p);
9404 		else if (p && unlikely(task_on_rq_queued(p) || current == p))
9405 			lsub_positive(&util_est, _task_util_est(p));
9406 
9407 		util = max(util, util_est);
9408 	}
9409 
9410 	return min(util, arch_scale_cpu_capacity(cpu));
9411 }
9412 
9413 unsigned long cpu_util_cfs(int cpu)
9414 {
9415 	return cpu_util(cpu, NULL, -1, 0);
9416 }
9417 
9418 unsigned long cpu_util_cfs_boost(int cpu)
9419 {
9420 	return cpu_util(cpu, NULL, -1, 1);
9421 }
9422 
9423 /*
9424  * cpu_util_without: compute cpu utilization without any contributions from *p
9425  * @cpu: the CPU which utilization is requested
9426  * @p: the task which utilization should be discounted
9427  *
9428  * The utilization of a CPU is defined by the utilization of tasks currently
9429  * enqueued on that CPU as well as tasks which are currently sleeping after an
9430  * execution on that CPU.
9431  *
9432  * This method returns the utilization of the specified CPU by discounting the
9433  * utilization of the specified task, whenever the task is currently
9434  * contributing to the CPU utilization.
9435  */
9436 static unsigned long cpu_util_without(int cpu, struct task_struct *p)
9437 {
9438 	/* Task has no contribution or is new */
9439 	if (cpu != task_cpu(p) || !READ_ONCE(p->se.avg.last_update_time))
9440 		p = NULL;
9441 
9442 	return cpu_util(cpu, p, -1, 0);
9443 }
9444 
9445 /*
9446  * This function computes an effective utilization for the given CPU, to be
9447  * used for frequency selection given the linear relation: f = u * f_max.
9448  *
9449  * The scheduler tracks the following metrics:
9450  *
9451  *   cpu_util_{cfs,rt,dl,irq}()
9452  *   cpu_bw_dl()
9453  *
9454  * Where the cfs,rt and dl util numbers are tracked with the same metric and
9455  * synchronized windows and are thus directly comparable.
9456  *
9457  * The cfs,rt,dl utilization are the running times measured with rq->clock_task
9458  * which excludes things like IRQ and steal-time. These latter are then accrued
9459  * in the IRQ utilization.
9460  *
9461  * The DL bandwidth number OTOH is not a measured metric but a value computed
9462  * based on the task model parameters and gives the minimal utilization
9463  * required to meet deadlines.
9464  */
9465 unsigned long effective_cpu_util(int cpu, unsigned long util_cfs,
9466 				 unsigned long *min,
9467 				 unsigned long *max)
9468 {
9469 	unsigned long util, irq, scale;
9470 	struct rq *rq = cpu_rq(cpu);
9471 
9472 	scale = arch_scale_cpu_capacity(cpu);
9473 
9474 	/*
9475 	 * Early check to see if IRQ/steal time saturates the CPU, can be
9476 	 * because of inaccuracies in how we track these -- see
9477 	 * update_irq_load_avg().
9478 	 */
9479 	irq = cpu_util_irq(rq);
9480 	if (unlikely(irq >= scale)) {
9481 		if (min)
9482 			*min = scale;
9483 		if (max)
9484 			*max = scale;
9485 		return scale;
9486 	}
9487 
9488 	if (min) {
9489 		/*
9490 		 * The minimum utilization returns the highest level between:
9491 		 * - the computed DL bandwidth needed with the IRQ pressure which
9492 		 *   steals time to the deadline task.
9493 		 * - The minimum performance requirement for CFS and/or RT.
9494 		 */
9495 		*min = max(irq + cpu_bw_dl(rq), uclamp_rq_get(rq, UCLAMP_MIN));
9496 
9497 		/*
9498 		 * When an RT task is runnable and uclamp is not used, we must
9499 		 * ensure that the task will run at maximum compute capacity.
9500 		 */
9501 		if (!uclamp_is_used() && rt_rq_is_runnable(&rq->rt))
9502 			*min = max(*min, scale);
9503 	}
9504 
9505 	/*
9506 	 * Because the time spend on RT/DL tasks is visible as 'lost' time to
9507 	 * CFS tasks and we use the same metric to track the effective
9508 	 * utilization (PELT windows are synchronized) we can directly add them
9509 	 * to obtain the CPU's actual utilization.
9510 	 */
9511 	util = util_cfs + cpu_util_rt(rq);
9512 	util += cpu_util_dl(rq);
9513 
9514 	/*
9515 	 * The maximum hint is a soft bandwidth requirement, which can be lower
9516 	 * than the actual utilization because of uclamp_max requirements.
9517 	 */
9518 	if (max)
9519 		*max = min(scale, uclamp_rq_get(rq, UCLAMP_MAX));
9520 
9521 	if (util >= scale)
9522 		return scale;
9523 
9524 	/*
9525 	 * There is still idle time; further improve the number by using the
9526 	 * IRQ metric. Because IRQ/steal time is hidden from the task clock we
9527 	 * need to scale the task numbers:
9528 	 *
9529 	 *              max - irq
9530 	 *   U' = irq + --------- * U
9531 	 *                 max
9532 	 */
9533 	util = scale_irq_capacity(util, irq, scale);
9534 	util += irq;
9535 
9536 	return min(scale, util);
9537 }
9538 
9539 unsigned long sched_cpu_util(int cpu)
9540 {
9541 	return effective_cpu_util(cpu, cpu_util_cfs(cpu), NULL, NULL);
9542 }
9543 
9544 /*
9545  * energy_env - Utilization landscape for energy estimation.
9546  * @task_busy_time: Utilization contribution by the task for which we test the
9547  *                  placement. Given by eenv_task_busy_time().
9548  * @pd_busy_time:   Utilization of the whole perf domain without the task
9549  *                  contribution. Given by eenv_pd_busy_time().
9550  * @cpu_cap:        Maximum CPU capacity for the perf domain.
9551  * @pd_cap:         Entire perf domain capacity. (pd->nr_cpus * cpu_cap).
9552  */
9553 struct energy_env {
9554 	unsigned long task_busy_time;
9555 	unsigned long pd_busy_time;
9556 	unsigned long cpu_cap;
9557 	unsigned long pd_cap;
9558 };
9559 
9560 /*
9561  * Compute the task busy time for compute_energy(). This time cannot be
9562  * injected directly into effective_cpu_util() because of the IRQ scaling.
9563  * The latter only makes sense with the most recent CPUs where the task has
9564  * run.
9565  */
9566 static inline void eenv_task_busy_time(struct energy_env *eenv,
9567 				       struct task_struct *p, int prev_cpu)
9568 {
9569 	unsigned long busy_time, max_cap = arch_scale_cpu_capacity(prev_cpu);
9570 	unsigned long irq = cpu_util_irq(cpu_rq(prev_cpu));
9571 
9572 	if (unlikely(irq >= max_cap))
9573 		busy_time = max_cap;
9574 	else
9575 		busy_time = scale_irq_capacity(task_util_est(p), irq, max_cap);
9576 
9577 	eenv->task_busy_time = busy_time;
9578 }
9579 
9580 /*
9581  * Compute the perf_domain (PD) busy time for compute_energy(). Based on the
9582  * utilization for each @pd_cpus, it however doesn't take into account
9583  * clamping since the ratio (utilization / cpu_capacity) is already enough to
9584  * scale the EM reported power consumption at the (eventually clamped)
9585  * cpu_capacity.
9586  *
9587  * The contribution of the task @p for which we want to estimate the
9588  * energy cost is removed (by cpu_util()) and must be calculated
9589  * separately (see eenv_task_busy_time). This ensures:
9590  *
9591  *   - A stable PD utilization, no matter which CPU of that PD we want to place
9592  *     the task on.
9593  *
9594  *   - A fair comparison between CPUs as the task contribution (task_util())
9595  *     will always be the same no matter which CPU utilization we rely on
9596  *     (util_avg or util_est).
9597  *
9598  * Set @eenv busy time for the PD that spans @pd_cpus. This busy time can't
9599  * exceed @eenv->pd_cap.
9600  */
9601 static inline void eenv_pd_busy_time(struct energy_env *eenv,
9602 				     struct cpumask *pd_cpus,
9603 				     struct task_struct *p)
9604 {
9605 	unsigned long busy_time = 0;
9606 	int cpu;
9607 
9608 	for_each_cpu(cpu, pd_cpus) {
9609 		unsigned long util = cpu_util(cpu, p, -1, 0);
9610 
9611 		busy_time += effective_cpu_util(cpu, util, NULL, NULL);
9612 	}
9613 
9614 	eenv->pd_busy_time = min(eenv->pd_cap, busy_time);
9615 }
9616 
9617 /*
9618  * Compute the maximum utilization for compute_energy() when the task @p
9619  * is placed on the cpu @dst_cpu.
9620  *
9621  * Returns the maximum utilization among @eenv->cpus. This utilization can't
9622  * exceed @eenv->cpu_cap.
9623  */
9624 static inline unsigned long
9625 eenv_pd_max_util(struct energy_env *eenv, struct cpumask *pd_cpus,
9626 		 struct task_struct *p, int dst_cpu)
9627 {
9628 	unsigned long max_util = 0;
9629 	int cpu;
9630 
9631 	for_each_cpu(cpu, pd_cpus) {
9632 		struct task_struct *tsk = (cpu == dst_cpu) ? p : NULL;
9633 		unsigned long util = cpu_util(cpu, p, dst_cpu, 1);
9634 		unsigned long eff_util, min, max;
9635 
9636 		/*
9637 		 * Performance domain frequency: utilization clamping
9638 		 * must be considered since it affects the selection
9639 		 * of the performance domain frequency.
9640 		 * NOTE: in case RT tasks are running, by default the min
9641 		 * utilization can be max OPP.
9642 		 */
9643 		eff_util = effective_cpu_util(cpu, util, &min, &max);
9644 
9645 		/* Task's uclamp can modify min and max value */
9646 		if (tsk && uclamp_is_used()) {
9647 			min = max(min, uclamp_eff_value(p, UCLAMP_MIN));
9648 
9649 			/*
9650 			 * If there is no active max uclamp constraint,
9651 			 * directly use task's one, otherwise keep max.
9652 			 */
9653 			if (uclamp_rq_is_idle(cpu_rq(cpu)))
9654 				max = uclamp_eff_value(p, UCLAMP_MAX);
9655 			else
9656 				max = max(max, uclamp_eff_value(p, UCLAMP_MAX));
9657 		}
9658 
9659 		eff_util = sugov_effective_cpu_perf(cpu, eff_util, min, max);
9660 		max_util = max(max_util, eff_util);
9661 	}
9662 
9663 	return min(max_util, eenv->cpu_cap);
9664 }
9665 
9666 /*
9667  * compute_energy(): Use the Energy Model to estimate the energy that @pd would
9668  * consume for a given utilization landscape @eenv. When @dst_cpu < 0, the task
9669  * contribution is ignored.
9670  */
9671 static inline unsigned long
9672 compute_energy(struct energy_env *eenv, struct perf_domain *pd,
9673 	       struct cpumask *pd_cpus, struct task_struct *p, int dst_cpu)
9674 {
9675 	unsigned long max_util = eenv_pd_max_util(eenv, pd_cpus, p, dst_cpu);
9676 	unsigned long busy_time = eenv->pd_busy_time;
9677 	unsigned long energy;
9678 
9679 	if (dst_cpu >= 0)
9680 		busy_time = min(eenv->pd_cap, busy_time + eenv->task_busy_time);
9681 
9682 	energy = em_cpu_energy(pd->em_pd, max_util, busy_time, eenv->cpu_cap);
9683 
9684 	trace_sched_compute_energy_tp(p, dst_cpu, energy, max_util, busy_time);
9685 
9686 	return energy;
9687 }
9688 
9689 /*
9690  * find_energy_efficient_cpu(): Find most energy-efficient target CPU for the
9691  * waking task. find_energy_efficient_cpu() looks for the CPU with maximum
9692  * spare capacity in each performance domain and uses it as a potential
9693  * candidate to execute the task. Then, it uses the Energy Model to figure
9694  * out which of the CPU candidates is the most energy-efficient.
9695  *
9696  * The rationale for this heuristic is as follows. In a performance domain,
9697  * all the most energy efficient CPU candidates (according to the Energy
9698  * Model) are those for which we'll request a low frequency. When there are
9699  * several CPUs for which the frequency request will be the same, we don't
9700  * have enough data to break the tie between them, because the Energy Model
9701  * only includes active power costs. With this model, if we assume that
9702  * frequency requests follow utilization (e.g. using schedutil), the CPU with
9703  * the maximum spare capacity in a performance domain is guaranteed to be among
9704  * the best candidates of the performance domain.
9705  *
9706  * In practice, it could be preferable from an energy standpoint to pack
9707  * small tasks on a CPU in order to let other CPUs go in deeper idle states,
9708  * but that could also hurt our chances to go cluster idle, and we have no
9709  * ways to tell with the current Energy Model if this is actually a good
9710  * idea or not. So, find_energy_efficient_cpu() basically favors
9711  * cluster-packing, and spreading inside a cluster. That should at least be
9712  * a good thing for latency, and this is consistent with the idea that most
9713  * of the energy savings of EAS come from the asymmetry of the system, and
9714  * not so much from breaking the tie between identical CPUs. That's also the
9715  * reason why EAS is enabled in the topology code only for systems where
9716  * SD_ASYM_CPUCAPACITY is set.
9717  *
9718  * NOTE: Forkees are not accepted in the energy-aware wake-up path because
9719  * they don't have any useful utilization data yet and it's not possible to
9720  * forecast their impact on energy consumption. Consequently, they will be
9721  * placed by sched_balance_find_dst_cpu() on the least loaded CPU, which might turn out
9722  * to be energy-inefficient in some use-cases. The alternative would be to
9723  * bias new tasks towards specific types of CPUs first, or to try to infer
9724  * their util_avg from the parent task, but those heuristics could hurt
9725  * other use-cases too. So, until someone finds a better way to solve this,
9726  * let's keep things simple by re-using the existing slow path.
9727  */
9728 static int find_energy_efficient_cpu(struct task_struct *p, int prev_cpu)
9729 {
9730 	struct cpumask *cpus = this_cpu_cpumask_var_ptr(select_rq_mask);
9731 	unsigned long prev_delta = ULONG_MAX, best_delta = ULONG_MAX;
9732 	unsigned long p_util_min = uclamp_is_used() ? uclamp_eff_value(p, UCLAMP_MIN) : 0;
9733 	unsigned long p_util_max = uclamp_is_used() ? uclamp_eff_value(p, UCLAMP_MAX) : 1024;
9734 	struct root_domain *rd = this_rq()->rd;
9735 	int cpu, best_energy_cpu, target = -1;
9736 	int prev_fits = -1, best_fits = -1;
9737 	unsigned long best_actual_cap = 0;
9738 	unsigned long prev_actual_cap = 0;
9739 	struct sched_domain *sd;
9740 	struct perf_domain *pd;
9741 	struct energy_env eenv;
9742 
9743 	pd = rcu_dereference_all(rd->pd);
9744 	if (!pd)
9745 		return target;
9746 
9747 	/*
9748 	 * Energy-aware wake-up happens on the lowest sched_domain starting
9749 	 * from sd_asym_cpucapacity spanning over this_cpu and prev_cpu.
9750 	 */
9751 	sd = rcu_dereference_all(*this_cpu_ptr(&sd_asym_cpucapacity));
9752 	while (sd && !cpumask_test_cpu(prev_cpu, sched_domain_span(sd)))
9753 		sd = sd->parent;
9754 	if (!sd)
9755 		return target;
9756 
9757 	target = prev_cpu;
9758 
9759 	sync_entity_load_avg(&p->se);
9760 	if (!task_util_est(p) && p_util_min == 0)
9761 		return target;
9762 
9763 	eenv_task_busy_time(&eenv, p, prev_cpu);
9764 
9765 	for (; pd; pd = pd->next) {
9766 		unsigned long util_min = p_util_min, util_max = p_util_max;
9767 		unsigned long cpu_cap, cpu_actual_cap, util;
9768 		long prev_spare_cap = -1, max_spare_cap = -1;
9769 		unsigned long rq_util_min, rq_util_max;
9770 		unsigned long cur_delta, base_energy;
9771 		int max_spare_cap_cpu = -1;
9772 		int fits, max_fits = -1;
9773 
9774 		if (!cpumask_and(cpus, perf_domain_span(pd), cpu_online_mask))
9775 			continue;
9776 
9777 		/* Account external pressure for the energy estimation */
9778 		cpu = cpumask_first(cpus);
9779 		cpu_actual_cap = get_actual_cpu_capacity(cpu);
9780 
9781 		eenv.cpu_cap = cpu_actual_cap;
9782 		eenv.pd_cap = 0;
9783 
9784 		for_each_cpu(cpu, cpus) {
9785 			struct rq *rq = cpu_rq(cpu);
9786 
9787 			eenv.pd_cap += cpu_actual_cap;
9788 
9789 			if (!cpumask_test_cpu(cpu, sched_domain_span(sd)))
9790 				continue;
9791 
9792 			if (!cpumask_test_cpu(cpu, p->cpus_ptr))
9793 				continue;
9794 
9795 			util = cpu_util(cpu, p, cpu, 0);
9796 			cpu_cap = capacity_of(cpu);
9797 
9798 			/*
9799 			 * Skip CPUs that cannot satisfy the capacity request.
9800 			 * IOW, placing the task there would make the CPU
9801 			 * overutilized. Take uclamp into account to see how
9802 			 * much capacity we can get out of the CPU; this is
9803 			 * aligned with sched_cpu_util().
9804 			 */
9805 			if (uclamp_is_used() && !uclamp_rq_is_idle(rq)) {
9806 				/*
9807 				 * Open code uclamp_rq_util_with() except for
9808 				 * the clamp() part. I.e.: apply max aggregation
9809 				 * only. util_fits_cpu() logic requires to
9810 				 * operate on non clamped util but must use the
9811 				 * max-aggregated uclamp_{min, max}.
9812 				 */
9813 				rq_util_min = uclamp_rq_get(rq, UCLAMP_MIN);
9814 				rq_util_max = uclamp_rq_get(rq, UCLAMP_MAX);
9815 
9816 				util_min = max(rq_util_min, p_util_min);
9817 				util_max = max(rq_util_max, p_util_max);
9818 			}
9819 
9820 			fits = util_fits_cpu(util, util_min, util_max, cpu);
9821 			if (!fits)
9822 				continue;
9823 
9824 			lsub_positive(&cpu_cap, util);
9825 
9826 			if (cpu == prev_cpu) {
9827 				/* Always use prev_cpu as a candidate. */
9828 				prev_spare_cap = cpu_cap;
9829 				prev_fits = fits;
9830 			} else if ((fits > max_fits) ||
9831 				   ((fits == max_fits) && ((long)cpu_cap > max_spare_cap))) {
9832 				/*
9833 				 * Find the CPU with the maximum spare capacity
9834 				 * among the remaining CPUs in the performance
9835 				 * domain.
9836 				 */
9837 				max_spare_cap = cpu_cap;
9838 				max_spare_cap_cpu = cpu;
9839 				max_fits = fits;
9840 			}
9841 		}
9842 
9843 		if (max_spare_cap_cpu < 0 && prev_spare_cap < 0)
9844 			continue;
9845 
9846 		eenv_pd_busy_time(&eenv, cpus, p);
9847 		/* Compute the 'base' energy of the pd, without @p */
9848 		base_energy = compute_energy(&eenv, pd, cpus, p, -1);
9849 
9850 		/* Evaluate the energy impact of using prev_cpu. */
9851 		if (prev_spare_cap > -1) {
9852 			prev_delta = compute_energy(&eenv, pd, cpus, p,
9853 						    prev_cpu);
9854 			/* CPU utilization has changed */
9855 			if (prev_delta < base_energy)
9856 				return target;
9857 			prev_delta -= base_energy;
9858 			prev_actual_cap = cpu_actual_cap;
9859 			best_delta = min(best_delta, prev_delta);
9860 		}
9861 
9862 		/* Evaluate the energy impact of using max_spare_cap_cpu. */
9863 		if (max_spare_cap_cpu >= 0 && max_spare_cap > prev_spare_cap) {
9864 			/* Current best energy cpu fits better */
9865 			if (max_fits < best_fits)
9866 				continue;
9867 
9868 			/*
9869 			 * Both don't fit performance hint (i.e. uclamp_min)
9870 			 * but best energy cpu has better capacity.
9871 			 */
9872 			if ((max_fits < 0) &&
9873 			    (cpu_actual_cap <= best_actual_cap))
9874 				continue;
9875 
9876 			cur_delta = compute_energy(&eenv, pd, cpus, p,
9877 						   max_spare_cap_cpu);
9878 			/* CPU utilization has changed */
9879 			if (cur_delta < base_energy)
9880 				return target;
9881 			cur_delta -= base_energy;
9882 
9883 			/*
9884 			 * Both fit for the task but best energy cpu has lower
9885 			 * energy impact.
9886 			 */
9887 			if ((max_fits > 0) && (best_fits > 0) &&
9888 			    (cur_delta >= best_delta))
9889 				continue;
9890 
9891 			best_delta = cur_delta;
9892 			best_energy_cpu = max_spare_cap_cpu;
9893 			best_fits = max_fits;
9894 			best_actual_cap = cpu_actual_cap;
9895 		}
9896 	}
9897 
9898 	if ((best_fits > prev_fits) ||
9899 	    ((best_fits > 0) && (best_delta < prev_delta)) ||
9900 	    ((best_fits < 0) && (best_actual_cap > prev_actual_cap)))
9901 		target = best_energy_cpu;
9902 
9903 	return target;
9904 }
9905 
9906 /*
9907  * select_task_rq_fair: Select target runqueue for the waking task in domains
9908  * that have the relevant SD flag set. In practice, this is SD_BALANCE_WAKE,
9909  * SD_BALANCE_FORK, or SD_BALANCE_EXEC.
9910  *
9911  * Balances load by selecting the idlest CPU in the idlest group, or under
9912  * certain conditions an idle sibling CPU if the domain has SD_WAKE_AFFINE set.
9913  *
9914  * Returns the target CPU number.
9915  */
9916 static int
9917 select_task_rq_fair(struct task_struct *p, int prev_cpu, int wake_flags)
9918 {
9919 	int sync = (wake_flags & WF_SYNC) && !(current->flags & PF_EXITING);
9920 	struct sched_domain *tmp, *sd = NULL;
9921 	int cpu = smp_processor_id();
9922 	int new_cpu = prev_cpu;
9923 	int want_affine = 0;
9924 	/* SD_flags and WF_flags share the first nibble */
9925 	int sd_flag = wake_flags & 0xF;
9926 
9927 	/*
9928 	 * required for stable ->cpus_allowed
9929 	 */
9930 	lockdep_assert_held(&p->pi_lock);
9931 	if (wake_flags & WF_TTWU) {
9932 		record_wakee(p);
9933 
9934 		if ((wake_flags & WF_CURRENT_CPU) &&
9935 		    cpumask_test_cpu(cpu, p->cpus_ptr))
9936 			return cpu;
9937 
9938 		if (!is_rd_overutilized(this_rq()->rd)) {
9939 			new_cpu = find_energy_efficient_cpu(p, prev_cpu);
9940 			if (new_cpu >= 0)
9941 				return new_cpu;
9942 			new_cpu = prev_cpu;
9943 		}
9944 
9945 		want_affine = !wake_wide(p) && cpumask_test_cpu(cpu, p->cpus_ptr);
9946 	}
9947 
9948 	for_each_domain(cpu, tmp) {
9949 		/*
9950 		 * If both 'cpu' and 'prev_cpu' are part of this domain,
9951 		 * cpu is a valid SD_WAKE_AFFINE target.
9952 		 */
9953 		if (want_affine && (tmp->flags & SD_WAKE_AFFINE) &&
9954 		    cpumask_test_cpu(prev_cpu, sched_domain_span(tmp))) {
9955 			if (cpu != prev_cpu)
9956 				new_cpu = wake_affine(tmp, p, cpu, prev_cpu, sync);
9957 
9958 			sd = NULL; /* Prefer wake_affine over balance flags */
9959 			break;
9960 		}
9961 
9962 		/*
9963 		 * Usually only true for WF_EXEC and WF_FORK, as sched_domains
9964 		 * usually do not have SD_BALANCE_WAKE set. That means wakeup
9965 		 * will usually go to the fast path.
9966 		 */
9967 		if (tmp->flags & sd_flag)
9968 			sd = tmp;
9969 		else if (!want_affine)
9970 			break;
9971 	}
9972 
9973 	/* Slow path */
9974 	if (unlikely(sd))
9975 		return sched_balance_find_dst_cpu(sd, p, cpu, prev_cpu, sd_flag);
9976 
9977 	/* Fast path */
9978 	if (wake_flags & WF_TTWU)
9979 		return select_idle_sibling(p, prev_cpu, new_cpu);
9980 
9981 	return new_cpu;
9982 }
9983 
9984 /*
9985  * Called immediately before a task is migrated to a new CPU; task_cpu(p) and
9986  * cfs_rq_of(p) references at time of call are still valid and identify the
9987  * previous CPU. The caller guarantees p->pi_lock or task_rq(p)->lock is held.
9988  */
9989 static void migrate_task_rq_fair(struct task_struct *p, int new_cpu)
9990 {
9991 	struct sched_entity *se = &p->se;
9992 
9993 	if (!task_on_rq_migrating(p)) {
9994 		remove_entity_load_avg(se);
9995 
9996 		/*
9997 		 * Here, the task's PELT values have been updated according to
9998 		 * the current rq's clock. But if that clock hasn't been
9999 		 * updated in a while, a substantial idle time will be missed,
10000 		 * leading to an inflation after wake-up on the new rq.
10001 		 *
10002 		 * Estimate the missing time from the cfs_rq last_update_time
10003 		 * and update sched_avg to improve the PELT continuity after
10004 		 * migration.
10005 		 */
10006 		migrate_se_pelt_lag(se);
10007 	}
10008 
10009 	/* Tell new CPU we are migrated */
10010 	se->avg.last_update_time = 0;
10011 
10012 	update_scan_period(p, new_cpu);
10013 }
10014 
10015 static void task_dead_fair(struct task_struct *p)
10016 {
10017 	struct sched_entity *se = &p->se;
10018 	remove_entity_load_avg(se);
10019 }
10020 
10021 /*
10022  * Set the max capacity the task is allowed to run at for misfit detection.
10023  */
10024 static void set_task_max_allowed_capacity(struct task_struct *p)
10025 {
10026 	struct asym_cap_data *entry;
10027 
10028 	if (!sched_asym_cpucap_active())
10029 		return;
10030 
10031 	rcu_read_lock();
10032 	list_for_each_entry_rcu(entry, &asym_cap_list, link) {
10033 		cpumask_t *cpumask;
10034 
10035 		cpumask = cpu_capacity_span(entry);
10036 		if (!cpumask_intersects(p->cpus_ptr, cpumask))
10037 			continue;
10038 
10039 		p->max_allowed_capacity = entry->capacity;
10040 		break;
10041 	}
10042 	rcu_read_unlock();
10043 }
10044 
10045 static void set_cpus_allowed_fair(struct task_struct *p, struct affinity_context *ctx)
10046 {
10047 	set_cpus_allowed_common(p, ctx);
10048 	set_task_max_allowed_capacity(p);
10049 }
10050 
10051 enum preempt_wakeup_action {
10052 	PREEMPT_WAKEUP_NONE,	/* No preemption. */
10053 	PREEMPT_WAKEUP_SHORT,	/* Ignore slice protection. */
10054 	PREEMPT_WAKEUP_PICK,	/* Let pick_eevdf() decide. */
10055 	PREEMPT_WAKEUP_RESCHED,	/* Force reschedule. */
10056 };
10057 
10058 static inline bool set_preempt_buddy(struct cfs_rq *cfs_rq, struct sched_entity *pse)
10059 {
10060 	/*
10061 	 * Keep existing buddy if the deadline is sooner than pse.
10062 	 * The older buddy may be cache cold and completely unrelated
10063 	 * to the current wakeup but that is unpredictable where as
10064 	 * obeying the deadline is more in line with EEVDF objectives.
10065 	 */
10066 	if (cfs_rq->next && entity_before(cfs_rq->next, pse))
10067 		return false;
10068 
10069 	set_next_buddy(cfs_rq, pse);
10070 	return true;
10071 }
10072 
10073 static inline bool set_short_buddy(struct cfs_rq *cfs_rq, struct sched_entity *pse)
10074 {
10075 	if (cfs_rq->next && cfs_rq->next->slice < pse->slice)
10076 		return false;
10077 
10078 	set_next_buddy(cfs_rq, pse);
10079 	return true;
10080 }
10081 
10082 /*
10083  * WF_SYNC|WF_TTWU indicates the waker expects to sleep but it is not
10084  * strictly enforced because the hint is either misunderstood or
10085  * multiple tasks must be woken up.
10086  */
10087 static inline enum preempt_wakeup_action
10088 preempt_sync(struct rq *rq, int wake_flags,
10089 	     struct sched_entity *pse, struct sched_entity *se)
10090 {
10091 	u64 threshold, delta;
10092 
10093 	/*
10094 	 * WF_SYNC without WF_TTWU is not expected so warn if it happens even
10095 	 * though it is likely harmless.
10096 	 */
10097 	WARN_ON_ONCE(!(wake_flags & WF_TTWU));
10098 
10099 	threshold = sysctl_sched_migration_cost;
10100 	delta = rq_clock_task(rq) - se->exec_start;
10101 	if ((s64)delta < 0)
10102 		delta = 0;
10103 
10104 	/*
10105 	 * WF_RQ_SELECTED implies the tasks are stacking on a CPU when they
10106 	 * could run on other CPUs. Reduce the threshold before preemption is
10107 	 * allowed to an arbitrary lower value as it is more likely (but not
10108 	 * guaranteed) the waker requires the wakee to finish.
10109 	 */
10110 	if (wake_flags & WF_RQ_SELECTED)
10111 		threshold >>= 2;
10112 
10113 	/*
10114 	 * As WF_SYNC is not strictly obeyed, allow some runtime for batch
10115 	 * wakeups to be issued.
10116 	 */
10117 	if (entity_before(pse, se) && delta >= threshold)
10118 		return PREEMPT_WAKEUP_RESCHED;
10119 
10120 	return PREEMPT_WAKEUP_NONE;
10121 }
10122 
10123 /*
10124  * Preempt the current task with a newly woken task if needed:
10125  */
10126 static void wakeup_preempt_fair(struct rq *rq, struct task_struct *p, int wake_flags)
10127 {
10128 	enum preempt_wakeup_action preempt_action = PREEMPT_WAKEUP_PICK;
10129 	struct task_struct *donor = rq->donor;
10130 	struct sched_entity *nse, *se = &donor->se, *pse = &p->se;
10131 	struct cfs_rq *cfs_rq = &rq->cfs;
10132 	int cse_is_idle, pse_is_idle;
10133 
10134 	/*
10135 	 * XXX Getting preempted by higher class, try and find idle CPU?
10136 	 */
10137 	if (p->sched_class != &fair_sched_class ||
10138 	    donor->sched_class != &fair_sched_class)
10139 		return;
10140 
10141 	if (unlikely(se == pse))
10142 		return;
10143 
10144 	/*
10145 	 * This is possible from callers such as attach_tasks(), in which we
10146 	 * unconditionally wakeup_preempt() after an enqueue (which may have
10147 	 * lead to a throttle).  This both saves work and prevents false
10148 	 * next-buddy nomination below.
10149 	 */
10150 	if (task_is_throttled(p))
10151 		return;
10152 
10153 	/*
10154 	 * We can come here with TIF_NEED_RESCHED already set from new task
10155 	 * wake up path.
10156 	 *
10157 	 * Note: this also catches the edge-case of curr being in a throttled
10158 	 * group (e.g. via set_curr_task), since update_curr() (in the
10159 	 * enqueue of curr) will have resulted in resched being set.  This
10160 	 * prevents us from potentially nominating it as a false LAST_BUDDY
10161 	 * below.
10162 	 */
10163 	if (!sched_feat(PREEMPT_SHORT) && test_tsk_need_resched(rq->curr))
10164 		return;
10165 
10166 	if (!sched_feat(WAKEUP_PREEMPTION))
10167 		return;
10168 
10169 	WARN_ON_ONCE(!pse);
10170 
10171 	cse_is_idle = se_is_idle(se);
10172 	pse_is_idle = se_is_idle(pse);
10173 
10174 	nse = se;
10175 	/*
10176 	 * Preempt an idle entity in favor of a non-idle entity (and don't preempt
10177 	 * in the inverse case).
10178 	 */
10179 	if (cse_is_idle && !pse_is_idle)
10180 		goto preempt;
10181 
10182 	update_curr_fair(rq);
10183 
10184 	if (cse_is_idle != pse_is_idle)
10185 		goto update;
10186 
10187 	/*
10188 	 * BATCH and IDLE tasks do not preempt others.
10189 	 */
10190 	if (unlikely(!normal_policy(p->policy)))
10191 		goto update;
10192 
10193 	/*
10194 	 * Do not preempt for tasks that are sched_delayed as it would violate
10195 	 * EEVDF to forcibly queue an ineligible task.
10196 	 */
10197 	if (pse->sched_delayed)
10198 		goto update;
10199 
10200 	/*
10201 	 * If @p has a shorter slice than current and @p is eligible, override
10202 	 * current's slice protection in order to allow preemption.
10203 	 */
10204 	if (sched_feat(PREEMPT_SHORT) && (pse->slice < se->slice)) {
10205 		preempt_action = PREEMPT_WAKEUP_SHORT;
10206 		goto pick;
10207 	}
10208 
10209 	/*
10210 	 * Ignore wakee preemption on WF_FORK as it is less likely that
10211 	 * there is shared data as exec often follow fork.
10212 	 */
10213 	if (wake_flags & WF_FORK)
10214 		goto update;
10215 
10216 	/* Prefer picking wakee soon if appropriate. */
10217 	if (sched_feat(NEXT_BUDDY) && set_preempt_buddy(cfs_rq, pse)) {
10218 		/*
10219 		 * Decide whether to obey WF_SYNC hint for a new buddy. Old
10220 		 * buddies are ignored as they may not be relevant to the
10221 		 * waker and less likely to be cache hot.
10222 		 */
10223 		if (wake_flags & WF_SYNC)
10224 			preempt_action = preempt_sync(rq, wake_flags, pse, se);
10225 	}
10226 
10227 	switch (preempt_action) {
10228 	case PREEMPT_WAKEUP_NONE:
10229 		return;
10230 	case PREEMPT_WAKEUP_RESCHED:
10231 		goto preempt;
10232 	case PREEMPT_WAKEUP_SHORT:
10233 		fallthrough;
10234 	case PREEMPT_WAKEUP_PICK:
10235 		break;
10236 	}
10237 
10238 pick:
10239 	if (cfs_rq->h_nr_queued) {
10240 		nse = pick_next_entity(rq, preempt_action != PREEMPT_WAKEUP_SHORT);
10241 		if (unlikely(!nse))
10242 			goto pick;
10243 
10244 		/* If @p has become the most eligible task, force preemption */
10245 		if (nse == pse)
10246 			goto preempt;
10247 	}
10248 
10249 	/*
10250 	 * If @p is eligible but not the next task to run then cancel protection
10251 	 * to prevent large scheduling latency
10252 	 */
10253 	if (preempt_action == PREEMPT_WAKEUP_SHORT && entity_eligible(cfs_rq, pse))
10254 		goto preempt;
10255 update:
10256 	if (sched_feat(RUN_TO_PARITY))
10257 		update_protect_slice(cfs_rq, se);
10258 
10259 	return;
10260 
10261 preempt:
10262 	cancel_protect_slice(se);
10263 
10264 	if (preempt_action == PREEMPT_WAKEUP_SHORT)
10265 		set_short_buddy(cfs_rq, pse);
10266 
10267 	resched_curr_lazy(rq);
10268 }
10269 
10270 struct task_struct *pick_task_fair(struct rq *rq, struct rq_flags *rf)
10271 	__must_hold(__rq_lockp(rq))
10272 {
10273 	struct cfs_rq *cfs_rq = &rq->cfs;
10274 	struct sched_entity *se;
10275 	struct task_struct *p;
10276 	int new_tasks;
10277 
10278 again:
10279 	if (!cfs_rq->h_nr_queued)
10280 		goto idle;
10281 
10282 	/* Might not have done put_prev_entity() */
10283 	if (cfs_rq->curr && cfs_rq->curr->on_rq)
10284 		update_curr_eevdf(cfs_rq);
10285 
10286 	se = pick_next_entity(rq, true);
10287 	if (!se)
10288 		goto again;
10289 
10290 	p = task_of(se);
10291 	return p;
10292 
10293 idle:
10294 	if (sched_core_enabled(rq))
10295 		return NULL;
10296 
10297 	new_tasks = sched_balance_newidle(rq, rf);
10298 	if (new_tasks < 0)
10299 		return RETRY_TASK;
10300 	if (new_tasks > 0)
10301 		goto again;
10302 	return NULL;
10303 }
10304 
10305 static struct task_struct *
10306 fair_server_pick_task(struct sched_dl_entity *dl_se, struct rq_flags *rf)
10307 	__must_hold(__rq_lockp(dl_se->rq))
10308 {
10309 	return pick_task_fair(dl_se->rq, rf);
10310 }
10311 
10312 void fair_server_init(struct rq *rq)
10313 {
10314 	struct sched_dl_entity *dl_se = &rq->fair_server;
10315 
10316 	init_dl_entity(dl_se);
10317 
10318 	dl_server_init(dl_se, rq, fair_server_pick_task);
10319 }
10320 
10321 /*
10322  * Account for a descheduled task:
10323  */
10324 static void put_prev_task_fair(struct rq *rq, struct task_struct *prev, struct task_struct *next)
10325 {
10326 	struct sched_entity *se = &prev->se;
10327 	struct cfs_rq *cfs_rq = &rq->cfs;
10328 	struct sched_entity *nse = NULL;
10329 
10330 #ifdef CONFIG_FAIR_GROUP_SCHED
10331 	if (next && next->sched_class == &fair_sched_class)
10332 		nse = &next->se;
10333 #endif
10334 
10335 	while (se) {
10336 		cfs_rq = cfs_rq_of(se);
10337 		if (!nse || cfs_rq->h_curr)
10338 			put_prev_entity(cfs_rq, se);
10339 #ifdef CONFIG_FAIR_GROUP_SCHED
10340 		if (nse) {
10341 			if (is_same_group(se, nse))
10342 				break;
10343 
10344 			int d = nse->depth - se->depth;
10345 			if (d >= 0) {
10346 				/* nse has equal or greater depth, ascend */
10347 				nse = parent_entity(nse);
10348 				/* if nse is the deeper, do not ascend se */
10349 				if (d > 0)
10350 					continue;
10351 			}
10352 		}
10353 #endif
10354 		se = parent_entity(se);
10355 	}
10356 
10357 	/* Put 'current' back into the tree. */
10358 	cfs_rq = &rq->cfs;
10359 	se = &prev->se;
10360 	WARN_ON_ONCE(cfs_rq->curr != se);
10361 	cfs_rq->curr = NULL;
10362 	if (se->on_rq)
10363 		__enqueue_entity(cfs_rq, se);
10364 }
10365 
10366 /*
10367  * sched_yield() is very simple
10368  */
10369 static void yield_task_fair(struct rq *rq)
10370 {
10371 	struct task_struct *curr = rq->donor;
10372 	struct sched_entity *se = &curr->se;
10373 	struct cfs_rq *cfs_rq = &rq->cfs;
10374 
10375 	/*
10376 	 * Are we the only task in the tree?
10377 	 */
10378 	if (unlikely(rq->nr_running == 1))
10379 		return;
10380 
10381 	clear_buddies(cfs_rq, se);
10382 
10383 	update_rq_clock(rq);
10384 	/*
10385 	 * Update run-time statistics of the 'current'.
10386 	 */
10387 	update_curr_eevdf(cfs_rq);
10388 	/*
10389 	 * Tell update_rq_clock() that we've just updated,
10390 	 * so we don't do microscopic update in schedule()
10391 	 * and double the fastpath cost.
10392 	 */
10393 	rq_clock_skip_update(rq);
10394 
10395 	/*
10396 	 * Forfeit the remaining vruntime, only if the entity is eligible. This
10397 	 * condition is necessary because in core scheduling we prefer to run
10398 	 * ineligible tasks rather than force idling. If this happens we may
10399 	 * end up in a loop where the core scheduler picks the yielding task,
10400 	 * which yields immediately again; without the condition the vruntime
10401 	 * ends up quickly running away.
10402 	 */
10403 	if (entity_eligible(cfs_rq, se)) {
10404 		se->vruntime = se->deadline;
10405 		update_deadline(cfs_rq, se);
10406 	}
10407 }
10408 
10409 static bool yield_to_task_fair(struct rq *rq, struct task_struct *p)
10410 {
10411 	struct sched_entity *se = &p->se;
10412 
10413 	/* !se->on_rq also covers throttled task */
10414 	if (!se->on_rq || se->sched_delayed)
10415 		return false;
10416 
10417 	/* Tell the scheduler that we'd really like se to run next. */
10418 	set_next_buddy(&task_rq(p)->cfs, se);
10419 
10420 	yield_task_fair(rq);
10421 
10422 	return true;
10423 }
10424 
10425 /**************************************************
10426  * Fair scheduling class load-balancing methods.
10427  *
10428  * BASICS
10429  *
10430  * The purpose of load-balancing is to achieve the same basic fairness the
10431  * per-CPU scheduler provides, namely provide a proportional amount of compute
10432  * time to each task. This is expressed in the following equation:
10433  *
10434  *   W_i,n/P_i == W_j,n/P_j for all i,j                               (1)
10435  *
10436  * Where W_i,n is the n-th weight average for CPU i. The instantaneous weight
10437  * W_i,0 is defined as:
10438  *
10439  *   W_i,0 = \Sum_j w_i,j                                             (2)
10440  *
10441  * Where w_i,j is the weight of the j-th runnable task on CPU i. This weight
10442  * is derived from the nice value as per sched_prio_to_weight[].
10443  *
10444  * The weight average is an exponential decay average of the instantaneous
10445  * weight:
10446  *
10447  *   W'_i,n = (2^n - 1) / 2^n * W_i,n + 1 / 2^n * W_i,0               (3)
10448  *
10449  * C_i is the compute capacity of CPU i, typically it is the
10450  * fraction of 'recent' time available for SCHED_OTHER task execution. But it
10451  * can also include other factors [XXX].
10452  *
10453  * To achieve this balance we define a measure of imbalance which follows
10454  * directly from (1):
10455  *
10456  *   imb_i,j = max{ avg(W/C), W_i/C_i } - min{ avg(W/C), W_j/C_j }    (4)
10457  *
10458  * We them move tasks around to minimize the imbalance. In the continuous
10459  * function space it is obvious this converges, in the discrete case we get
10460  * a few fun cases generally called infeasible weight scenarios.
10461  *
10462  * [XXX expand on:
10463  *     - infeasible weights;
10464  *     - local vs global optima in the discrete case. ]
10465  *
10466  *
10467  * SCHED DOMAINS
10468  *
10469  * In order to solve the imbalance equation (4), and avoid the obvious O(n^2)
10470  * for all i,j solution, we create a tree of CPUs that follows the hardware
10471  * topology where each level pairs two lower groups (or better). This results
10472  * in O(log n) layers. Furthermore we reduce the number of CPUs going up the
10473  * tree to only the first of the previous level and we decrease the frequency
10474  * of load-balance at each level inversely proportional to the number of CPUs in
10475  * the groups.
10476  *
10477  * This yields:
10478  *
10479  *     log_2 n     1     n
10480  *   \Sum       { --- * --- * 2^i } = O(n)                            (5)
10481  *     i = 0      2^i   2^i
10482  *                               `- size of each group
10483  *         |         |     `- number of CPUs doing load-balance
10484  *         |         `- freq
10485  *         `- sum over all levels
10486  *
10487  * Coupled with a limit on how many tasks we can migrate every balance pass,
10488  * this makes (5) the runtime complexity of the balancer.
10489  *
10490  * An important property here is that each CPU is still (indirectly) connected
10491  * to every other CPU in at most O(log n) steps:
10492  *
10493  * The adjacency matrix of the resulting graph is given by:
10494  *
10495  *             log_2 n
10496  *   A_i,j = \Union     (i % 2^k == 0) && i / 2^(k+1) == j / 2^(k+1)  (6)
10497  *             k = 0
10498  *
10499  * And you'll find that:
10500  *
10501  *   A^(log_2 n)_i,j != 0  for all i,j                                (7)
10502  *
10503  * Showing there's indeed a path between every CPU in at most O(log n) steps.
10504  * The task movement gives a factor of O(m), giving a convergence complexity
10505  * of:
10506  *
10507  *   O(nm log n),  n := nr_cpus, m := nr_tasks                        (8)
10508  *
10509  *
10510  * WORK CONSERVING
10511  *
10512  * In order to avoid CPUs going idle while there's still work to do, new idle
10513  * balancing is more aggressive and has the newly idle CPU iterate up the domain
10514  * tree itself instead of relying on other CPUs to bring it work.
10515  *
10516  * This adds some complexity to both (5) and (8) but it reduces the total idle
10517  * time.
10518  *
10519  * [XXX more?]
10520  *
10521  *
10522  * CGROUPS
10523  *
10524  * Cgroups make a horror show out of (2), instead of a simple sum we get:
10525  *
10526  *                                s_k,i
10527  *   W_i,0 = \Sum_j \Prod_k w_k * -----                               (9)
10528  *                                 S_k
10529  *
10530  * Where
10531  *
10532  *   s_k,i = \Sum_j w_i,j,k  and  S_k = \Sum_i s_k,i                 (10)
10533  *
10534  * w_i,j,k is the weight of the j-th runnable task in the k-th cgroup on CPU i.
10535  *
10536  * The big problem is S_k, its a global sum needed to compute a local (W_i)
10537  * property.
10538  *
10539  * [XXX write more on how we solve this.. _after_ merging pjt's patches that
10540  *      rewrite all of this once again.]
10541  */
10542 
10543 static unsigned long __read_mostly max_load_balance_interval = HZ/10;
10544 
10545 enum fbq_type { regular, remote, all };
10546 
10547 /*
10548  * 'group_type' describes the group of CPUs at the moment of load balancing.
10549  *
10550  * The enum is ordered by pulling priority, with the group with lowest priority
10551  * first so the group_type can simply be compared when selecting the busiest
10552  * group. See update_sd_pick_busiest().
10553  */
10554 enum group_type {
10555 	/* The group has spare capacity that can be used to run more tasks.  */
10556 	group_has_spare = 0,
10557 	/*
10558 	 * The group is fully used and the tasks don't compete for more CPU
10559 	 * cycles. Nevertheless, some tasks might wait before running.
10560 	 */
10561 	group_fully_busy,
10562 	/*
10563 	 * One task doesn't fit with CPU's capacity and must be migrated to a
10564 	 * more powerful CPU.
10565 	 */
10566 	group_misfit_task,
10567 	/*
10568 	 * Balance SMT group that's fully busy. Can benefit from migration
10569 	 * a task on SMT with busy sibling to another CPU on idle core.
10570 	 */
10571 	group_smt_balance,
10572 	/*
10573 	 * SD_ASYM_PACKING only: One local CPU with higher capacity is available,
10574 	 * and the task should be migrated to it instead of running on the
10575 	 * current CPU.
10576 	 */
10577 	group_asym_packing,
10578 	/*
10579 	 * The tasks' affinity constraints previously prevented the scheduler
10580 	 * from balancing the load across the system.
10581 	 */
10582 	group_imbalanced,
10583 	/*
10584 	 * There are tasks running on non-preferred LLC, possible to move
10585 	 * them to their preferred LLC without creating too much imbalance.
10586 	 * The priority of group_llc_balance is lower than that of
10587 	 * group_overloaded and higher than that of all other group types.
10588 	 * This is because group_llc_balance may exacerbate load imbalance.
10589 	 * If the LLC balancing attempt fails, the nr_balance_failed
10590 	 * mechanism will trigger other group types to rebalance the load.
10591 	 */
10592 	group_llc_balance,
10593 	/*
10594 	 * The CPU is overloaded and can't provide expected CPU cycles to all
10595 	 * tasks.
10596 	 */
10597 	group_overloaded
10598 };
10599 
10600 enum migration_type {
10601 	migrate_load = 0,
10602 	migrate_util,
10603 	migrate_task,
10604 	migrate_misfit,
10605 	migrate_llc_task
10606 };
10607 
10608 #define LBF_ALL_PINNED	0x01
10609 #define LBF_NEED_BREAK	0x02
10610 #define LBF_DST_PINNED  0x04
10611 #define LBF_SOME_PINNED	0x08
10612 #define LBF_ACTIVE_LB	0x10
10613 #define LBF_LLC_PINNED	0x20
10614 #define LBF_ACTIVE_LB_LLC	0x40
10615 
10616 struct lb_env {
10617 	struct sched_domain	*sd;
10618 
10619 	struct rq		*src_rq;
10620 	int			src_cpu;
10621 
10622 	int			dst_cpu;
10623 	struct rq		*dst_rq;
10624 	bool			dst_core_idle;
10625 
10626 	struct cpumask		*dst_grpmask;
10627 	int			new_dst_cpu;
10628 	enum cpu_idle_type	idle;
10629 	long			imbalance;
10630 	/* The set of CPUs under consideration for load-balancing */
10631 	struct cpumask		*cpus;
10632 
10633 	unsigned int		flags;
10634 
10635 	unsigned int		loop;
10636 	unsigned int		loop_break;
10637 	unsigned int		loop_max;
10638 
10639 	enum fbq_type		fbq_type;
10640 	enum migration_type	migration_type;
10641 	struct list_head	tasks;
10642 };
10643 
10644 /*
10645  * Is this task likely cache-hot:
10646  */
10647 static int task_hot(struct task_struct *p, struct lb_env *env)
10648 {
10649 	s64 delta;
10650 
10651 	lockdep_assert_rq_held(env->src_rq);
10652 
10653 	if (p->sched_class != &fair_sched_class)
10654 		return 0;
10655 
10656 	if (unlikely(task_has_idle_policy(p)))
10657 		return 0;
10658 
10659 	/* SMT siblings share cache */
10660 	if (env->sd->flags & SD_SHARE_CPUCAPACITY)
10661 		return 0;
10662 
10663 	/*
10664 	 * Buddy candidates are cache hot:
10665 	 */
10666 	if (sched_feat(CACHE_HOT_BUDDY) && env->dst_rq->nr_running &&
10667 	    (&p->se == cfs_rq_of(&p->se)->next))
10668 		return 1;
10669 
10670 	if (sysctl_sched_migration_cost == -1)
10671 		return 1;
10672 
10673 	/*
10674 	 * Don't migrate task if the task's cookie does not match
10675 	 * with the destination CPU's core cookie.
10676 	 */
10677 	if (!sched_core_cookie_match(cpu_rq(env->dst_cpu), p))
10678 		return 1;
10679 
10680 	if (sysctl_sched_migration_cost == 0)
10681 		return 0;
10682 
10683 	delta = rq_clock_task(env->src_rq) - p->se.exec_start;
10684 
10685 	return delta < (s64)sysctl_sched_migration_cost;
10686 }
10687 
10688 #ifdef CONFIG_NUMA_BALANCING
10689 /*
10690  * Returns a positive value, if task migration degrades locality.
10691  * Returns 0, if task migration is not affected by locality.
10692  * Returns a negative value, if task migration improves locality i.e migration preferred.
10693  */
10694 static long migrate_degrades_locality(struct task_struct *p, struct lb_env *env)
10695 {
10696 	struct numa_group *numa_group = rcu_dereference_all(p->numa_group);
10697 	unsigned long src_weight, dst_weight;
10698 	int src_nid, dst_nid, dist;
10699 
10700 	if (!static_branch_likely(&sched_numa_balancing))
10701 		return 0;
10702 
10703 	if (!p->numa_faults || !(env->sd->flags & SD_NUMA))
10704 		return 0;
10705 
10706 	src_nid = cpu_to_node(env->src_cpu);
10707 	dst_nid = cpu_to_node(env->dst_cpu);
10708 
10709 	if (src_nid == dst_nid)
10710 		return 0;
10711 
10712 	/* Migrating away from the preferred node is always bad. */
10713 	if (src_nid == p->numa_preferred_nid) {
10714 		if (env->src_rq->nr_running > env->src_rq->nr_preferred_running)
10715 			return 1;
10716 		else
10717 			return 0;
10718 	}
10719 
10720 	/* Encourage migration to the preferred node. */
10721 	if (dst_nid == p->numa_preferred_nid)
10722 		return -1;
10723 
10724 	/* Leaving a core idle is often worse than degrading locality. */
10725 	if (env->idle == CPU_IDLE)
10726 		return 0;
10727 
10728 	dist = node_distance(src_nid, dst_nid);
10729 	if (numa_group) {
10730 		src_weight = group_weight(p, src_nid, dist);
10731 		dst_weight = group_weight(p, dst_nid, dist);
10732 	} else {
10733 		src_weight = task_weight(p, src_nid, dist);
10734 		dst_weight = task_weight(p, dst_nid, dist);
10735 	}
10736 
10737 	return src_weight - dst_weight;
10738 }
10739 
10740 #else /* !CONFIG_NUMA_BALANCING: */
10741 static inline long migrate_degrades_locality(struct task_struct *p,
10742 					     struct lb_env *env)
10743 {
10744 	return 0;
10745 }
10746 #endif /* !CONFIG_NUMA_BALANCING */
10747 
10748 /*
10749  * Check whether the task is ineligible on the destination cpu
10750  *
10751  * When the PLACE_LAG scheduling feature is enabled and
10752  * dst_cfs_rq->nr_queued is greater than 1, if the task
10753  * is ineligible, it will also be ineligible when
10754  * it is migrated to the destination cpu.
10755  */
10756 static inline int task_is_ineligible_on_dst_cpu(struct task_struct *p, int dest_cpu)
10757 {
10758 	struct cfs_rq *dst_cfs_rq = &cpu_rq(dest_cpu)->cfs;
10759 
10760 	if (sched_feat(PLACE_LAG) && dst_cfs_rq->h_nr_queued &&
10761 	    !entity_eligible(&task_rq(p)->cfs, &p->se))
10762 		return 1;
10763 
10764 	return 0;
10765 }
10766 
10767 #ifdef CONFIG_SCHED_CACHE
10768 /*
10769  * The margin used when comparing LLC utilization with CPU capacity.
10770  * It determines the LLC load level where active LLC aggregation is
10771  * done.
10772  * Derived from fits_capacity().
10773  *
10774  * (default: ~50%, tunable via debugfs)
10775  */
10776 static bool fits_llc_capacity(unsigned long util, unsigned long max)
10777 {
10778 	u32 aggr_pct = llc_overaggr_pct;
10779 
10780 	/*
10781 	 * For single core systems, raise the aggregation
10782 	 * threshold to accommodate more tasks.
10783 	 */
10784 	if (cpu_smt_num_threads == 1)
10785 		aggr_pct = (aggr_pct * 3 / 2);
10786 
10787 	return util * 100 < max * aggr_pct;
10788 }
10789 
10790 /*
10791  * The margin used when comparing utilization.
10792  * is 'util1' noticeably greater than 'util2'
10793  * Derived from capacity_greater().
10794  * Bias is in perentage.
10795  */
10796 /* Allows dst util to be bigger than src util by up to bias percent */
10797 #define util_greater(util1, util2) \
10798 	((util1) * 100 > (util2) * (100 + llc_imb_pct))
10799 
10800 static __maybe_unused bool get_llc_stats(int cpu, unsigned long *util,
10801 					 unsigned long *cap)
10802 {
10803 	struct sched_domain_shared *sd_share;
10804 
10805 	sd_share = rcu_dereference_all(per_cpu(sd_llc_shared, cpu));
10806 	if (!sd_share)
10807 		return false;
10808 
10809 	*util = READ_ONCE(sd_share->util_avg);
10810 	*cap = READ_ONCE(sd_share->capacity);
10811 
10812 	return true;
10813 }
10814 
10815 /*
10816  * Decision matrix according to the LLC utilization. To
10817  * decide whether we can do task aggregation across LLC.
10818  *
10819  * By default, 50% is the threshold for treating the LLC
10820  * as busy. The reason for choosing 50% is to avoid saturation
10821  * of SMT-2, and it is also a safe cutoff for other SMT-n
10822  * platforms. SMT-1 has higher threshold because it is
10823  * supposed to accommodate more tasks, see fits_llc_capacity().
10824  *
10825  * 20% is the utilization imbalance percentage to decide
10826  * if the preferred LLC is busier than the non-preferred LLC.
10827  * 20 is a little higher than the LLC domain's imbalance_pct
10828  * 17. The hysteresis is used to avoid task bouncing between the
10829  * preferred LLC and the non-preferred LLC, and it will
10830  * be turned into tunable debugfs.
10831  *
10832  * 1. moving towards the preferred LLC, dst is the preferred
10833  *    LLC, src is not.
10834  *
10835  * src \ dst      30%  40%  50%  60%
10836  * 30%            Y    Y    Y    N
10837  * 40%            Y    Y    Y    Y
10838  * 50%            Y    Y    G    G
10839  * 60%            Y    Y    G    G
10840  *
10841  * 2. moving out of the preferred LLC, src is the preferred
10842  *    LLC, dst is not:
10843  *
10844  * src \ dst      30%  40%  50%  60%
10845  * 30%            N    N    N    N
10846  * 40%            N    N    N    N
10847  * 50%            N    N    G    G
10848  * 60%            Y    N    G    G
10849  *
10850  * src :      src_util
10851  * dst :      dst_util
10852  * Y :        Yes, migrate
10853  * N :        No, do not migrate
10854  * G :        let the Generic load balance to even the load.
10855  *
10856  * The intention is that if both LLCs are quite busy, cache aware
10857  * load balance should not be performed, and generic load balance
10858  * should take effect. However, if one is busy and the other is not,
10859  * the preferred LLC capacity(50%) and imbalance criteria(20%) should
10860  * be considered to determine whether LLC aggregation should be
10861  * performed to bias the load towards the preferred LLC.
10862  */
10863 
10864 /* migration decision, 3 states are orthogonal. */
10865 enum llc_mig {
10866 	mig_forbid = 0,		/* N: Don't migrate task, respect LLC preference */
10867 	mig_llc,		/* Y: Do LLC preference based migration */
10868 	mig_unrestricted	/* G: Don't restrict generic load balance migration */
10869 };
10870 
10871 /*
10872  * Check if task can be moved from the source LLC to the
10873  * destination LLC without breaking cache aware preferrence.
10874  * src_cpu and dst_cpu are arbitrary CPUs within the source
10875  * and destination LLCs, respectively.
10876  */
10877 static enum llc_mig can_migrate_llc(int src_cpu, int dst_cpu,
10878 				    unsigned long tsk_util,
10879 				    bool to_pref)
10880 {
10881 	unsigned long src_util, dst_util, src_cap, dst_cap;
10882 
10883 	if (!get_llc_stats(src_cpu, &src_util, &src_cap) ||
10884 	    !get_llc_stats(dst_cpu, &dst_util, &dst_cap))
10885 		return mig_unrestricted;
10886 
10887 	src_util = src_util < tsk_util ? 0 : src_util - tsk_util;
10888 	dst_util = dst_util + tsk_util;
10889 
10890 	if (!fits_llc_capacity(dst_util, dst_cap) &&
10891 	    !fits_llc_capacity(src_util, src_cap))
10892 		return mig_unrestricted;
10893 
10894 	if (to_pref) {
10895 		/*
10896 		 * Don't migrate if we will get preferred LLC too
10897 		 * heavily loaded and if the dest is much busier
10898 		 * than the src, in which case migration will
10899 		 * increase the imbalance too much.
10900 		 */
10901 		if (!fits_llc_capacity(dst_util, dst_cap) &&
10902 		    util_greater(dst_util, src_util))
10903 			return mig_forbid;
10904 	} else {
10905 		/*
10906 		 * Don't migrate if we will leave preferred LLC
10907 		 * too idle, or if this migration leads to the
10908 		 * non-preferred LLC falls within sysctl_aggr_imb percent
10909 		 * of preferred LLC, leading to migration again
10910 		 * back to preferred LLC.
10911 		 */
10912 		if (fits_llc_capacity(src_util, src_cap) ||
10913 		    !util_greater(src_util, dst_util))
10914 			return mig_forbid;
10915 	}
10916 	return mig_llc;
10917 }
10918 
10919 static inline bool task_misfits_asym_cpu(struct lb_env *env, struct task_struct *p)
10920 {
10921 	/*
10922 	 * On asymmetric CPU capacity domains, do not let cache-aware
10923 	 * balancing pull the task onto a destination CPU that cannot
10924 	 * accommodate it. Doing so would turn the task into a misfit on
10925 	 * the destination, trading a cache-locality gain for a capacity
10926 	 * loss. If the task already does not fit its source CPU, the move
10927 	 * cannot make things worse, so let the LLC preference decide.
10928 	 */
10929 	if ((env->sd->flags & SD_ASYM_CPUCAPACITY) && p &&
10930 	    !task_fits_cpu(p, env->dst_cpu) &&
10931 	    task_fits_cpu(p, env->src_cpu))
10932 		return true;
10933 
10934 	return false;
10935 }
10936 
10937 /*
10938  * Check if task p can migrate from source LLC to
10939  * destination LLC in terms of cache aware load balance.
10940  */
10941 static enum llc_mig can_migrate_llc_task(struct lb_env *env,
10942 					 struct task_struct *p)
10943 {
10944 	struct sched_cache_group *grp;
10945 	bool to_pref;
10946 	int cpu, src_cpu, dst_cpu;
10947 
10948 	if (task_misfits_asym_cpu(env, p))
10949 		return mig_forbid;
10950 
10951 	src_cpu = env->src_cpu;
10952 	dst_cpu = env->dst_cpu;
10953 	grp = rcu_dereference_all(p->sched_cache_grp);
10954 	if (!grp)
10955 		return mig_unrestricted;
10956 
10957 	cpu = READ_ONCE(grp->cpu);
10958 	if (cpu < 0 || cpus_share_cache(src_cpu, dst_cpu))
10959 		return mig_unrestricted;
10960 
10961 	/* skip cache aware load balance for too many threads */
10962 	if (invalid_llc_nr(grp, p, dst_cpu) ||
10963 	    exceed_llc_capacity(grp, dst_cpu)) {
10964 		if (READ_ONCE(grp->cpu) != -1)
10965 			WRITE_ONCE(grp->cpu, -1);
10966 		return mig_unrestricted;
10967 	}
10968 
10969 	if (cpus_share_cache(dst_cpu, cpu))
10970 		to_pref = true;
10971 	else if (cpus_share_cache(src_cpu, cpu))
10972 		to_pref = false;
10973 	else
10974 		return mig_unrestricted;
10975 
10976 	return can_migrate_llc(src_cpu, dst_cpu,
10977 			       task_util(p), to_pref);
10978 }
10979 
10980 /*
10981  * Check if active load balance breaks LLC locality in
10982  * terms of cache aware load balance. The load level and
10983  * imbalance do not warrant breaking LLC preference per
10984  * the can_migrate_llc() policy. Here, the benefit of
10985  * LLC locality outweighs the power efficiency gained from
10986  * migrating the only runnable task away.
10987  */
10988 static inline bool
10989 alb_break_llc(struct lb_env *env)
10990 {
10991 	if (!sched_cache_enabled())
10992 		return false;
10993 
10994 	if (cpus_share_cache(env->src_cpu, env->dst_cpu))
10995 		return false;
10996 	/*
10997 	 * All tasks prefer to stay on their current CPU.
10998 	 * Do not pull a task from its preferred CPU if:
10999 	 * 1. It is the only task running and does not exceed
11000 	 *    imbalance allowance; OR
11001 	 * 2. Migrating it away from its preferred LLC would violate
11002 	 *    the cache-aware scheduling policy.
11003 	 */
11004 	if (env->src_rq->nr_pref_llc_running &&
11005 	    env->src_rq->nr_pref_llc_running == env->src_rq->cfs.h_nr_runnable) {
11006 		unsigned long util = 0;
11007 		struct task_struct *cur;
11008 
11009 		/*
11010 		 * Migrating misfit tasks from current CPU
11011 		 * to CPU with a better fit.
11012 		 * Prioritize that over LLC preference.
11013 		 */
11014 		if (env->migration_type == migrate_misfit)
11015 			return false;
11016 
11017 		if (env->src_rq->nr_running <= 1)
11018 			return true;
11019 
11020 		cur = rcu_dereference_all(env->src_rq->curr);
11021 		if (cur && cur->sched_class == &fair_sched_class)
11022 			util = task_util(cur);
11023 
11024 		if (task_misfits_asym_cpu(env, cur) ||
11025 		    can_migrate_llc(env->src_cpu, env->dst_cpu,
11026 				    util, false) == mig_forbid)
11027 			return true;
11028 	}
11029 
11030 	return false;
11031 }
11032 
11033 /*
11034  * Returns true if p's preferred LLC does not match the destination CPU
11035  * under migrate_llc_task semantics. Passive LB passes migrate_llc_task
11036  * in env->migration_type, while active LB carries LBF_ACTIVE_LB_LLC in
11037  * env->flags to avoid overwriting env->migration_type.
11038  */
11039 static inline bool
11040 migrate_llc_task_wrong_dst(struct task_struct *p, struct lb_env *env)
11041 {
11042 	return sched_cache_enabled() &&
11043 	       (env->migration_type == migrate_llc_task ||
11044 		env->flags & LBF_ACTIVE_LB_LLC) &&
11045 	       READ_ONCE(p->preferred_llc) != llc_id(env->dst_cpu);
11046 }
11047 
11048 /*
11049  * Check if migrating task p from env->src_cpu to
11050  * env->dst_cpu breaks LLC localiy.
11051  */
11052 static bool migrate_degrades_llc(struct task_struct *p, struct lb_env *env)
11053 {
11054 	if (!sched_cache_enabled())
11055 		return false;
11056 
11057 	if (task_has_sched_core(p))
11058 		return false;
11059 	/*
11060 	 * Skip over tasks that would degrade LLC locality;
11061 	 * only when nr_balanced_failed is sufficiently high do we
11062 	 * ignore this constraint.
11063 	 *
11064 	 * Threshold of cache_nice_tries is set to 1 higher
11065 	 * than nr_balance_failed to avoid excessive task
11066 	 * migration at the same time.
11067 	 */
11068 	if (env->sd->nr_balance_failed >= env->sd->cache_nice_tries + 1)
11069 		return false;
11070 
11071 	/*
11072 	 * We know the env->src_cpu has some tasks prefer to
11073 	 * run on env->dst_cpu, skip the tasks do not prefer
11074 	 * env->dst_cpu, and find the one that prefers.
11075 	 */
11076 	if (migrate_llc_task_wrong_dst(p, env))
11077 		return true;
11078 
11079 	if (can_migrate_llc_task(env, p) != mig_forbid)
11080 		return false;
11081 
11082 	return true;
11083 }
11084 
11085 #else
11086 static inline bool get_llc_stats(int cpu, unsigned long *util,
11087 				 unsigned long *cap)
11088 {
11089 	return false;
11090 }
11091 
11092 static inline bool
11093 alb_break_llc(struct lb_env *env)
11094 {
11095 	return false;
11096 }
11097 
11098 static inline bool
11099 migrate_llc_task_wrong_dst(struct task_struct *p, struct lb_env *env)
11100 {
11101 	return false;
11102 }
11103 
11104 static inline bool
11105 migrate_degrades_llc(struct task_struct *p, struct lb_env *env)
11106 {
11107 	return false;
11108 }
11109 #endif
11110 /*
11111  * can_migrate_task - may task p from runqueue rq be migrated to this_cpu?
11112  */
11113 static
11114 int can_migrate_task(struct task_struct *p, struct lb_env *env)
11115 {
11116 	long degrades, hot;
11117 
11118 	lockdep_assert_rq_held(env->src_rq);
11119 	if (p->sched_task_hot)
11120 		p->sched_task_hot = 0;
11121 
11122 	/*
11123 	 * We do not migrate tasks that are:
11124 	 * 1) delayed dequeued unless we migrate load, or
11125 	 * 2) target cfs_rq is in throttled hierarchy, or
11126 	 * 3) cannot be migrated to this CPU due to cpus_ptr, or
11127 	 * 4) running (obviously), or
11128 	 * 5) are cache-hot on their current CPU, or
11129 	 * 6) are blocked on mutexes (if SCHED_PROXY_EXEC is enabled)
11130 	 */
11131 	if ((p->se.sched_delayed) && (env->migration_type != migrate_load))
11132 		return 0;
11133 
11134 	if (lb_throttled_hierarchy(p, env->dst_cpu))
11135 		return 0;
11136 
11137 	/*
11138 	 * We want to prioritize the migration of eligible tasks.
11139 	 * For ineligible tasks we soft-limit them and only allow
11140 	 * them to migrate when nr_balance_failed is non-zero to
11141 	 * avoid load-balancing trying very hard to balance the load.
11142 	 */
11143 	if (!env->sd->nr_balance_failed &&
11144 	    task_is_ineligible_on_dst_cpu(p, env->dst_cpu))
11145 		return 0;
11146 
11147 	/* Disregard percpu kthreads; they are where they need to be. */
11148 	if (kthread_is_per_cpu(p))
11149 		return 0;
11150 
11151 	if (task_is_blocked(p))
11152 		return 0;
11153 
11154 	if (!cpumask_test_cpu(env->dst_cpu, p->cpus_ptr)) {
11155 		int cpu;
11156 
11157 		schedstat_inc(p->stats.nr_failed_migrations_affine);
11158 
11159 		env->flags |= LBF_SOME_PINNED;
11160 
11161 		/*
11162 		 * Remember if this task can be migrated to any other CPU in
11163 		 * our sched_group. We may want to revisit it if we couldn't
11164 		 * meet load balance goals by pulling other tasks on src_cpu.
11165 		 *
11166 		 * Avoid computing new_dst_cpu
11167 		 * - for NEWLY_IDLE
11168 		 * - if we have already computed one in current iteration
11169 		 * - if it's an active balance
11170 		 */
11171 		if (env->idle == CPU_NEWLY_IDLE ||
11172 		    env->flags & (LBF_DST_PINNED | LBF_ACTIVE_LB))
11173 			return 0;
11174 
11175 		/* Prevent to re-select dst_cpu via env's CPUs: */
11176 		cpu = cpumask_first_and_and(env->dst_grpmask, env->cpus, p->cpus_ptr);
11177 
11178 		if (cpu < nr_cpu_ids) {
11179 			env->flags |= LBF_DST_PINNED;
11180 			env->new_dst_cpu = cpu;
11181 		}
11182 
11183 		return 0;
11184 	}
11185 
11186 	/* Record that we found at least one task that could run on dst_cpu */
11187 	env->flags &= ~LBF_ALL_PINNED;
11188 
11189 	if (task_on_cpu(env->src_rq, p) ||
11190 	    task_current_donor(env->src_rq, p)) {
11191 		schedstat_inc(p->stats.nr_failed_migrations_running);
11192 		return 0;
11193 	}
11194 
11195 	/*
11196 	 * Aggressive migration if:
11197 	 * 1) active balance
11198 	 * 2) destination numa is preferred
11199 	 * 3) task is cache cold, or
11200 	 * 4) too many balance attempts have failed.
11201 	 */
11202 	if (env->flags & LBF_ACTIVE_LB)
11203 		return !migrate_llc_task_wrong_dst(p, env);
11204 
11205 	degrades = migrate_degrades_locality(p, env);
11206 	if (!degrades) {
11207 		/*
11208 		 * If the NUMA locality is not broken,
11209 		 * further check if migration would hurt
11210 		 * LLC locality.
11211 		 */
11212 		if (migrate_degrades_llc(p, env)) {
11213 			/*
11214 			 * If regular load balancing fails to pull a task
11215 			 * due to LLC locality, this is expected behavior
11216 			 * and we set LBF_LLC_PINNED so we don't increase
11217 			 * nr_balance_failed unecessarily.
11218 			 */
11219 			if (env->migration_type != migrate_llc_task)
11220 				env->flags |= LBF_LLC_PINNED;
11221 
11222 			return 0;
11223 		}
11224 
11225 		hot = task_hot(p, env);
11226 	} else {
11227 		hot = degrades > 0;
11228 	}
11229 
11230 	if (!hot || env->sd->nr_balance_failed > env->sd->cache_nice_tries) {
11231 		if (hot)
11232 			p->sched_task_hot = 1;
11233 		return 1;
11234 	}
11235 
11236 	schedstat_inc(p->stats.nr_failed_migrations_hot);
11237 	return 0;
11238 }
11239 
11240 /*
11241  * detach_task() -- detach the task for the migration specified in env
11242  */
11243 static void detach_task(struct task_struct *p, struct lb_env *env)
11244 {
11245 	lockdep_assert_rq_held(env->src_rq);
11246 
11247 	if (p->sched_task_hot) {
11248 		p->sched_task_hot = 0;
11249 		schedstat_inc(env->sd->lb_hot_gained[env->idle]);
11250 		schedstat_inc(p->stats.nr_forced_migrations);
11251 	}
11252 
11253 	WARN_ON(task_current(env->src_rq, p));
11254 	WARN_ON(task_current_donor(env->src_rq, p));
11255 
11256 	deactivate_task(env->src_rq, p, DEQUEUE_NOCLOCK);
11257 	set_task_cpu(p, env->dst_cpu);
11258 }
11259 
11260 /*
11261  * detach_one_task() -- tries to dequeue exactly one task from env->src_rq, as
11262  * part of active balancing operations within "domain".
11263  *
11264  * Returns a task if successful and NULL otherwise.
11265  */
11266 static struct task_struct *detach_one_task(struct lb_env *env)
11267 {
11268 	struct task_struct *p;
11269 
11270 	lockdep_assert_rq_held(env->src_rq);
11271 
11272 	list_for_each_entry_reverse(p,
11273 			&env->src_rq->cfs_tasks, se.group_node) {
11274 		if (!can_migrate_task(p, env))
11275 			continue;
11276 
11277 		detach_task(p, env);
11278 
11279 		/*
11280 		 * Right now, this is only the second place where
11281 		 * lb_gained[env->idle] is updated (other is detach_tasks)
11282 		 * so we can safely collect stats here rather than
11283 		 * inside detach_tasks().
11284 		 */
11285 		schedstat_inc(env->sd->lb_gained[env->idle]);
11286 		return p;
11287 	}
11288 	return NULL;
11289 }
11290 
11291 /*
11292  * detach_tasks() -- tries to detach up to imbalance load/util/tasks from
11293  * busiest_rq, as part of a balancing operation within domain "sd".
11294  *
11295  * Returns number of detached tasks if successful and 0 otherwise.
11296  */
11297 static int detach_tasks(struct lb_env *env)
11298 {
11299 	struct list_head *tasks = &env->src_rq->cfs_tasks;
11300 	unsigned long util, load;
11301 	struct task_struct *p;
11302 	int detached = 0;
11303 
11304 	lockdep_assert_rq_held(env->src_rq);
11305 
11306 	/*
11307 	 * Source run queue has been emptied by another CPU, clear
11308 	 * LBF_ALL_PINNED flag as we will not test any task.
11309 	 */
11310 	if (env->src_rq->nr_running <= 1) {
11311 		env->flags &= ~LBF_ALL_PINNED;
11312 		return 0;
11313 	}
11314 
11315 	if (env->imbalance <= 0)
11316 		return 0;
11317 
11318 	while (!list_empty(tasks)) {
11319 		/*
11320 		 * We don't want to steal all, otherwise we may be treated likewise,
11321 		 * which could at worst lead to a livelock crash.
11322 		 */
11323 		if (env->idle && env->src_rq->nr_running <= 1)
11324 			break;
11325 
11326 		env->loop++;
11327 		/* We've more or less seen every task there is, call it quits */
11328 		if (env->loop > env->loop_max)
11329 			break;
11330 
11331 		/* take a breather every nr_migrate tasks */
11332 		if (env->loop > env->loop_break) {
11333 			env->loop_break += SCHED_NR_MIGRATE_BREAK;
11334 			env->flags |= LBF_NEED_BREAK;
11335 			break;
11336 		}
11337 
11338 		p = list_last_entry(tasks, struct task_struct, se.group_node);
11339 
11340 		if (!can_migrate_task(p, env))
11341 			goto next;
11342 
11343 		switch (env->migration_type) {
11344 		case migrate_load:
11345 			/*
11346 			 * Depending of the number of CPUs and tasks and the
11347 			 * cgroup hierarchy, task_h_load() can return a null
11348 			 * value. Make sure that env->imbalance decreases
11349 			 * otherwise detach_tasks() will stop only after
11350 			 * detaching up to loop_max tasks.
11351 			 */
11352 			load = max_t(unsigned long, task_h_load(p), 1);
11353 
11354 			if (sched_feat(LB_MIN) &&
11355 			    load < 16 && !env->sd->nr_balance_failed)
11356 				goto next;
11357 
11358 			/*
11359 			 * Make sure that we don't migrate too much load.
11360 			 * Nevertheless, let relax the constraint if
11361 			 * scheduler fails to find a good waiting task to
11362 			 * migrate.
11363 			 */
11364 			if (shr_bound(load, env->sd->nr_balance_failed) > env->imbalance)
11365 				goto next;
11366 
11367 			env->imbalance -= load;
11368 			break;
11369 
11370 		case migrate_util:
11371 			util = task_util_est(p);
11372 
11373 			if (shr_bound(util, env->sd->nr_balance_failed) > env->imbalance)
11374 				goto next;
11375 
11376 			env->imbalance -= util;
11377 			break;
11378 
11379 		case migrate_task:
11380 			env->imbalance--;
11381 			break;
11382 
11383 		case migrate_misfit:
11384 			/* This is not a misfit task */
11385 			if (task_fits_cpu(p, env->src_cpu))
11386 				goto next;
11387 
11388 			env->imbalance = 0;
11389 			break;
11390 
11391 		case migrate_llc_task:
11392 			env->imbalance--;
11393 			break;
11394 		}
11395 
11396 		detach_task(p, env);
11397 		list_add(&p->se.group_node, &env->tasks);
11398 
11399 		detached++;
11400 
11401 #ifdef CONFIG_PREEMPTION
11402 		/*
11403 		 * NEWIDLE balancing is a source of latency, so preemptible
11404 		 * kernels will stop after the first task is detached to minimize
11405 		 * the critical section.
11406 		 */
11407 		if (env->idle == CPU_NEWLY_IDLE)
11408 			break;
11409 #endif
11410 
11411 		/*
11412 		 * We only want to steal up to the prescribed amount of
11413 		 * load/util/tasks.
11414 		 */
11415 		if (env->imbalance <= 0)
11416 			break;
11417 
11418 		continue;
11419 next:
11420 		if (p->sched_task_hot)
11421 			schedstat_inc(p->stats.nr_failed_migrations_hot);
11422 
11423 		list_move(&p->se.group_node, tasks);
11424 	}
11425 
11426 	/*
11427 	 * Right now, this is one of only two places we collect this stat
11428 	 * so we can safely collect detach_one_task() stats here rather
11429 	 * than inside detach_one_task().
11430 	 */
11431 	schedstat_add(env->sd->lb_gained[env->idle], detached);
11432 
11433 	return detached;
11434 }
11435 
11436 /*
11437  * attach_tasks() -- attaches all tasks detached by detach_tasks() to their
11438  * new rq.
11439  */
11440 static void attach_tasks(struct lb_env *env)
11441 {
11442 	struct list_head *tasks = &env->tasks;
11443 	struct task_struct *p;
11444 	struct rq_flags rf;
11445 
11446 	rq_lock(env->dst_rq, &rf);
11447 	update_rq_clock(env->dst_rq);
11448 
11449 	while (!list_empty(tasks)) {
11450 		p = list_first_entry(tasks, struct task_struct, se.group_node);
11451 		list_del_init(&p->se.group_node);
11452 
11453 		attach_task(env->dst_rq, p);
11454 	}
11455 
11456 	rq_unlock(env->dst_rq, &rf);
11457 }
11458 
11459 #ifdef CONFIG_NO_HZ_COMMON
11460 static inline bool cfs_rq_has_blocked_load(struct cfs_rq *cfs_rq)
11461 {
11462 	if (cfs_rq->avg.load_avg)
11463 		return true;
11464 
11465 	if (cfs_rq->avg.util_avg)
11466 		return true;
11467 
11468 	return false;
11469 }
11470 
11471 static inline bool others_have_blocked(struct rq *rq)
11472 {
11473 	if (cpu_util_rt(rq))
11474 		return true;
11475 
11476 	if (cpu_util_dl(rq))
11477 		return true;
11478 
11479 	if (hw_load_avg(rq))
11480 		return true;
11481 
11482 	if (cpu_util_irq(rq))
11483 		return true;
11484 
11485 	return false;
11486 }
11487 
11488 static inline void update_blocked_load_tick(struct rq *rq)
11489 {
11490 	WRITE_ONCE(rq->last_blocked_load_update_tick, jiffies);
11491 }
11492 
11493 static inline void update_has_blocked_load_status(struct rq *rq, bool has_blocked_load)
11494 {
11495 	if (!has_blocked_load)
11496 		rq->has_blocked_load = 0;
11497 }
11498 #else /* !CONFIG_NO_HZ_COMMON: */
11499 static inline bool cfs_rq_has_blocked_load(struct cfs_rq *cfs_rq) { return false; }
11500 static inline bool others_have_blocked(struct rq *rq) { return false; }
11501 static inline void update_blocked_load_tick(struct rq *rq) {}
11502 static inline void update_has_blocked_load_status(struct rq *rq, bool has_blocked_load) {}
11503 #endif /* !CONFIG_NO_HZ_COMMON */
11504 
11505 static bool __update_blocked_others(struct rq *rq, bool *done)
11506 {
11507 	bool updated;
11508 
11509 	/*
11510 	 * update_load_avg() can call cpufreq_update_util(). Make sure that RT,
11511 	 * DL and IRQ signals have been updated before updating CFS.
11512 	 */
11513 	updated = update_other_load_avgs(rq);
11514 
11515 	if (others_have_blocked(rq))
11516 		*done = false;
11517 
11518 	return updated;
11519 }
11520 
11521 #ifdef CONFIG_FAIR_GROUP_SCHED
11522 
11523 static bool __update_blocked_fair(struct rq *rq, bool *done)
11524 {
11525 	struct cfs_rq *cfs_rq, *pos;
11526 	bool decayed = false;
11527 
11528 	/*
11529 	 * Iterates the task_group tree in a bottom up fashion, see
11530 	 * list_add_leaf_cfs_rq() for details.
11531 	 */
11532 	for_each_leaf_cfs_rq_safe(rq, cfs_rq, pos) {
11533 		struct sched_entity *se;
11534 
11535 		if (update_cfs_rq_load_avg(cfs_rq_clock_pelt(cfs_rq), cfs_rq)) {
11536 			update_tg_load_avg(cfs_rq);
11537 
11538 			if (cfs_rq->nr_queued == 0)
11539 				update_idle_cfs_rq_clock_pelt(cfs_rq);
11540 
11541 			if (cfs_rq == &rq->cfs)
11542 				decayed = true;
11543 		}
11544 
11545 		/* Propagate pending load changes to the parent, if any: */
11546 		se = cfs_rq_se(cfs_rq);
11547 		if (se && !skip_blocked_update(se))
11548 			update_load_avg(cfs_rq_of(se), se, UPDATE_TG);
11549 
11550 		/*
11551 		 * There can be a lot of idle CPU cgroups.  Don't let fully
11552 		 * decayed cfs_rqs linger on the list.
11553 		 */
11554 		if (cfs_rq_is_decayed(cfs_rq))
11555 			list_del_leaf_cfs_rq(cfs_rq);
11556 
11557 		/* Don't need periodic decay once load/util_avg are null */
11558 		if (cfs_rq_has_blocked_load(cfs_rq))
11559 			*done = false;
11560 	}
11561 
11562 	return decayed;
11563 }
11564 
11565 /*
11566  * Compute the hierarchical load factor for cfs_rq and all its ascendants.
11567  * This needs to be done in a top-down fashion because the load of a child
11568  * group is a fraction of its parents load.
11569  */
11570 static void update_cfs_rq_h_load(struct cfs_rq *cfs_rq)
11571 {
11572 	struct sched_entity *se = cfs_rq_se(cfs_rq);
11573 	unsigned long now = jiffies;
11574 	unsigned long load;
11575 
11576 	if (cfs_rq->last_h_load_update == now)
11577 		return;
11578 
11579 	WRITE_ONCE(cfs_rq->h_load_next, NULL);
11580 	for_each_sched_entity(se) {
11581 		cfs_rq = cfs_rq_of(se);
11582 		WRITE_ONCE(cfs_rq->h_load_next, se);
11583 		if (cfs_rq->last_h_load_update == now)
11584 			break;
11585 	}
11586 
11587 	if (!se) {
11588 		cfs_rq->h_load = cfs_rq_load_avg(cfs_rq);
11589 		cfs_rq->last_h_load_update = now;
11590 	}
11591 
11592 	while ((se = READ_ONCE(cfs_rq->h_load_next)) != NULL) {
11593 		load = cfs_rq->h_load;
11594 		load = div64_ul(load * se->avg.load_avg,
11595 				cfs_rq_load_avg(cfs_rq) + 1);
11596 		cfs_rq = group_cfs_rq(se);
11597 		cfs_rq->h_load = load;
11598 		cfs_rq->last_h_load_update = now;
11599 	}
11600 }
11601 
11602 static unsigned long task_h_load(struct task_struct *p)
11603 {
11604 	struct cfs_rq *cfs_rq = task_cfs_rq(p);
11605 
11606 	update_cfs_rq_h_load(cfs_rq);
11607 	return div64_ul(p->se.avg.load_avg * cfs_rq->h_load,
11608 			cfs_rq_load_avg(cfs_rq) + 1);
11609 }
11610 #else /* !CONFIG_FAIR_GROUP_SCHED: */
11611 static bool __update_blocked_fair(struct rq *rq, bool *done)
11612 {
11613 	struct cfs_rq *cfs_rq = &rq->cfs;
11614 	bool decayed;
11615 
11616 	decayed = update_cfs_rq_load_avg(cfs_rq_clock_pelt(cfs_rq), cfs_rq);
11617 	if (cfs_rq_has_blocked_load(cfs_rq))
11618 		*done = false;
11619 
11620 	return decayed;
11621 }
11622 
11623 static unsigned long task_h_load(struct task_struct *p)
11624 {
11625 	return p->se.avg.load_avg;
11626 }
11627 #endif /* !CONFIG_FAIR_GROUP_SCHED */
11628 
11629 static void __sched_balance_update_blocked_averages(struct rq *rq)
11630 {
11631 	bool decayed = false, done = true;
11632 
11633 	update_blocked_load_tick(rq);
11634 
11635 	decayed |= __update_blocked_others(rq, &done);
11636 	decayed |= __update_blocked_fair(rq, &done);
11637 
11638 	update_has_blocked_load_status(rq, !done);
11639 	if (decayed)
11640 		cpufreq_update_util(rq, 0);
11641 }
11642 
11643 static void sched_balance_update_blocked_averages(int cpu)
11644 {
11645 	struct rq *rq = cpu_rq(cpu);
11646 
11647 	guard(rq_lock_irqsave)(rq);
11648 	update_rq_clock(rq);
11649 	__sched_balance_update_blocked_averages(rq);
11650 }
11651 
11652 /********** Helpers for sched_balance_find_src_group ************************/
11653 
11654 /*
11655  * sg_lb_stats - stats of a sched_group required for load-balancing:
11656  */
11657 struct sg_lb_stats {
11658 	unsigned long avg_load;			/* Avg load            over the CPUs of the group */
11659 	unsigned long group_load;		/* Total load          over the CPUs of the group */
11660 	unsigned long group_capacity;		/* Capacity            over the CPUs of the group */
11661 	unsigned long group_util;		/* Total utilization   over the CPUs of the group */
11662 	unsigned long group_runnable;		/* Total runnable time over the CPUs of the group */
11663 	unsigned int sum_nr_running;		/* Nr of all tasks running in the group */
11664 	unsigned int sum_h_nr_running;		/* Nr of CFS tasks running in the group */
11665 	unsigned int idle_cpus;                 /* Nr of idle CPUs         in the group */
11666 	unsigned int group_weight;
11667 	enum group_type group_type;
11668 	unsigned int group_asym_packing;	/* Tasks should be moved to preferred CPU */
11669 	unsigned int group_smt_balance;		/* Task on busy SMT be moved */
11670 	unsigned int group_llc_balance;		/* Tasks should be moved to preferred LLC */
11671 	unsigned long group_misfit_task_load;	/* A CPU has a task too big for its capacity */
11672 	unsigned int group_overutilized;	/* At least one CPU is overutilized in the group */
11673 #ifdef CONFIG_NUMA_BALANCING
11674 	unsigned int nr_numa_running;
11675 	unsigned int nr_preferred_running;
11676 #endif
11677 #ifdef CONFIG_SCHED_CACHE
11678 	unsigned int nr_pref_dst_llc;
11679 #endif
11680 };
11681 
11682 /*
11683  * sd_lb_stats - stats of a sched_domain required for load-balancing:
11684  */
11685 struct sd_lb_stats {
11686 	struct sched_group *busiest;		/* Busiest group in this sd */
11687 	struct sched_group *local;		/* Local group in this sd */
11688 	unsigned long total_load;		/* Total load of all groups in sd */
11689 	unsigned long total_capacity;		/* Total capacity of all groups in sd */
11690 	unsigned long avg_load;			/* Average load across all groups in sd */
11691 	unsigned int prefer_sibling;		/* Tasks should go to sibling first */
11692 
11693 	struct sg_lb_stats busiest_stat;	/* Statistics of the busiest group */
11694 	struct sg_lb_stats local_stat;		/* Statistics of the local group */
11695 };
11696 
11697 static inline void init_sd_lb_stats(struct sd_lb_stats *sds)
11698 {
11699 	/*
11700 	 * Skimp on the clearing to avoid duplicate work. We can avoid clearing
11701 	 * local_stat because update_sg_lb_stats() does a full clear/assignment.
11702 	 * We must however set busiest_stat::group_type and
11703 	 * busiest_stat::idle_cpus to the worst busiest group because
11704 	 * update_sd_pick_busiest() reads these before assignment.
11705 	 */
11706 	*sds = (struct sd_lb_stats){
11707 		.busiest = NULL,
11708 		.local = NULL,
11709 		.total_load = 0UL,
11710 		.total_capacity = 0UL,
11711 		.busiest_stat = {
11712 			.idle_cpus = UINT_MAX,
11713 			.group_type = group_has_spare,
11714 		},
11715 	};
11716 }
11717 
11718 static unsigned long scale_rt_capacity(int cpu)
11719 {
11720 	unsigned long max = get_actual_cpu_capacity(cpu);
11721 	struct rq *rq = cpu_rq(cpu);
11722 	unsigned long used, free;
11723 	unsigned long irq;
11724 
11725 	irq = cpu_util_irq(rq);
11726 
11727 	if (unlikely(irq >= max))
11728 		return 1;
11729 
11730 	/*
11731 	 * avg_rt.util_avg and avg_dl.util_avg track binary signals
11732 	 * (running and not running) with weights 0 and 1024 respectively.
11733 	 */
11734 	used = cpu_util_rt(rq);
11735 	used += cpu_util_dl(rq);
11736 
11737 	if (unlikely(used >= max))
11738 		return 1;
11739 
11740 	free = max - used;
11741 
11742 	return scale_irq_capacity(free, irq, max);
11743 }
11744 
11745 static void update_cpu_capacity(struct sched_domain *sd, int cpu)
11746 {
11747 	unsigned long capacity = scale_rt_capacity(cpu);
11748 	struct sched_group *sdg = sd->groups;
11749 
11750 	if (!capacity)
11751 		capacity = 1;
11752 
11753 	cpu_rq(cpu)->cpu_capacity = capacity;
11754 	trace_sched_cpu_capacity_tp(cpu_rq(cpu));
11755 
11756 	sdg->sgc->capacity = capacity;
11757 	sdg->sgc->min_capacity = capacity;
11758 	sdg->sgc->max_capacity = capacity;
11759 }
11760 
11761 void update_group_capacity(struct sched_domain *sd, int cpu)
11762 {
11763 	struct sched_domain *child = sd->child;
11764 	struct sched_group *group, *sdg = sd->groups;
11765 	unsigned long capacity, min_capacity, max_capacity;
11766 	unsigned long interval;
11767 
11768 	interval = msecs_to_jiffies(sd->balance_interval);
11769 	interval = clamp(interval, 1UL, max_load_balance_interval);
11770 	sdg->sgc->next_update = jiffies + interval;
11771 
11772 	if (!child) {
11773 		update_cpu_capacity(sd, cpu);
11774 		return;
11775 	}
11776 
11777 	capacity = 0;
11778 	min_capacity = ULONG_MAX;
11779 	max_capacity = 0;
11780 
11781 	if (child->flags & SD_NUMA) {
11782 		/*
11783 		 * SD_NUMA domains cannot assume that child groups
11784 		 * span the current group.
11785 		 */
11786 
11787 		for_each_cpu(cpu, sched_group_span(sdg)) {
11788 			unsigned long cpu_cap = capacity_of(cpu);
11789 
11790 			capacity += cpu_cap;
11791 			min_capacity = min(cpu_cap, min_capacity);
11792 			max_capacity = max(cpu_cap, max_capacity);
11793 		}
11794 	} else  {
11795 		/*
11796 		 * !SD_NUMA domains can assume that child groups
11797 		 * span the current group.
11798 		 */
11799 
11800 		group = child->groups;
11801 		do {
11802 			struct sched_group_capacity *sgc = group->sgc;
11803 
11804 			capacity += sgc->capacity;
11805 			min_capacity = min(sgc->min_capacity, min_capacity);
11806 			max_capacity = max(sgc->max_capacity, max_capacity);
11807 			group = group->next;
11808 		} while (group != child->groups);
11809 	}
11810 
11811 	sdg->sgc->capacity = capacity;
11812 	sdg->sgc->min_capacity = min_capacity;
11813 	sdg->sgc->max_capacity = max_capacity;
11814 }
11815 
11816 /*
11817  * Check whether the capacity of the rq has been noticeably reduced by side
11818  * activity. The imbalance_pct is used for the threshold.
11819  * Return true is the capacity is reduced
11820  */
11821 static inline int
11822 check_cpu_capacity(struct rq *rq, struct sched_domain *sd)
11823 {
11824 	return ((rq->cpu_capacity * sd->imbalance_pct) <
11825 				(arch_scale_cpu_capacity(cpu_of(rq)) * 100));
11826 }
11827 
11828 /* Check if the rq has a misfit task */
11829 static inline bool check_misfit_status(struct rq *rq)
11830 {
11831 	return rq->misfit_task_load;
11832 }
11833 
11834 /*
11835  * Group imbalance indicates (and tries to solve) the problem where balancing
11836  * groups is inadequate due to ->cpus_ptr constraints.
11837  *
11838  * Imagine a situation of two groups of 4 CPUs each and 4 tasks each with a
11839  * cpumask covering 1 CPU of the first group and 3 CPUs of the second group.
11840  * Something like:
11841  *
11842  *	{ 0 1 2 3 } { 4 5 6 7 }
11843  *	        *     * * *
11844  *
11845  * If we were to balance group-wise we'd place two tasks in the first group and
11846  * two tasks in the second group. Clearly this is undesired as it will overload
11847  * cpu 3 and leave one of the CPUs in the second group unused.
11848  *
11849  * The current solution to this issue is detecting the skew in the first group
11850  * by noticing the lower domain failed to reach balance and had difficulty
11851  * moving tasks due to affinity constraints.
11852  *
11853  * When this is so detected; this group becomes a candidate for busiest; see
11854  * update_sd_pick_busiest(). And calculate_imbalance() and
11855  * sched_balance_find_src_group() avoid some of the usual balance conditions to allow it
11856  * to create an effective group imbalance.
11857  *
11858  * This is a somewhat tricky proposition since the next run might not find the
11859  * group imbalance and decide the groups need to be balanced again. A most
11860  * subtle and fragile situation.
11861  */
11862 
11863 static inline int sg_imbalanced(struct sched_group *group)
11864 {
11865 	return group->sgc->imbalance;
11866 }
11867 
11868 /*
11869  * group_has_capacity returns true if the group has spare capacity that could
11870  * be used by some tasks.
11871  * We consider that a group has spare capacity if the number of task is
11872  * smaller than the number of CPUs or if the utilization is lower than the
11873  * available capacity for CFS tasks.
11874  * For the latter, we use a threshold to stabilize the state, to take into
11875  * account the variance of the tasks' load and to return true if the available
11876  * capacity in meaningful for the load balancer.
11877  * As an example, an available capacity of 1% can appear but it doesn't make
11878  * any benefit for the load balance.
11879  */
11880 static inline bool
11881 group_has_capacity(unsigned int imbalance_pct, struct sg_lb_stats *sgs)
11882 {
11883 	if (sgs->sum_nr_running < sgs->group_weight)
11884 		return true;
11885 
11886 	if ((sgs->group_capacity * imbalance_pct) <
11887 			(sgs->group_runnable * 100))
11888 		return false;
11889 
11890 	if ((sgs->group_capacity * 100) >
11891 			(sgs->group_util * imbalance_pct))
11892 		return true;
11893 
11894 	return false;
11895 }
11896 
11897 /*
11898  *  group_is_overloaded returns true if the group has more tasks than it can
11899  *  handle.
11900  *  group_is_overloaded is not equals to !group_has_capacity because a group
11901  *  with the exact right number of tasks, has no more spare capacity but is not
11902  *  overloaded so both group_has_capacity and group_is_overloaded return
11903  *  false.
11904  */
11905 static inline bool
11906 group_is_overloaded(unsigned int imbalance_pct, struct sg_lb_stats *sgs)
11907 {
11908 	/*
11909 	 * With EAS and uclamp, 1 CPU in the group must be overutilized to
11910 	 * consider the group overloaded.
11911 	 */
11912 	if (sched_energy_enabled() && !sgs->group_overutilized)
11913 		return false;
11914 
11915 	if (sgs->sum_nr_running <= sgs->group_weight)
11916 		return false;
11917 
11918 	if ((sgs->group_capacity * 100) <
11919 			(sgs->group_util * imbalance_pct))
11920 		return true;
11921 
11922 	if ((sgs->group_capacity * imbalance_pct) <
11923 			(sgs->group_runnable * 100))
11924 		return true;
11925 
11926 	return false;
11927 }
11928 
11929 static inline enum
11930 group_type group_classify(unsigned int imbalance_pct,
11931 			  struct sched_group *group,
11932 			  struct sg_lb_stats *sgs)
11933 {
11934 	if (group_is_overloaded(imbalance_pct, sgs))
11935 		return group_overloaded;
11936 
11937 	if (sgs->group_llc_balance)
11938 		return group_llc_balance;
11939 
11940 	if (sg_imbalanced(group))
11941 		return group_imbalanced;
11942 
11943 	if (sgs->group_asym_packing)
11944 		return group_asym_packing;
11945 
11946 	if (sgs->group_smt_balance)
11947 		return group_smt_balance;
11948 
11949 	if (sgs->group_misfit_task_load)
11950 		return group_misfit_task;
11951 
11952 	if (!group_has_capacity(imbalance_pct, sgs))
11953 		return group_fully_busy;
11954 
11955 	return group_has_spare;
11956 }
11957 
11958 /**
11959  * sched_use_asym_prio - Check whether asym_packing priority must be used
11960  * @sd:		The scheduling domain of the load balancing
11961  * @cpu:	A CPU
11962  *
11963  * Always use CPU priority when balancing load between SMT siblings. When
11964  * balancing load between cores, it is not sufficient that @cpu is idle. Only
11965  * use CPU priority if the whole core is idle.
11966  *
11967  * Returns: True if the priority of @cpu must be followed. False otherwise.
11968  */
11969 static bool sched_use_asym_prio(struct sched_domain *sd, int cpu)
11970 {
11971 	if (!(sd->flags & SD_ASYM_PACKING))
11972 		return false;
11973 
11974 	if (!sched_smt_active())
11975 		return true;
11976 
11977 	return sd->flags & SD_SHARE_CPUCAPACITY || is_core_idle(cpu);
11978 }
11979 
11980 static inline bool sched_asym(struct sched_domain *sd, int dst_cpu, int src_cpu)
11981 {
11982 	/*
11983 	 * First check if @dst_cpu can do asym_packing load balance. Only do it
11984 	 * if it has higher priority than @src_cpu.
11985 	 */
11986 	return sched_use_asym_prio(sd, dst_cpu) &&
11987 		sched_asym_prefer(dst_cpu, src_cpu);
11988 }
11989 
11990 /**
11991  * sched_group_asym - Check if the destination CPU can do asym_packing balance
11992  * @env:	The load balancing environment
11993  * @sgs:	Load-balancing statistics of the candidate busiest group
11994  * @group:	The candidate busiest group
11995  *
11996  * @env::dst_cpu can do asym_packing if it has higher priority than the
11997  * preferred CPU of @group.
11998  *
11999  * Return: true if @env::dst_cpu can do with asym_packing load balance. False
12000  * otherwise.
12001  */
12002 static inline bool
12003 sched_group_asym(struct lb_env *env, struct sg_lb_stats *sgs, struct sched_group *group)
12004 {
12005 	/*
12006 	 * CPU priorities do not make sense for SMT cores with more than one
12007 	 * busy sibling.
12008 	 */
12009 	if ((group->flags & SD_SHARE_CPUCAPACITY) &&
12010 	    (sgs->group_weight - sgs->idle_cpus != 1))
12011 		return false;
12012 
12013 	return sched_asym(env->sd, env->dst_cpu, READ_ONCE(group->asym_prefer_cpu));
12014 }
12015 
12016 /* One group has more than one SMT CPU while the other group does not */
12017 static inline bool smt_vs_nonsmt_groups(struct sched_group *sg1,
12018 				    struct sched_group *sg2)
12019 {
12020 	if (!sg1 || !sg2)
12021 		return false;
12022 
12023 	return (sg1->flags & SD_SHARE_CPUCAPACITY) !=
12024 		(sg2->flags & SD_SHARE_CPUCAPACITY);
12025 }
12026 
12027 static inline bool smt_balance(struct lb_env *env, struct sg_lb_stats *sgs,
12028 			       struct sched_group *group)
12029 {
12030 	if (!env->idle)
12031 		return false;
12032 
12033 	/*
12034 	 * For SMT source group, it is better to move a task
12035 	 * to a CPU that doesn't have multiple tasks sharing its CPU capacity.
12036 	 * Note that if a group has a single SMT, SD_SHARE_CPUCAPACITY
12037 	 * will not be on.
12038 	 */
12039 	if (group->flags & SD_SHARE_CPUCAPACITY &&
12040 	    sgs->sum_h_nr_running > 1)
12041 		return true;
12042 
12043 	return false;
12044 }
12045 
12046 static inline long sibling_imbalance(struct lb_env *env,
12047 				    struct sd_lb_stats *sds,
12048 				    struct sg_lb_stats *busiest,
12049 				    struct sg_lb_stats *local)
12050 {
12051 	int ncores_busiest, ncores_local;
12052 	long imbalance;
12053 
12054 	if (!env->idle || !busiest->sum_nr_running)
12055 		return 0;
12056 
12057 	ncores_busiest = sds->busiest->cores;
12058 	ncores_local = sds->local->cores;
12059 
12060 	if (ncores_busiest == ncores_local) {
12061 		imbalance = busiest->sum_nr_running;
12062 		lsub_positive(&imbalance, local->sum_nr_running);
12063 		return imbalance;
12064 	}
12065 
12066 	/* Balance such that nr_running/ncores ratio are same on both groups */
12067 	imbalance = ncores_local * busiest->sum_nr_running;
12068 	lsub_positive(&imbalance, ncores_busiest * local->sum_nr_running);
12069 	/* Normalize imbalance and do rounding on normalization */
12070 	imbalance = 2 * imbalance + ncores_local + ncores_busiest;
12071 	imbalance /= ncores_local + ncores_busiest;
12072 
12073 	/* Take advantage of resource in an empty sched group */
12074 	if (imbalance <= 1 && local->sum_nr_running == 0 &&
12075 	    busiest->sum_nr_running > 1)
12076 		imbalance = 2;
12077 
12078 	return imbalance;
12079 }
12080 
12081 static inline bool
12082 sched_reduced_capacity(struct rq *rq, struct sched_domain *sd)
12083 {
12084 	/*
12085 	 * When there is more than 1 task, the group_overloaded case already
12086 	 * takes care of cpu with reduced capacity
12087 	 */
12088 	if (rq->cfs.h_nr_runnable != 1)
12089 		return false;
12090 
12091 	return check_cpu_capacity(rq, sd);
12092 }
12093 
12094 #ifdef CONFIG_SCHED_CACHE
12095 /*
12096  * Record the statistics for this scheduler group for later
12097  * use. These values guide load balancing on aggregating tasks
12098  * to a LLC.
12099  */
12100 static void record_sg_llc_stats(struct lb_env *env,
12101 				struct sg_lb_stats *sgs,
12102 				struct sched_group *group)
12103 {
12104 	struct sched_domain_shared *sd_share;
12105 	int cpu;
12106 
12107 	if (!sched_cache_enabled() || env->idle == CPU_NEWLY_IDLE)
12108 		return;
12109 
12110 	/* Only care about sched domain spanning multiple LLCs */
12111 	if (env->sd->child != rcu_dereference_all(per_cpu(sd_llc, env->dst_cpu)))
12112 		return;
12113 
12114 	/*
12115 	 * At this point we know this group spans a LLC domain.
12116 	 * Record the statistic of this group in its corresponding
12117 	 * shared LLC domain.
12118 	 * Note: sd_share cannot be obtained via sd->child->shared,
12119 	 * because the latter refers to the domain that covers the
12120 	 * local group. Instead, sd_share should be located using
12121 	 * the first CPU of the LLC group.
12122 	 */
12123 	cpu = cpumask_first(sched_group_span(group));
12124 	sd_share = rcu_dereference_all(per_cpu(sd_llc_shared, cpu));
12125 	if (!sd_share)
12126 		return;
12127 
12128 	if (READ_ONCE(sd_share->util_avg) != sgs->group_util)
12129 		WRITE_ONCE(sd_share->util_avg, sgs->group_util);
12130 
12131 	if (unlikely(READ_ONCE(sd_share->capacity) != sgs->group_capacity))
12132 		WRITE_ONCE(sd_share->capacity, sgs->group_capacity);
12133 }
12134 
12135 /*
12136  * Do LLC balance on sched group that contains LLC, and have tasks preferring
12137  * to run on LLC in idle dst_cpu.
12138  */
12139 static inline bool llc_balance(struct lb_env *env, struct sg_lb_stats *sgs,
12140 			       struct sched_group *group)
12141 {
12142 	if (!sched_cache_enabled())
12143 		return false;
12144 
12145 	if (env->sd->flags & SD_SHARE_LLC)
12146 		return false;
12147 
12148 	/*
12149 	 * On asymmetric domains, group_misfit_task_load
12150 	 * should be prioritized to move tasks to CPU that fit them
12151 	 * over aggregating tasks to their preferred LLC.
12152 	 */
12153 	if ((env->sd->flags & SD_ASYM_CPUCAPACITY) &&
12154 	    sgs->group_misfit_task_load)
12155 		return false;
12156 
12157 	/*
12158 	 * Skip cache aware tagging if nr_balanced_failed is sufficiently high.
12159 	 * Threshold of cache_nice_tries is set to 1 higher than nr_balance_failed
12160 	 * to avoid excessive task migration at the same time.
12161 	 */
12162 	if (env->sd->nr_balance_failed >= env->sd->cache_nice_tries + 1)
12163 		return false;
12164 
12165 	if (sgs->nr_pref_dst_llc &&
12166 	    can_migrate_llc(cpumask_first(sched_group_span(group)),
12167 			    env->dst_cpu, 0, true) == mig_llc)
12168 		return true;
12169 
12170 	return false;
12171 }
12172 
12173 static bool update_llc_busiest(struct lb_env *env,
12174 			       struct sg_lb_stats *busiest,
12175 			       struct sg_lb_stats *sgs)
12176 {
12177 	/*
12178 	 * There are more tasks that want to run on dst_cpu's LLC.
12179 	 */
12180 	return sgs->nr_pref_dst_llc > busiest->nr_pref_dst_llc;
12181 }
12182 #else
12183 static inline void record_sg_llc_stats(struct lb_env *env, struct sg_lb_stats *sgs,
12184 				       struct sched_group *group)
12185 {
12186 }
12187 
12188 static inline bool llc_balance(struct lb_env *env, struct sg_lb_stats *sgs,
12189 			       struct sched_group *group)
12190 {
12191 	return false;
12192 }
12193 
12194 static bool update_llc_busiest(struct lb_env *env,
12195 			       struct sg_lb_stats *busiest,
12196 			       struct sg_lb_stats *sgs)
12197 {
12198 	return false;
12199 }
12200 #endif
12201 
12202 /**
12203  * update_sg_lb_stats - Update sched_group's statistics for load balancing.
12204  * @env: The load balancing environment.
12205  * @sds: Load-balancing data with statistics of the local group.
12206  * @group: sched_group whose statistics are to be updated.
12207  * @sgs: variable to hold the statistics for this group.
12208  * @sg_overloaded: sched_group is overloaded
12209  */
12210 static inline void update_sg_lb_stats(struct lb_env *env,
12211 				      struct sd_lb_stats *sds,
12212 				      struct sched_group *group,
12213 				      struct sg_lb_stats *sgs,
12214 				      bool *sg_overloaded)
12215 {
12216 	int i, nr_running, local_group, sd_flags = env->sd->flags;
12217 	bool balancing_at_rd = !env->sd->parent;
12218 
12219 	memset(sgs, 0, sizeof(*sgs));
12220 
12221 	local_group = group == sds->local;
12222 
12223 	for_each_cpu_and(i, sched_group_span(group), env->cpus) {
12224 		struct rq *rq = cpu_rq(i);
12225 		unsigned long load = cpu_load(rq);
12226 
12227 		sgs->group_load += load;
12228 		sgs->group_util += cpu_util_cfs(i);
12229 		sgs->group_runnable += cpu_runnable(rq);
12230 		sgs->sum_h_nr_running += rq->cfs.h_nr_runnable;
12231 
12232 		nr_running = rq->nr_running;
12233 		sgs->sum_nr_running += nr_running;
12234 
12235 		if (cpu_overutilized(i))
12236 			sgs->group_overutilized = 1;
12237 
12238 #ifdef CONFIG_SCHED_CACHE
12239 		if (sched_cache_enabled()) {
12240 			struct sched_domain *sd_tmp;
12241 			int dst_llc;
12242 
12243 			dst_llc = llc_id(env->dst_cpu);
12244 			if (llc_id(i) != dst_llc) {
12245 				sd_tmp = rcu_dereference_all(rq->sd);
12246 				if (sd_tmp && (unsigned int)dst_llc < sd_tmp->llc_max)
12247 					sgs->nr_pref_dst_llc += sd_tmp->llc_counts[dst_llc];
12248 			}
12249 		}
12250 #endif
12251 
12252 		/*
12253 		 * No need to call idle_cpu() if nr_running is not 0
12254 		 */
12255 		if (!nr_running && idle_cpu(i)) {
12256 			sgs->idle_cpus++;
12257 			/* Idle cpu can't have misfit task */
12258 			continue;
12259 		}
12260 
12261 		/* Overload indicator is only updated at root domain */
12262 		if (balancing_at_rd && nr_running > 1)
12263 			*sg_overloaded = 1;
12264 
12265 #ifdef CONFIG_NUMA_BALANCING
12266 		/* Only fbq_classify_group() uses this to classify NUMA groups */
12267 		if (sd_flags & SD_NUMA) {
12268 			sgs->nr_numa_running += rq->nr_numa_running;
12269 			sgs->nr_preferred_running += rq->nr_preferred_running;
12270 		}
12271 #endif
12272 		if (local_group)
12273 			continue;
12274 
12275 		if (sd_flags & SD_ASYM_CPUCAPACITY) {
12276 			if (rq->misfit_task_load) {
12277 				/*
12278 				 * Always mark the root domain overloaded so big
12279 				 * CPUs can pick up misfit tasks via newly idle
12280 				 * balance.
12281 				 */
12282 				if (balancing_at_rd)
12283 					*sg_overloaded = 1;
12284 
12285 				/*
12286 				 * Only account misfit load if @dst_cpu can
12287 				 * help; otherwise, the group may be classified
12288 				 * as misfit_task and update_sd_pick_busiest()
12289 				 * will skip it.
12290 				 */
12291 				if (capacity_greater(capacity_of(env->dst_cpu),
12292 						     group->sgc->max_capacity) &&
12293 				    (sgs->group_misfit_task_load < rq->misfit_task_load))
12294 					sgs->group_misfit_task_load = rq->misfit_task_load;
12295 			}
12296 		} else if (env->idle && sched_reduced_capacity(rq, env->sd)) {
12297 			/* Check for a task running on a CPU with reduced capacity */
12298 			if (sgs->group_misfit_task_load < load)
12299 				sgs->group_misfit_task_load = load;
12300 		}
12301 	}
12302 
12303 	sgs->group_capacity = group->sgc->capacity;
12304 
12305 	sgs->group_weight = group->group_weight;
12306 
12307 	if (!local_group) {
12308 		/* Check if dst CPU is idle and preferred to this group */
12309 		if (env->idle && sgs->sum_h_nr_running &&
12310 		    sched_group_asym(env, sgs, group))
12311 			sgs->group_asym_packing = 1;
12312 
12313 		/* Check for loaded SMT group to be balanced to dst CPU */
12314 		if (smt_balance(env, sgs, group))
12315 			sgs->group_smt_balance = 1;
12316 
12317 		/* Check for tasks in this group can be moved to their preferred LLC */
12318 		if (llc_balance(env, sgs, group))
12319 			sgs->group_llc_balance = 1;
12320 	}
12321 
12322 	sgs->group_type = group_classify(env->sd->imbalance_pct, group, sgs);
12323 
12324 	record_sg_llc_stats(env, sgs, group);
12325 	/* Computing avg_load makes sense only when group is overloaded */
12326 	if (sgs->group_type == group_overloaded)
12327 		sgs->avg_load = (sgs->group_load * SCHED_CAPACITY_SCALE) /
12328 				sgs->group_capacity;
12329 }
12330 
12331 /**
12332  * update_sd_pick_busiest - return 1 on busiest group
12333  * @env: The load balancing environment.
12334  * @sds: sched_domain statistics
12335  * @sg: sched_group candidate to be checked for being the busiest
12336  * @sgs: sched_group statistics
12337  *
12338  * Determine if @sg is a busier group than the previously selected
12339  * busiest group.
12340  *
12341  * Return: %true if @sg is a busier group than the previously selected
12342  * busiest group. %false otherwise.
12343  */
12344 static bool update_sd_pick_busiest(struct lb_env *env,
12345 				   struct sd_lb_stats *sds,
12346 				   struct sched_group *sg,
12347 				   struct sg_lb_stats *sgs)
12348 {
12349 	struct sg_lb_stats *busiest = &sds->busiest_stat;
12350 
12351 	/* Make sure that there is at least one task to pull */
12352 	if (!sgs->sum_h_nr_running)
12353 		return false;
12354 
12355 	/*
12356 	 * Don't try to pull misfit tasks we can't help.
12357 	 * We can use max_capacity here as reduction in capacity on some
12358 	 * CPUs in the group should either be possible to resolve
12359 	 * internally or be covered by avg_load imbalance (eventually).
12360 	 *
12361 	 * When SMT is active, only pull a misfit to dst_cpu if it is on a
12362 	 * fully idle core; otherwise the effective capacity of the core is
12363 	 * reduced and we may not actually provide more capacity than the
12364 	 * source.
12365 	 */
12366 	if ((env->sd->flags & SD_ASYM_CPUCAPACITY) &&
12367 	    (sgs->group_type == group_misfit_task) &&
12368 	    (!env->dst_core_idle ||
12369 	     !capacity_greater(capacity_of(env->dst_cpu), sg->sgc->max_capacity) ||
12370 	     sds->local_stat.group_type != group_has_spare))
12371 		return false;
12372 
12373 	/*
12374 	 * Candidate sg has no more than one task per CPU and has higher
12375 	 * per-CPU capacity. Migrating tasks to less capable CPUs may harm
12376 	 * throughput. Maximize throughput, power/energy consequences are not
12377 	 * considered.
12378 	 */
12379 	if ((env->sd->flags & SD_ASYM_CPUCAPACITY) &&
12380 	    (sgs->group_type <= group_fully_busy) &&
12381 	    (capacity_greater(sg->sgc->min_capacity, capacity_of(env->dst_cpu))))
12382 		return false;
12383 
12384 	if (sgs->group_type > busiest->group_type)
12385 		return true;
12386 
12387 	if (sgs->group_type < busiest->group_type)
12388 		return false;
12389 
12390 	/*
12391 	 * The candidate and the current busiest group are the same type of
12392 	 * group. Let check which one is the busiest according to the type.
12393 	 */
12394 
12395 	switch (sgs->group_type) {
12396 	case group_overloaded:
12397 		/* Select the overloaded group with highest avg_load. */
12398 		return sgs->avg_load > busiest->avg_load;
12399 
12400 	case group_llc_balance:
12401 		/* Select the group with most tasks preferring dst LLC */
12402 		return update_llc_busiest(env, busiest, sgs);
12403 
12404 	case group_imbalanced:
12405 		/*
12406 		 * Select the 1st imbalanced group as we don't have any way to
12407 		 * choose one more than another.
12408 		 */
12409 		return false;
12410 
12411 	case group_asym_packing:
12412 		/* Prefer to move from lowest priority CPU's work */
12413 		return sched_asym_prefer(READ_ONCE(sds->busiest->asym_prefer_cpu),
12414 					 READ_ONCE(sg->asym_prefer_cpu));
12415 
12416 	case group_misfit_task:
12417 		/*
12418 		 * If we have more than one misfit sg go with the biggest
12419 		 * misfit.
12420 		 */
12421 		return sgs->group_misfit_task_load > busiest->group_misfit_task_load;
12422 
12423 	case group_smt_balance:
12424 		/*
12425 		 * Check if we have spare CPUs on either SMT group to
12426 		 * choose has spare or fully busy handling.
12427 		 */
12428 		if (sgs->idle_cpus != 0 || busiest->idle_cpus != 0)
12429 			goto has_spare;
12430 
12431 		fallthrough;
12432 
12433 	case group_fully_busy:
12434 		/*
12435 		 * Select the fully busy group with highest avg_load. In
12436 		 * theory, there is no need to pull task from such kind of
12437 		 * group because tasks have all compute capacity that they need
12438 		 * but we can still improve the overall throughput by reducing
12439 		 * contention when accessing shared HW resources.
12440 		 *
12441 		 * XXX for now avg_load is not computed and always 0 so we
12442 		 * select the 1st one, except if @sg is composed of SMT
12443 		 * siblings.
12444 		 */
12445 
12446 		if (sgs->avg_load < busiest->avg_load)
12447 			return false;
12448 
12449 		if (sgs->avg_load == busiest->avg_load) {
12450 			/*
12451 			 * SMT sched groups need more help than non-SMT groups.
12452 			 * If @sg happens to also be SMT, either choice is good.
12453 			 */
12454 			if (sds->busiest->flags & SD_SHARE_CPUCAPACITY)
12455 				return false;
12456 		}
12457 
12458 		break;
12459 
12460 	case group_has_spare:
12461 		/*
12462 		 * Do not pick sg with SMT CPUs over sg with pure CPUs,
12463 		 * as we do not want to pull task off SMT core with one task
12464 		 * and make the core idle.
12465 		 */
12466 		if (smt_vs_nonsmt_groups(sds->busiest, sg)) {
12467 			if (sg->flags & SD_SHARE_CPUCAPACITY && sgs->sum_h_nr_running <= 1)
12468 				return false;
12469 			else
12470 				return true;
12471 		}
12472 has_spare:
12473 
12474 		/*
12475 		 * Select not overloaded group with lowest number of idle CPUs
12476 		 * and highest number of running tasks. We could also compare
12477 		 * the spare capacity which is more stable but it can end up
12478 		 * that the group has less spare capacity but finally more idle
12479 		 * CPUs which means less opportunity to pull tasks.
12480 		 */
12481 		if (sgs->idle_cpus > busiest->idle_cpus)
12482 			return false;
12483 		else if ((sgs->idle_cpus == busiest->idle_cpus) &&
12484 			 (sgs->sum_nr_running <= busiest->sum_nr_running))
12485 			return false;
12486 
12487 		break;
12488 	}
12489 
12490 	return true;
12491 }
12492 
12493 #ifdef CONFIG_NUMA_BALANCING
12494 static inline enum fbq_type fbq_classify_group(struct sg_lb_stats *sgs)
12495 {
12496 	if (sgs->sum_h_nr_running > sgs->nr_numa_running)
12497 		return regular;
12498 	if (sgs->sum_h_nr_running > sgs->nr_preferred_running)
12499 		return remote;
12500 	return all;
12501 }
12502 
12503 static inline enum fbq_type fbq_classify_rq(struct rq *rq)
12504 {
12505 	if (rq->nr_running > rq->nr_numa_running)
12506 		return regular;
12507 	if (rq->nr_running > rq->nr_preferred_running)
12508 		return remote;
12509 	return all;
12510 }
12511 #else /* !CONFIG_NUMA_BALANCING: */
12512 static inline enum fbq_type fbq_classify_group(struct sg_lb_stats *sgs)
12513 {
12514 	return all;
12515 }
12516 
12517 static inline enum fbq_type fbq_classify_rq(struct rq *rq)
12518 {
12519 	return regular;
12520 }
12521 #endif /* !CONFIG_NUMA_BALANCING */
12522 
12523 
12524 struct sg_lb_stats;
12525 
12526 /*
12527  * task_running_on_cpu - return 1 if @p is running on @cpu.
12528  */
12529 
12530 static unsigned int task_running_on_cpu(int cpu, struct task_struct *p)
12531 {
12532 	/* Task has no contribution or is new */
12533 	if (cpu != task_cpu(p) || !READ_ONCE(p->se.avg.last_update_time))
12534 		return 0;
12535 
12536 	if (task_on_rq_queued(p))
12537 		return 1;
12538 
12539 	return 0;
12540 }
12541 
12542 /**
12543  * idle_cpu_without - would a given CPU be idle without p ?
12544  * @cpu: the processor on which idleness is tested.
12545  * @p: task which should be ignored.
12546  *
12547  * Return: 1 if the CPU would be idle. 0 otherwise.
12548  */
12549 static int idle_cpu_without(int cpu, struct task_struct *p)
12550 {
12551 	struct rq *rq = cpu_rq(cpu);
12552 
12553 	if (rq->curr != rq->idle && rq->curr != p)
12554 		return 0;
12555 
12556 	/*
12557 	 * rq->nr_running can't be used but an updated version without the
12558 	 * impact of p on cpu must be used instead. The updated nr_running
12559 	 * be computed and tested before calling idle_cpu_without().
12560 	 */
12561 
12562 	if (rq->ttwu_pending)
12563 		return 0;
12564 
12565 	return 1;
12566 }
12567 
12568 /*
12569  * update_sg_wakeup_stats - Update sched_group's statistics for wakeup.
12570  * @sd: The sched_domain level to look for idlest group.
12571  * @group: sched_group whose statistics are to be updated.
12572  * @sgs: variable to hold the statistics for this group.
12573  * @p: The task for which we look for the idlest group/CPU.
12574  */
12575 static inline void update_sg_wakeup_stats(struct sched_domain *sd,
12576 					  struct sched_group *group,
12577 					  struct sg_lb_stats *sgs,
12578 					  struct task_struct *p)
12579 {
12580 	int i, nr_running;
12581 
12582 	memset(sgs, 0, sizeof(*sgs));
12583 
12584 	/* Assume that task can't fit any CPU of the group */
12585 	if (sd->flags & SD_ASYM_CPUCAPACITY)
12586 		sgs->group_misfit_task_load = 1;
12587 
12588 	for_each_cpu_and(i, sched_group_span(group), p->cpus_ptr) {
12589 		struct rq *rq = cpu_rq(i);
12590 		unsigned int local;
12591 
12592 		sgs->group_load += cpu_load_without(rq, p);
12593 		sgs->group_util += cpu_util_without(i, p);
12594 		sgs->group_runnable += cpu_runnable_without(rq, p);
12595 		local = task_running_on_cpu(i, p);
12596 		sgs->sum_h_nr_running += rq->cfs.h_nr_runnable - local;
12597 
12598 		nr_running = rq->nr_running - local;
12599 		sgs->sum_nr_running += nr_running;
12600 
12601 		/*
12602 		 * No need to call idle_cpu_without() if nr_running is not 0
12603 		 */
12604 		if (!nr_running && idle_cpu_without(i, p))
12605 			sgs->idle_cpus++;
12606 
12607 		/* Check if task fits in the CPU */
12608 		if (sd->flags & SD_ASYM_CPUCAPACITY &&
12609 		    sgs->group_misfit_task_load &&
12610 		    task_fits_cpu(p, i))
12611 			sgs->group_misfit_task_load = 0;
12612 
12613 	}
12614 
12615 	sgs->group_capacity = group->sgc->capacity;
12616 
12617 	sgs->group_weight = group->group_weight;
12618 
12619 	sgs->group_type = group_classify(sd->imbalance_pct, group, sgs);
12620 
12621 	/*
12622 	 * Computing avg_load makes sense only when group is fully busy or
12623 	 * overloaded
12624 	 */
12625 	if (sgs->group_type == group_fully_busy ||
12626 		sgs->group_type == group_overloaded)
12627 		sgs->avg_load = (sgs->group_load * SCHED_CAPACITY_SCALE) /
12628 				sgs->group_capacity;
12629 }
12630 
12631 static bool update_pick_idlest(struct sched_group *idlest,
12632 			       struct sg_lb_stats *idlest_sgs,
12633 			       struct sched_group *group,
12634 			       struct sg_lb_stats *sgs)
12635 {
12636 	if (sgs->group_type < idlest_sgs->group_type)
12637 		return true;
12638 
12639 	if (sgs->group_type > idlest_sgs->group_type)
12640 		return false;
12641 
12642 	/*
12643 	 * The candidate and the current idlest group are the same type of
12644 	 * group. Let check which one is the idlest according to the type.
12645 	 */
12646 
12647 	switch (sgs->group_type) {
12648 	case group_overloaded:
12649 	case group_fully_busy:
12650 		/* Select the group with lowest avg_load. */
12651 		if (idlest_sgs->avg_load <= sgs->avg_load)
12652 			return false;
12653 		break;
12654 
12655 	case group_llc_balance:
12656 	case group_imbalanced:
12657 	case group_asym_packing:
12658 	case group_smt_balance:
12659 		/* Those types are not used in the slow wakeup path */
12660 		return false;
12661 
12662 	case group_misfit_task:
12663 		/* Select group with the highest max capacity */
12664 		if (idlest->sgc->max_capacity >= group->sgc->max_capacity)
12665 			return false;
12666 		break;
12667 
12668 	case group_has_spare:
12669 		/* Select group with most idle CPUs */
12670 		if (idlest_sgs->idle_cpus > sgs->idle_cpus)
12671 			return false;
12672 
12673 		/* Select group with lowest group_util */
12674 		if (idlest_sgs->idle_cpus == sgs->idle_cpus &&
12675 			idlest_sgs->group_util <= sgs->group_util)
12676 			return false;
12677 
12678 		break;
12679 	}
12680 
12681 	return true;
12682 }
12683 
12684 /*
12685  * sched_balance_find_dst_group() finds and returns the least busy CPU group within the
12686  * domain.
12687  *
12688  * Assumes p is allowed on at least one CPU in sd.
12689  */
12690 static struct sched_group *
12691 sched_balance_find_dst_group(struct sched_domain *sd, struct task_struct *p, int this_cpu)
12692 {
12693 	struct sched_group *idlest = NULL, *local = NULL, *group = sd->groups;
12694 	struct sg_lb_stats local_sgs, tmp_sgs;
12695 	struct sg_lb_stats *sgs;
12696 	unsigned long imbalance;
12697 	struct sg_lb_stats idlest_sgs = {
12698 			.avg_load = UINT_MAX,
12699 			.group_type = group_overloaded,
12700 	};
12701 
12702 	do {
12703 		int local_group;
12704 
12705 		/* Skip over this group if it has no CPUs allowed */
12706 		if (!cpumask_intersects(sched_group_span(group),
12707 					p->cpus_ptr))
12708 			continue;
12709 
12710 		/* Skip over this group if no cookie matched */
12711 		if (!sched_group_cookie_match(cpu_rq(this_cpu), p, group))
12712 			continue;
12713 
12714 		local_group = cpumask_test_cpu(this_cpu,
12715 					       sched_group_span(group));
12716 
12717 		if (local_group) {
12718 			sgs = &local_sgs;
12719 			local = group;
12720 		} else {
12721 			sgs = &tmp_sgs;
12722 		}
12723 
12724 		update_sg_wakeup_stats(sd, group, sgs, p);
12725 
12726 		if (!local_group && update_pick_idlest(idlest, &idlest_sgs, group, sgs)) {
12727 			idlest = group;
12728 			idlest_sgs = *sgs;
12729 		}
12730 
12731 	} while (group = group->next, group != sd->groups);
12732 
12733 
12734 	/* There is no idlest group to push tasks to */
12735 	if (!idlest)
12736 		return NULL;
12737 
12738 	/* The local group has been skipped because of CPU affinity */
12739 	if (!local)
12740 		return idlest;
12741 
12742 	/*
12743 	 * If the local group is idler than the selected idlest group
12744 	 * don't try and push the task.
12745 	 */
12746 	if (local_sgs.group_type < idlest_sgs.group_type)
12747 		return NULL;
12748 
12749 	/*
12750 	 * If the local group is busier than the selected idlest group
12751 	 * try and push the task.
12752 	 */
12753 	if (local_sgs.group_type > idlest_sgs.group_type)
12754 		return idlest;
12755 
12756 	switch (local_sgs.group_type) {
12757 	case group_overloaded:
12758 	case group_fully_busy:
12759 
12760 		/* Calculate allowed imbalance based on load */
12761 		imbalance = scale_load_down(NICE_0_LOAD) *
12762 				(sd->imbalance_pct-100) / 100;
12763 
12764 		/*
12765 		 * When comparing groups across NUMA domains, it's possible for
12766 		 * the local domain to be very lightly loaded relative to the
12767 		 * remote domains but "imbalance" skews the comparison making
12768 		 * remote CPUs look much more favourable. When considering
12769 		 * cross-domain, add imbalance to the load on the remote node
12770 		 * and consider staying local.
12771 		 */
12772 
12773 		if ((sd->flags & SD_NUMA) &&
12774 		    ((idlest_sgs.avg_load + imbalance) >= local_sgs.avg_load))
12775 			return NULL;
12776 
12777 		/*
12778 		 * If the local group is less loaded than the selected
12779 		 * idlest group don't try and push any tasks.
12780 		 */
12781 		if (idlest_sgs.avg_load >= (local_sgs.avg_load + imbalance))
12782 			return NULL;
12783 
12784 		if (100 * local_sgs.avg_load <= sd->imbalance_pct * idlest_sgs.avg_load)
12785 			return NULL;
12786 		break;
12787 
12788 	case group_llc_balance:
12789 	case group_imbalanced:
12790 	case group_asym_packing:
12791 	case group_smt_balance:
12792 		/* Those type are not used in the slow wakeup path */
12793 		return NULL;
12794 
12795 	case group_misfit_task:
12796 		/* Select group with the highest max capacity */
12797 		if (local->sgc->max_capacity >= idlest->sgc->max_capacity)
12798 			return NULL;
12799 		break;
12800 
12801 	case group_has_spare:
12802 #ifdef CONFIG_NUMA
12803 		if (sd->flags & SD_NUMA) {
12804 			int imb_numa_nr = sd->imb_numa_nr;
12805 #ifdef CONFIG_NUMA_BALANCING
12806 			int idlest_cpu;
12807 			/*
12808 			 * If there is spare capacity at NUMA, try to select
12809 			 * the preferred node
12810 			 */
12811 			if (cpu_to_node(this_cpu) == p->numa_preferred_nid)
12812 				return NULL;
12813 
12814 			idlest_cpu = cpumask_first(sched_group_span(idlest));
12815 			if (cpu_to_node(idlest_cpu) == p->numa_preferred_nid)
12816 				return idlest;
12817 #endif /* CONFIG_NUMA_BALANCING */
12818 			/*
12819 			 * Otherwise, keep the task close to the wakeup source
12820 			 * and improve locality if the number of running tasks
12821 			 * would remain below threshold where an imbalance is
12822 			 * allowed while accounting for the possibility the
12823 			 * task is pinned to a subset of CPUs. If there is a
12824 			 * real need of migration, periodic load balance will
12825 			 * take care of it.
12826 			 */
12827 			if (p->nr_cpus_allowed != NR_CPUS) {
12828 				unsigned int w = cpumask_weight_and(p->cpus_ptr,
12829 								sched_group_span(local));
12830 				imb_numa_nr = min(w, sd->imb_numa_nr);
12831 			}
12832 
12833 			imbalance = abs(local_sgs.idle_cpus - idlest_sgs.idle_cpus);
12834 			if (!adjust_numa_imbalance(imbalance,
12835 						   local_sgs.sum_nr_running + 1,
12836 						   imb_numa_nr)) {
12837 				return NULL;
12838 			}
12839 		}
12840 #endif /* CONFIG_NUMA */
12841 
12842 		/*
12843 		 * Select group with highest number of idle CPUs. We could also
12844 		 * compare the utilization which is more stable but it can end
12845 		 * up that the group has less spare capacity but finally more
12846 		 * idle CPUs which means more opportunity to run task.
12847 		 */
12848 		if (local_sgs.idle_cpus >= idlest_sgs.idle_cpus)
12849 			return NULL;
12850 		break;
12851 	}
12852 
12853 	return idlest;
12854 }
12855 
12856 static void update_idle_cpu_scan(struct lb_env *env,
12857 				 unsigned long sum_util)
12858 {
12859 	struct sched_domain_shared *sd_share;
12860 	struct sched_domain *sd = env->sd;
12861 	int llc_weight, pct;
12862 	u64 x, y, tmp;
12863 	/*
12864 	 * Update the number of CPUs to scan in LLC domain, which could
12865 	 * be used as a hint in select_idle_cpu(). The update of sd_share
12866 	 * could be expensive because it is within a shared cache line.
12867 	 * So the write of this hint only occurs during periodic load
12868 	 * balancing, rather than CPU_NEWLY_IDLE, because the latter
12869 	 * can fire way more frequently than the former.
12870 	 */
12871 	if (!sched_feat(SIS_UTIL) || env->idle == CPU_NEWLY_IDLE)
12872 		return;
12873 
12874 	sd_share = sd->shared;
12875 	if (!sd_share)
12876 		return;
12877 
12878 	/*
12879 	 * The number of CPUs to search drops as sum_util increases, when
12880 	 * sum_util hits 85% or above, the scan stops.
12881 	 * The reason to choose 85% as the threshold is because this is the
12882 	 * imbalance_pct(117) when a LLC sched group is overloaded.
12883 	 *
12884 	 * let y = SCHED_CAPACITY_SCALE - p * x^2                       [1]
12885 	 * and y'= y / SCHED_CAPACITY_SCALE
12886 	 *
12887 	 * x is the ratio of sum_util compared to the CPU capacity:
12888 	 * x = sum_util / (llc_weight * SCHED_CAPACITY_SCALE)
12889 	 * y' is the ratio of CPUs to be scanned in the LLC domain,
12890 	 * and the number of CPUs to scan is calculated by:
12891 	 *
12892 	 * nr_scan = llc_weight * y'                                    [2]
12893 	 *
12894 	 * When x hits the threshold of overloaded, AKA, when
12895 	 * x = 100 / pct, y drops to 0. According to [1],
12896 	 * p should be SCHED_CAPACITY_SCALE * pct^2 / 10000
12897 	 *
12898 	 * Scale x by SCHED_CAPACITY_SCALE:
12899 	 * x' = sum_util / llc_weight;                                  [3]
12900 	 *
12901 	 * and finally [1] becomes:
12902 	 * y = SCHED_CAPACITY_SCALE -
12903 	 *     x'^2 * pct^2 / (10000 * SCHED_CAPACITY_SCALE)            [4]
12904 	 *
12905 	 */
12906 	/* equation [3] */
12907 	x = sum_util;
12908 	llc_weight = sd->span_weight;
12909 	do_div(x, llc_weight);
12910 
12911 	/* equation [4] */
12912 	pct = sd->imbalance_pct;
12913 	tmp = x * x * pct * pct;
12914 	do_div(tmp, 10000 * SCHED_CAPACITY_SCALE);
12915 	tmp = min_t(long, tmp, SCHED_CAPACITY_SCALE);
12916 	y = SCHED_CAPACITY_SCALE - tmp;
12917 
12918 	/* equation [2] */
12919 	y *= llc_weight;
12920 	do_div(y, SCHED_CAPACITY_SCALE);
12921 	if ((int)y != sd_share->nr_idle_scan)
12922 		WRITE_ONCE(sd_share->nr_idle_scan, (int)y);
12923 }
12924 
12925 /**
12926  * update_sd_lb_stats - Update sched_domain's statistics for load balancing.
12927  * @env: The load balancing environment.
12928  * @sds: variable to hold the statistics for this sched_domain.
12929  */
12930 
12931 static inline void update_sd_lb_stats(struct lb_env *env, struct sd_lb_stats *sds)
12932 {
12933 	struct sched_group *sg = env->sd->groups;
12934 	struct sg_lb_stats *local = &sds->local_stat;
12935 	struct sg_lb_stats tmp_sgs;
12936 	unsigned long sum_util = 0;
12937 	bool sg_overloaded = 0, sg_overutilized = 0;
12938 
12939 	env->dst_core_idle = !sched_smt_active() || is_core_idle(env->dst_cpu);
12940 
12941 	do {
12942 		struct sg_lb_stats *sgs = &tmp_sgs;
12943 		int local_group;
12944 
12945 		local_group = cpumask_test_cpu(env->dst_cpu, sched_group_span(sg));
12946 		if (local_group) {
12947 			sds->local = sg;
12948 			sgs = local;
12949 
12950 			if (env->idle != CPU_NEWLY_IDLE ||
12951 			    time_after_eq(jiffies, sg->sgc->next_update))
12952 				update_group_capacity(env->sd, env->dst_cpu);
12953 		}
12954 
12955 		update_sg_lb_stats(env, sds, sg, sgs, &sg_overloaded);
12956 
12957 		if (!local_group && update_sd_pick_busiest(env, sds, sg, sgs)) {
12958 			sds->busiest = sg;
12959 			sds->busiest_stat = *sgs;
12960 		}
12961 
12962 		sg_overutilized |= sgs->group_overutilized;
12963 
12964 		/* Now, start updating sd_lb_stats */
12965 		sds->total_load += sgs->group_load;
12966 		sds->total_capacity += sgs->group_capacity;
12967 
12968 		sum_util += sgs->group_util;
12969 		sg = sg->next;
12970 	} while (sg != env->sd->groups);
12971 
12972 	/*
12973 	 * Indicate that the child domain of the busiest group prefers tasks
12974 	 * go to a child's sibling domains first. NB the flags of a sched group
12975 	 * are those of the child domain.
12976 	 */
12977 	if (sds->busiest)
12978 		sds->prefer_sibling = !!(sds->busiest->flags & SD_PREFER_SIBLING);
12979 
12980 
12981 	if (env->sd->flags & SD_NUMA)
12982 		env->fbq_type = fbq_classify_group(&sds->busiest_stat);
12983 
12984 	if (!env->sd->parent) {
12985 		/* update overload indicator if we are at root domain */
12986 		set_rd_overloaded(env->dst_rq->rd, sg_overloaded);
12987 
12988 		/* Update over-utilization (tipping point, U >= 0) indicator */
12989 		set_rd_overutilized(env->dst_rq->rd, sg_overutilized);
12990 	} else if (sg_overutilized) {
12991 		set_rd_overutilized(env->dst_rq->rd, sg_overutilized);
12992 	}
12993 
12994 	update_idle_cpu_scan(env, sum_util);
12995 }
12996 
12997 /**
12998  * calculate_imbalance - Calculate the amount of imbalance present within the
12999  *			 groups of a given sched_domain during load balance.
13000  * @env: load balance environment
13001  * @sds: statistics of the sched_domain whose imbalance is to be calculated.
13002  */
13003 static inline void calculate_imbalance(struct lb_env *env, struct sd_lb_stats *sds)
13004 {
13005 	struct sg_lb_stats *local, *busiest;
13006 
13007 	local = &sds->local_stat;
13008 	busiest = &sds->busiest_stat;
13009 
13010 	if (busiest->group_type == group_misfit_task) {
13011 		if (env->sd->flags & SD_ASYM_CPUCAPACITY) {
13012 			/* Set imbalance to allow misfit tasks to be balanced. */
13013 			env->migration_type = migrate_misfit;
13014 			env->imbalance = 1;
13015 		} else {
13016 			/*
13017 			 * Set load imbalance to allow moving task from cpu
13018 			 * with reduced capacity.
13019 			 */
13020 			env->migration_type = migrate_load;
13021 			env->imbalance = busiest->group_misfit_task_load;
13022 		}
13023 		return;
13024 	}
13025 
13026 	if (busiest->group_type == group_asym_packing) {
13027 		/*
13028 		 * In case of asym capacity, we will try to migrate all load to
13029 		 * the preferred CPU.
13030 		 */
13031 		env->migration_type = migrate_task;
13032 		env->imbalance = busiest->sum_h_nr_running;
13033 		return;
13034 	}
13035 
13036 	if (busiest->group_type == group_smt_balance) {
13037 		/* Reduce number of tasks sharing CPU capacity */
13038 		env->migration_type = migrate_task;
13039 		env->imbalance = 1;
13040 		return;
13041 	}
13042 
13043 #ifdef CONFIG_SCHED_CACHE
13044 	if (busiest->group_type == group_llc_balance) {
13045 		/* Move a task that prefer local LLC */
13046 		env->migration_type = migrate_llc_task;
13047 		env->imbalance = 1;
13048 		return;
13049 	}
13050 #endif
13051 
13052 	if (busiest->group_type == group_imbalanced) {
13053 		/*
13054 		 * In the group_imb case we cannot rely on group-wide averages
13055 		 * to ensure CPU-load equilibrium, try to move any task to fix
13056 		 * the imbalance. The next load balance will take care of
13057 		 * balancing back the system.
13058 		 */
13059 		env->migration_type = migrate_task;
13060 		env->imbalance = 1;
13061 		return;
13062 	}
13063 
13064 	/*
13065 	 * Try to use spare capacity of local group without overloading it or
13066 	 * emptying busiest.
13067 	 */
13068 	if (local->group_type == group_has_spare) {
13069 		if ((busiest->group_type > group_fully_busy) &&
13070 		    !(env->sd->flags & SD_SHARE_LLC)) {
13071 			/*
13072 			 * If busiest is overloaded, try to fill spare
13073 			 * capacity. This might end up creating spare capacity
13074 			 * in busiest or busiest still being overloaded but
13075 			 * there is no simple way to directly compute the
13076 			 * amount of load to migrate in order to balance the
13077 			 * system.
13078 			 */
13079 			env->migration_type = migrate_util;
13080 			env->imbalance = max(local->group_capacity, local->group_util) -
13081 					 local->group_util;
13082 
13083 			/*
13084 			 * In some cases, the group's utilization is max or even
13085 			 * higher than capacity because of migrations but the
13086 			 * local CPU is (newly) idle. There is at least one
13087 			 * waiting task in this overloaded busiest group. Let's
13088 			 * try to pull it.
13089 			 */
13090 			if (env->idle && env->imbalance == 0) {
13091 				env->migration_type = migrate_task;
13092 				env->imbalance = 1;
13093 			}
13094 
13095 			return;
13096 		}
13097 
13098 		if (busiest->group_weight == 1 || sds->prefer_sibling) {
13099 			/*
13100 			 * When prefer sibling, evenly spread running tasks on
13101 			 * groups.
13102 			 */
13103 			env->migration_type = migrate_task;
13104 			env->imbalance = sibling_imbalance(env, sds, busiest, local);
13105 		} else {
13106 
13107 			/*
13108 			 * If there is no overload, we just want to even the number of
13109 			 * idle CPUs.
13110 			 */
13111 			env->migration_type = migrate_task;
13112 			env->imbalance = max_t(long, 0,
13113 					       (local->idle_cpus - busiest->idle_cpus));
13114 		}
13115 
13116 #ifdef CONFIG_NUMA
13117 		/* Consider allowing a small imbalance between NUMA groups */
13118 		if (env->sd->flags & SD_NUMA) {
13119 			env->imbalance = adjust_numa_imbalance(env->imbalance,
13120 							       local->sum_nr_running + 1,
13121 							       env->sd->imb_numa_nr);
13122 		}
13123 #endif
13124 
13125 		/* Number of tasks to move to restore balance */
13126 		env->imbalance >>= 1;
13127 
13128 		return;
13129 	}
13130 
13131 	/*
13132 	 * Local is fully busy but has to take more load to relieve the
13133 	 * busiest group
13134 	 */
13135 	if (local->group_type < group_overloaded) {
13136 		/*
13137 		 * Local will become overloaded so the avg_load metrics are
13138 		 * finally needed.
13139 		 */
13140 
13141 		local->avg_load = (local->group_load * SCHED_CAPACITY_SCALE) /
13142 				  local->group_capacity;
13143 
13144 		/*
13145 		 * If the local group is more loaded than the selected
13146 		 * busiest group don't try to pull any tasks.
13147 		 */
13148 		if (local->avg_load >= busiest->avg_load) {
13149 			env->imbalance = 0;
13150 			return;
13151 		}
13152 
13153 		sds->avg_load = (sds->total_load * SCHED_CAPACITY_SCALE) /
13154 				sds->total_capacity;
13155 
13156 		/*
13157 		 * If the local group is more loaded than the average system
13158 		 * load, don't try to pull any tasks.
13159 		 */
13160 		if (local->avg_load >= sds->avg_load) {
13161 			env->imbalance = 0;
13162 			return;
13163 		}
13164 
13165 	}
13166 
13167 	/*
13168 	 * Both group are or will become overloaded and we're trying to get all
13169 	 * the CPUs to the average_load, so we don't want to push ourselves
13170 	 * above the average load, nor do we wish to reduce the max loaded CPU
13171 	 * below the average load. At the same time, we also don't want to
13172 	 * reduce the group load below the group capacity. Thus we look for
13173 	 * the minimum possible imbalance.
13174 	 */
13175 	env->migration_type = migrate_load;
13176 	env->imbalance = min(
13177 		(busiest->avg_load - sds->avg_load) * busiest->group_capacity,
13178 		(sds->avg_load - local->avg_load) * local->group_capacity
13179 	) / SCHED_CAPACITY_SCALE;
13180 }
13181 
13182 /******* sched_balance_find_src_group() helpers end here *********************/
13183 
13184 /*
13185  * Decision matrix according to the local and busiest group type:
13186  *
13187  * busiest \ local has_spare fully_busy misfit asym imbalanced overloaded
13188  * has_spare        nr_idle   balanced   N/A    N/A  balanced   balanced
13189  * fully_busy       nr_idle   nr_idle    N/A    N/A  balanced   balanced
13190  * misfit_task      force     N/A        N/A    N/A  N/A        N/A
13191  * asym_packing     force     force      N/A    N/A  force      force
13192  * imbalanced       force     force      N/A    N/A  force      force
13193  * overloaded       force     force      N/A    N/A  force      avg_load
13194  *
13195  * N/A :      Not Applicable because already filtered while updating
13196  *            statistics.
13197  * balanced : The system is balanced for these 2 groups.
13198  * force :    Calculate the imbalance as load migration is probably needed.
13199  * avg_load : Only if imbalance is significant enough.
13200  * nr_idle :  dst_cpu is not busy and the number of idle CPUs is quite
13201  *            different in groups.
13202  */
13203 
13204 /**
13205  * sched_balance_find_src_group - Returns the busiest group within the sched_domain
13206  * if there is an imbalance.
13207  * @env: The load balancing environment.
13208  *
13209  * Also calculates the amount of runnable load which should be moved
13210  * to restore balance.
13211  *
13212  * Return:	- The busiest group if imbalance exists.
13213  */
13214 static struct sched_group *sched_balance_find_src_group(struct lb_env *env)
13215 {
13216 	struct sg_lb_stats *local, *busiest;
13217 	struct sd_lb_stats sds;
13218 
13219 	init_sd_lb_stats(&sds);
13220 
13221 	/*
13222 	 * Compute the various statistics relevant for load balancing at
13223 	 * this level.
13224 	 */
13225 	update_sd_lb_stats(env, &sds);
13226 
13227 	/* There is no busy sibling group to pull tasks from */
13228 	if (!sds.busiest)
13229 		goto out_balanced;
13230 
13231 	busiest = &sds.busiest_stat;
13232 
13233 	/* Misfit tasks should be dealt with regardless of the avg load */
13234 	if (busiest->group_type == group_misfit_task)
13235 		goto force_balance;
13236 
13237 	if (!is_rd_overutilized(env->dst_rq->rd) &&
13238 	    rcu_dereference_all(env->dst_rq->rd->pd))
13239 		goto out_balanced;
13240 
13241 	/* ASYM feature bypasses nice load balance check */
13242 	if (busiest->group_type == group_asym_packing)
13243 		goto force_balance;
13244 
13245 	/*
13246 	 * If the busiest group is imbalanced the below checks don't
13247 	 * work because they assume all things are equal, which typically
13248 	 * isn't true due to cpus_ptr constraints and the like.
13249 	 */
13250 	if (busiest->group_type == group_imbalanced)
13251 		goto force_balance;
13252 
13253 	local = &sds.local_stat;
13254 	/*
13255 	 * If the local group is busier than the selected busiest group
13256 	 * don't try and pull any tasks.
13257 	 */
13258 	if (local->group_type > busiest->group_type)
13259 		goto out_balanced;
13260 
13261 	/*
13262 	 * When groups are overloaded, use the avg_load to ensure fairness
13263 	 * between tasks.
13264 	 */
13265 	if (local->group_type == group_overloaded) {
13266 		/*
13267 		 * If the local group is more loaded than the selected
13268 		 * busiest group don't try to pull any tasks.
13269 		 */
13270 		if (local->avg_load >= busiest->avg_load)
13271 			goto out_balanced;
13272 
13273 		/* XXX broken for overlapping NUMA groups */
13274 		sds.avg_load = (sds.total_load * SCHED_CAPACITY_SCALE) /
13275 				sds.total_capacity;
13276 
13277 		/*
13278 		 * Don't pull any tasks if this group is already above the
13279 		 * domain average load.
13280 		 */
13281 		if (local->avg_load >= sds.avg_load)
13282 			goto out_balanced;
13283 
13284 		/*
13285 		 * If the busiest group is more loaded, use imbalance_pct to be
13286 		 * conservative.
13287 		 */
13288 		if (100 * busiest->avg_load <=
13289 				env->sd->imbalance_pct * local->avg_load)
13290 			goto out_balanced;
13291 	}
13292 
13293 	/*
13294 	 * Try to move all excess tasks to a sibling domain of the busiest
13295 	 * group's child domain.
13296 	 */
13297 	if (sds.prefer_sibling && local->group_type == group_has_spare &&
13298 	    (busiest->group_type == group_llc_balance ||
13299 	    sibling_imbalance(env, &sds, busiest, local) > 1))
13300 		goto force_balance;
13301 
13302 	if (busiest->group_type != group_overloaded) {
13303 		if (!env->idle) {
13304 			/*
13305 			 * If the busiest group is not overloaded (and as a
13306 			 * result the local one too) but this CPU is already
13307 			 * busy, let another idle CPU try to pull task.
13308 			 */
13309 			goto out_balanced;
13310 		}
13311 
13312 		if (busiest->group_type == group_smt_balance &&
13313 		    smt_vs_nonsmt_groups(sds.local, sds.busiest)) {
13314 			/* Let non SMT CPU pull from SMT CPU sharing with sibling */
13315 			goto force_balance;
13316 		}
13317 
13318 		if (busiest->group_weight > 1 &&
13319 		    local->idle_cpus <= (busiest->idle_cpus + 1)) {
13320 			/*
13321 			 * If the busiest group is not overloaded
13322 			 * and there is no imbalance between this and busiest
13323 			 * group wrt idle CPUs, it is balanced. The imbalance
13324 			 * becomes significant if the diff is greater than 1
13325 			 * otherwise we might end up to just move the imbalance
13326 			 * on another group. Of course this applies only if
13327 			 * there is more than 1 CPU per group.
13328 			 */
13329 			goto out_balanced;
13330 		}
13331 
13332 		if (busiest->sum_h_nr_running == 1) {
13333 			/*
13334 			 * busiest doesn't have any tasks waiting to run
13335 			 */
13336 			goto out_balanced;
13337 		}
13338 	}
13339 
13340 force_balance:
13341 	/* Looks like there is an imbalance. Compute it */
13342 	calculate_imbalance(env, &sds);
13343 	return env->imbalance ? sds.busiest : NULL;
13344 
13345 out_balanced:
13346 	env->imbalance = 0;
13347 	return NULL;
13348 }
13349 
13350 /*
13351  * sched_balance_find_src_rq - find the busiest runqueue among the CPUs in the group.
13352  */
13353 static struct rq *sched_balance_find_src_rq(struct lb_env *env,
13354 				     struct sched_group *group)
13355 {
13356 	struct rq *busiest = NULL, *rq;
13357 	unsigned long busiest_util = 0, busiest_load = 0, busiest_capacity = 1;
13358 	unsigned int __maybe_unused busiest_pref_llc = 0;
13359 	struct sched_domain __maybe_unused *sd_tmp;
13360 	unsigned int busiest_nr = 0;
13361 	int __maybe_unused dst_llc;
13362 	int i;
13363 
13364 	for_each_cpu_and(i, sched_group_span(group), env->cpus) {
13365 		unsigned long capacity, load, util;
13366 		unsigned int nr_running;
13367 		enum fbq_type rt;
13368 
13369 		rq = cpu_rq(i);
13370 		rt = fbq_classify_rq(rq);
13371 
13372 		/*
13373 		 * We classify groups/runqueues into three groups:
13374 		 *  - regular: there are !numa tasks
13375 		 *  - remote:  there are numa tasks that run on the 'wrong' node
13376 		 *  - all:     there is no distinction
13377 		 *
13378 		 * In order to avoid migrating ideally placed numa tasks,
13379 		 * ignore those when there's better options.
13380 		 *
13381 		 * If we ignore the actual busiest queue to migrate another
13382 		 * task, the next balance pass can still reduce the busiest
13383 		 * queue by moving tasks around inside the node.
13384 		 *
13385 		 * If we cannot move enough load due to this classification
13386 		 * the next pass will adjust the group classification and
13387 		 * allow migration of more tasks.
13388 		 *
13389 		 * Both cases only affect the total convergence complexity.
13390 		 */
13391 		if (rt > env->fbq_type)
13392 			continue;
13393 
13394 		nr_running = rq->cfs.h_nr_runnable;
13395 		if (!nr_running)
13396 			continue;
13397 
13398 		capacity = capacity_of(i);
13399 
13400 		/*
13401 		 * For ASYM_CPUCAPACITY domains, don't pick a CPU that could
13402 		 * eventually lead to active_balancing high->low capacity.
13403 		 * Higher per-CPU capacity is considered better than balancing
13404 		 * average load.
13405 		 */
13406 		if (env->sd->flags & SD_ASYM_CPUCAPACITY &&
13407 		    nr_running == 1) {
13408 			bool cluster_equal_cap = static_branch_unlikely(&sched_cluster_active) &&
13409 						 (get_actual_cpu_capacity(env->dst_cpu) ==
13410 						  get_actual_cpu_capacity(i));
13411 			bool smt_degraded_cap = sched_smt_active() && !is_core_idle(i);
13412 
13413 			/*
13414 			 * Busy SMT siblings reduce the capacity of CPU @i. Do
13415 			 * not skip it in this case.
13416 			 *
13417 			 * CONFIG_SCHED_CLUSTER requires balancing load across
13418 			 * clusters of identical capacity, accounting for
13419 			 * hardware and cpufreq pressure.
13420 			 */
13421 			if (!smt_degraded_cap && !cluster_equal_cap &&
13422 			    !capacity_greater(capacity_of(env->dst_cpu), capacity))
13423 				continue;
13424 		}
13425 
13426 		/*
13427 		 * Make sure we only pull tasks from a CPU of lower priority
13428 		 * when balancing between SMT siblings.
13429 		 *
13430 		 * If balancing between cores, let lower priority CPUs help
13431 		 * SMT cores with more than one busy sibling.
13432 		 */
13433 		if (sched_asym(env->sd, i, env->dst_cpu) && nr_running == 1)
13434 			continue;
13435 
13436 		switch (env->migration_type) {
13437 		case migrate_load:
13438 			/*
13439 			 * When comparing with load imbalance, use cpu_load()
13440 			 * which is not scaled with the CPU capacity.
13441 			 */
13442 			load = cpu_load(rq);
13443 
13444 			if (nr_running == 1 && load > env->imbalance &&
13445 			    !check_cpu_capacity(rq, env->sd))
13446 				break;
13447 
13448 			/*
13449 			 * For the load comparisons with the other CPUs,
13450 			 * consider the cpu_load() scaled with the CPU
13451 			 * capacity, so that the load can be moved away
13452 			 * from the CPU that is potentially running at a
13453 			 * lower capacity.
13454 			 *
13455 			 * Thus we're looking for max(load_i / capacity_i),
13456 			 * crosswise multiplication to rid ourselves of the
13457 			 * division works out to:
13458 			 * load_i * capacity_j > load_j * capacity_i;
13459 			 * where j is our previous maximum.
13460 			 */
13461 			if (load * busiest_capacity > busiest_load * capacity) {
13462 				busiest_load = load;
13463 				busiest_capacity = capacity;
13464 				busiest = rq;
13465 			}
13466 			break;
13467 
13468 		case migrate_util:
13469 			util = cpu_util_cfs_boost(i);
13470 
13471 			/*
13472 			 * Don't try to pull utilization from a CPU with one
13473 			 * running task. Whatever its utilization, we will fail
13474 			 * detach the task.
13475 			 */
13476 			if (nr_running <= 1)
13477 				continue;
13478 
13479 			if (busiest_util < util) {
13480 				busiest_util = util;
13481 				busiest = rq;
13482 			}
13483 			break;
13484 
13485 		case migrate_task:
13486 			if (busiest_nr < nr_running) {
13487 				busiest_nr = nr_running;
13488 				busiest = rq;
13489 			}
13490 			break;
13491 
13492 		case migrate_misfit:
13493 			/*
13494 			 * For ASYM_CPUCAPACITY domains with misfit tasks we
13495 			 * simply seek the "biggest" misfit task.
13496 			 */
13497 			if (rq->misfit_task_load > busiest_load) {
13498 				busiest_load = rq->misfit_task_load;
13499 				busiest = rq;
13500 			}
13501 
13502 			break;
13503 
13504 		case migrate_llc_task:
13505 #ifdef CONFIG_SCHED_CACHE
13506 			sd_tmp = rcu_dereference_all(rq->sd);
13507 			dst_llc = llc_id(env->dst_cpu);
13508 
13509 			if (sd_tmp && (unsigned)dst_llc < sd_tmp->llc_max) {
13510 				unsigned int this_pref_llc =
13511 					sd_tmp->llc_counts[dst_llc];
13512 
13513 				if (busiest_pref_llc < this_pref_llc) {
13514 					busiest_pref_llc = this_pref_llc;
13515 					busiest = rq;
13516 				}
13517 			}
13518 #endif
13519 			break;
13520 
13521 		}
13522 	}
13523 
13524 	return busiest;
13525 }
13526 
13527 /*
13528  * Max backoff if we encounter pinned tasks. Pretty arbitrary value, but
13529  * so long as it is large enough.
13530  */
13531 #define MAX_PINNED_INTERVAL	512
13532 
13533 static inline bool
13534 asym_active_balance(struct lb_env *env)
13535 {
13536 	/*
13537 	 * ASYM_PACKING needs to force migrate tasks from busy but lower
13538 	 * priority CPUs in order to pack all tasks in the highest priority
13539 	 * CPUs. When done between cores, do it only if the whole core if the
13540 	 * whole core is idle.
13541 	 *
13542 	 * If @env::src_cpu is an SMT core with busy siblings, let
13543 	 * the lower priority @env::dst_cpu help it. Do not follow
13544 	 * CPU priority.
13545 	 */
13546 	return env->idle && sched_use_asym_prio(env->sd, env->dst_cpu) &&
13547 	       (sched_asym_prefer(env->dst_cpu, env->src_cpu) ||
13548 		!sched_use_asym_prio(env->sd, env->src_cpu));
13549 }
13550 
13551 static inline bool
13552 imbalanced_active_balance(struct lb_env *env)
13553 {
13554 	struct sched_domain *sd = env->sd;
13555 
13556 	/*
13557 	 * The imbalanced case includes the case of pinned tasks preventing a fair
13558 	 * distribution of the load on the system but also the even distribution of the
13559 	 * threads on a system with spare capacity
13560 	 */
13561 	if ((env->migration_type == migrate_task) &&
13562 	    (sd->nr_balance_failed > sd->cache_nice_tries+2))
13563 		return 1;
13564 
13565 	return 0;
13566 }
13567 
13568 static int need_active_balance(struct lb_env *env)
13569 {
13570 	struct sched_domain *sd = env->sd;
13571 
13572 	if (alb_break_llc(env))
13573 		return 0;
13574 
13575 	if (asym_active_balance(env))
13576 		return 1;
13577 
13578 	if (imbalanced_active_balance(env))
13579 		return 1;
13580 
13581 	/*
13582 	 * The dst_cpu is idle and the src_cpu CPU has only 1 CFS task.
13583 	 * It's worth migrating the task if the src_cpu's capacity is reduced
13584 	 * because of other sched_class or IRQs if more capacity stays
13585 	 * available on dst_cpu.
13586 	 */
13587 	if (env->idle &&
13588 	    (env->src_rq->cfs.h_nr_runnable == 1)) {
13589 		if ((check_cpu_capacity(env->src_rq, sd)) &&
13590 		    (capacity_of(env->src_cpu)*sd->imbalance_pct < capacity_of(env->dst_cpu)*100))
13591 			return 1;
13592 	}
13593 
13594 	if (env->migration_type == migrate_misfit ||
13595 	    env->migration_type == migrate_llc_task)
13596 		return 1;
13597 
13598 	return 0;
13599 }
13600 
13601 static int active_load_balance_cpu_stop(void *data);
13602 static int active_load_balance_llc_cpu_stop(void *data);
13603 
13604 /*
13605  * migration_type is checked elsewhere to decide migration policy, so
13606  * it shouldn't be repurposed just to flag an LLC-directed active
13607  * balance across the stopper. Pick the callback here instead.
13608  */
13609 static inline cpu_stop_fn_t alb_stop_fn(struct lb_env *env)
13610 {
13611 	if (env->migration_type == migrate_llc_task)
13612 		return active_load_balance_llc_cpu_stop;
13613 
13614 	return active_load_balance_cpu_stop;
13615 }
13616 
13617 static int should_we_balance(struct lb_env *env)
13618 {
13619 	struct cpumask *swb_cpus = this_cpu_cpumask_var_ptr(should_we_balance_tmpmask);
13620 	struct sched_group *sg = env->sd->groups;
13621 	int cpu, idle_smt = -1;
13622 
13623 	/*
13624 	 * Ensure the balancing environment is consistent; can happen
13625 	 * when the softirq triggers 'during' hotplug.
13626 	 */
13627 	if (!cpumask_test_cpu(env->dst_cpu, env->cpus))
13628 		return 0;
13629 
13630 	/*
13631 	 * In the newly idle case, we will allow all the CPUs
13632 	 * to do the newly idle load balance.
13633 	 *
13634 	 * However, we bail out if we already have tasks or a wakeup pending,
13635 	 * to optimize wakeup latency.
13636 	 */
13637 	if (env->idle == CPU_NEWLY_IDLE) {
13638 		if (env->dst_rq->nr_running > 0 || env->dst_rq->ttwu_pending)
13639 			return 0;
13640 		return 1;
13641 	}
13642 
13643 	cpumask_copy(swb_cpus, group_balance_mask(sg));
13644 	/* Try to find first idle CPU */
13645 	for_each_cpu_and(cpu, swb_cpus, env->cpus) {
13646 		if (!idle_cpu(cpu))
13647 			continue;
13648 
13649 		/*
13650 		 * Don't balance to idle SMT in busy core right away when
13651 		 * balancing cores, but remember the first idle SMT CPU for
13652 		 * later consideration.  Find CPU on an idle core first.
13653 		 */
13654 		if (sched_smt_active() &&
13655 		    !(env->sd->flags & SD_SHARE_CPUCAPACITY) &&
13656 		    !is_core_idle(cpu)) {
13657 			if (idle_smt == -1)
13658 				idle_smt = cpu;
13659 			/*
13660 			 * If the core is not idle, and first SMT sibling which is
13661 			 * idle has been found, then its not needed to check other
13662 			 * SMT siblings for idleness:
13663 			 */
13664 			cpumask_andnot(swb_cpus, swb_cpus, cpu_smt_mask(cpu));
13665 			continue;
13666 		}
13667 
13668 		/*
13669 		 * Are we the first idle core in a non-SMT domain or higher,
13670 		 * or the first idle CPU in a SMT domain?
13671 		 */
13672 		return cpu == env->dst_cpu;
13673 	}
13674 
13675 	/* Are we the first idle CPU with busy siblings? */
13676 	if (idle_smt != -1)
13677 		return idle_smt == env->dst_cpu;
13678 
13679 	/* Are we the first CPU of this group ? */
13680 	return group_balance_cpu(sg) == env->dst_cpu;
13681 }
13682 
13683 static void update_lb_imbalance_stat(struct lb_env *env, struct sched_domain *sd,
13684 				     enum cpu_idle_type idle)
13685 {
13686 	if (!schedstat_enabled())
13687 		return;
13688 
13689 	switch (env->migration_type) {
13690 	case migrate_load:
13691 		__schedstat_add(sd->lb_imbalance_load[idle], env->imbalance);
13692 		break;
13693 	case migrate_util:
13694 		__schedstat_add(sd->lb_imbalance_util[idle], env->imbalance);
13695 		break;
13696 	case migrate_task:
13697 		__schedstat_add(sd->lb_imbalance_task[idle], env->imbalance);
13698 		break;
13699 	case migrate_misfit:
13700 		__schedstat_add(sd->lb_imbalance_misfit[idle], env->imbalance);
13701 		break;
13702 	case migrate_llc_task:
13703 		break;
13704 	}
13705 }
13706 
13707 /*
13708  * This flag serializes load-balancing passes over large domains
13709  * (above the NODE topology level) - only one load-balancing instance
13710  * may run at a time, to reduce overhead on very large systems with
13711  * lots of CPUs and large NUMA distances.
13712  *
13713  * - Note that load-balancing passes triggered while another one
13714  *   is executing are skipped and not re-tried.
13715  *
13716  * - Also note that this does not serialize rebalance_domains()
13717  *   execution, as non-SD_SERIALIZE domains will still be
13718  *   load-balanced in parallel.
13719  */
13720 static atomic_t sched_balance_running = ATOMIC_INIT(0);
13721 
13722 /*
13723  * Check this_cpu to ensure it is balanced within domain. Attempt to move
13724  * tasks if there is an imbalance.
13725  */
13726 static int sched_balance_rq(int this_cpu, struct rq *this_rq,
13727 			struct sched_domain *sd, enum cpu_idle_type idle,
13728 			int *continue_balancing)
13729 {
13730 	int ld_moved, cur_ld_moved, active_balance = 0;
13731 	struct sched_domain *sd_parent = sd->parent;
13732 	struct sched_group *group;
13733 	struct rq *busiest;
13734 	struct rq_flags rf;
13735 	struct cpumask *cpus = this_cpu_cpumask_var_ptr(load_balance_mask);
13736 	struct lb_env env = {
13737 		.sd		= sd,
13738 		.dst_cpu	= this_cpu,
13739 		.dst_rq		= this_rq,
13740 		.dst_grpmask    = group_balance_mask(sd->groups),
13741 		.idle		= idle,
13742 		.loop_break	= SCHED_NR_MIGRATE_BREAK,
13743 		.cpus		= cpus,
13744 		.fbq_type	= all,
13745 		.tasks		= LIST_HEAD_INIT(env.tasks),
13746 	};
13747 	bool need_unlock = false;
13748 
13749 	cpumask_and(cpus, sched_domain_span(sd), cpu_active_mask);
13750 
13751 	schedstat_inc(sd->lb_count[idle]);
13752 
13753 redo:
13754 	if (!should_we_balance(&env)) {
13755 		*continue_balancing = 0;
13756 		goto out_balanced;
13757 	}
13758 
13759 	if (!need_unlock && (sd->flags & SD_SERIALIZE)) {
13760 		int zero = 0;
13761 		if (!atomic_try_cmpxchg_acquire(&sched_balance_running, &zero, 1))
13762 			goto out_balanced;
13763 
13764 		need_unlock = true;
13765 	}
13766 
13767 	group = sched_balance_find_src_group(&env);
13768 	if (!group) {
13769 		schedstat_inc(sd->lb_nobusyg[idle]);
13770 		goto out_balanced;
13771 	}
13772 
13773 	busiest = sched_balance_find_src_rq(&env, group);
13774 	if (!busiest) {
13775 		schedstat_inc(sd->lb_nobusyq[idle]);
13776 		goto out_balanced;
13777 	}
13778 
13779 	WARN_ON_ONCE(busiest == env.dst_rq);
13780 
13781 	update_lb_imbalance_stat(&env, sd, idle);
13782 
13783 	env.src_cpu = busiest->cpu;
13784 	env.src_rq = busiest;
13785 
13786 	ld_moved = 0;
13787 	/* Clear this flag as soon as we find a pullable task */
13788 	env.flags |= LBF_ALL_PINNED;
13789 	if (busiest->nr_running > 1) {
13790 		/*
13791 		 * Attempt to move tasks. If sched_balance_find_src_group has found
13792 		 * an imbalance but busiest->nr_running <= 1, the group is
13793 		 * still unbalanced. ld_moved simply stays zero, so it is
13794 		 * correctly treated as an imbalance.
13795 		 */
13796 		env.loop_max  = min(sysctl_sched_nr_migrate, busiest->nr_running);
13797 
13798 more_balance:
13799 		rq_lock_irqsave(busiest, &rf);
13800 		update_rq_clock(busiest);
13801 
13802 		/*
13803 		 * cur_ld_moved - load moved in current iteration
13804 		 * ld_moved     - cumulative load moved across iterations
13805 		 */
13806 		cur_ld_moved = detach_tasks(&env);
13807 
13808 		/*
13809 		 * We've detached some tasks from busiest_rq. Every
13810 		 * task is masked "TASK_ON_RQ_MIGRATING", so we can safely
13811 		 * unlock busiest->lock, and we are able to be sure
13812 		 * that nobody can manipulate the tasks in parallel.
13813 		 * See task_rq_lock() family for the details.
13814 		 */
13815 
13816 		rq_unlock(busiest, &rf);
13817 
13818 		if (cur_ld_moved) {
13819 			attach_tasks(&env);
13820 			ld_moved += cur_ld_moved;
13821 		}
13822 
13823 		local_irq_restore(rf.flags);
13824 
13825 		if (env.flags & LBF_NEED_BREAK) {
13826 			env.flags &= ~LBF_NEED_BREAK;
13827 			goto more_balance;
13828 		}
13829 
13830 		/*
13831 		 * Revisit (affine) tasks on src_cpu that couldn't be moved to
13832 		 * us and move them to an alternate dst_cpu in our sched_group
13833 		 * where they can run. The upper limit on how many times we
13834 		 * iterate on same src_cpu is dependent on number of CPUs in our
13835 		 * sched_group.
13836 		 *
13837 		 * This changes load balance semantics a bit on who can move
13838 		 * load to a given_cpu. In addition to the given_cpu itself
13839 		 * (or a ilb_cpu acting on its behalf where given_cpu is
13840 		 * nohz-idle), we now have balance_cpu in a position to move
13841 		 * load to given_cpu. In rare situations, this may cause
13842 		 * conflicts (balance_cpu and given_cpu/ilb_cpu deciding
13843 		 * _independently_ and at _same_ time to move some load to
13844 		 * given_cpu) causing excess load to be moved to given_cpu.
13845 		 * This however should not happen so much in practice and
13846 		 * moreover subsequent load balance cycles should correct the
13847 		 * excess load moved.
13848 		 */
13849 		if ((env.flags & LBF_DST_PINNED) && env.imbalance > 0) {
13850 
13851 			/* Prevent to re-select dst_cpu via env's CPUs */
13852 			__cpumask_clear_cpu(env.dst_cpu, env.cpus);
13853 
13854 			env.dst_rq	 = cpu_rq(env.new_dst_cpu);
13855 			env.dst_cpu	 = env.new_dst_cpu;
13856 			env.flags	&= ~LBF_DST_PINNED;
13857 			env.loop	 = 0;
13858 			env.loop_break	 = SCHED_NR_MIGRATE_BREAK;
13859 
13860 			/*
13861 			 * Go back to "more_balance" rather than "redo" since we
13862 			 * need to continue with same src_cpu.
13863 			 */
13864 			goto more_balance;
13865 		}
13866 
13867 		/*
13868 		 * We failed to reach balance because of affinity.
13869 		 */
13870 		if (sd_parent) {
13871 			int *group_imbalance = &sd_parent->groups->sgc->imbalance;
13872 
13873 			if ((env.flags & LBF_SOME_PINNED) && env.imbalance > 0)
13874 				*group_imbalance = 1;
13875 		}
13876 
13877 		/* All tasks on this runqueue were pinned by CPU affinity */
13878 		if (unlikely(env.flags & LBF_ALL_PINNED)) {
13879 			__cpumask_clear_cpu(cpu_of(busiest), cpus);
13880 			/*
13881 			 * Attempting to continue load balancing at the current
13882 			 * sched_domain level only makes sense if there are
13883 			 * active CPUs remaining as possible busiest CPUs to
13884 			 * pull load from which are not contained within the
13885 			 * destination group that is receiving any migrated
13886 			 * load.
13887 			 */
13888 			if (!cpumask_subset(cpus, env.dst_grpmask)) {
13889 				env.loop = 0;
13890 				env.loop_break = SCHED_NR_MIGRATE_BREAK;
13891 				goto redo;
13892 			}
13893 			goto out_all_pinned;
13894 		}
13895 	}
13896 
13897 	if (ld_moved) {
13898 		sd->nr_balance_failed = 0;
13899 		goto out_unbalanced;
13900 	}
13901 
13902 	schedstat_inc(sd->lb_failed[idle]);
13903 	/*
13904 	 * Increment the failure counter only on periodic balance.
13905 	 * We do not want newidle balance, which can be very
13906 	 * frequent, pollute the failure counter causing
13907 	 * excessive cache_hot migrations and active balances.
13908 	 *
13909 	 * Similarly for migration_misfit which is not related to
13910 	 * load/util migration, don't pollute nr_balance_failed.
13911 	 *
13912 	 * The same for cache aware scheduling's allowance for
13913 	 * load imbalance. If regular load balance does not
13914 	 * migrate task due to LLC locality, it is a expected
13915 	 * behavior and don't pollute nr_balance_failed.
13916 	 * See can_migrate_task().
13917 	 */
13918 	if (idle != CPU_NEWLY_IDLE &&
13919 	    env.migration_type != migrate_misfit &&
13920 	    !(env.flags & LBF_LLC_PINNED))
13921 		sd->nr_balance_failed++;
13922 
13923 	if (!need_active_balance(&env))
13924 		goto out_unbalanced;
13925 
13926 	scoped_guard (raw_spin_rq_lock_irqsave, busiest) {
13927 		/*
13928 		 * Don't kick the active_load_balance_cpu_stop,
13929 		 * if the curr task on busiest CPU can't be
13930 		 * moved to this_cpu:
13931 		 */
13932 		if (!cpumask_test_cpu(this_cpu, busiest->curr->cpus_ptr))
13933 			goto out_one_pinned;
13934 
13935 		/* Record that we found at least one task that could run on this_cpu */
13936 		env.flags &= ~LBF_ALL_PINNED;
13937 
13938 		/*
13939 		 * ->active_balance synchronizes accesses to
13940 		 * ->active_balance_work.  Once set, it's cleared
13941 		 * only after active load balance is finished.
13942 		 */
13943 		if (busiest->active_balance)
13944 			goto out_unbalanced;
13945 
13946 		/*
13947 		 * @busiest dropped its rq_lock in the middle of
13948 		 * scheduling out its ->curr task (->on_rq := 0), no
13949 		 * need to forcefully punt it away with active balance.
13950 		 */
13951 		if (!busiest->curr->on_rq)
13952 			goto out_unbalanced;
13953 
13954 		busiest->active_balance = 1;
13955 		busiest->push_cpu = this_cpu;
13956 		active_balance = 1;
13957 		preempt_disable();
13958 	}
13959 	if (active_balance) {
13960 		stop_one_cpu_nowait(cpu_of(busiest),
13961 				    alb_stop_fn(&env), busiest,
13962 				    &busiest->active_balance_work);
13963 	}
13964 	preempt_enable();
13965 
13966 out_unbalanced:
13967 	/* We were unbalanced, so reset the balancing interval */
13968 	sd->balance_interval = sd->min_interval;
13969 	goto out;
13970 
13971 out_balanced:
13972 	/*
13973 	 * We reach balance although we may have faced some affinity
13974 	 * constraints. Clear the imbalance flag only if other tasks got
13975 	 * a chance to move and fix the imbalance.
13976 	 */
13977 	if (sd_parent && !(env.flags & LBF_ALL_PINNED)) {
13978 		int *group_imbalance = &sd_parent->groups->sgc->imbalance;
13979 
13980 		if (*group_imbalance)
13981 			*group_imbalance = 0;
13982 	}
13983 
13984 out_all_pinned:
13985 	/*
13986 	 * We reach balance because all tasks are pinned at this level so
13987 	 * we can't migrate them. Let the imbalance flag set so parent level
13988 	 * can try to migrate them.
13989 	 */
13990 	schedstat_inc(sd->lb_balanced[idle]);
13991 
13992 	sd->nr_balance_failed = 0;
13993 
13994 out_one_pinned:
13995 	ld_moved = 0;
13996 
13997 	/*
13998 	 * sched_balance_newidle() disregards balance intervals, so we could
13999 	 * repeatedly reach this code, which would lead to balance_interval
14000 	 * skyrocketing in a short amount of time. Skip the balance_interval
14001 	 * increase logic to avoid that.
14002 	 *
14003 	 * Similarly misfit migration which is not necessarily an indication of
14004 	 * the system being busy and requires lb to backoff to let it settle
14005 	 * down.
14006 	 */
14007 	if (env.idle == CPU_NEWLY_IDLE ||
14008 	    env.migration_type == migrate_misfit)
14009 		goto out;
14010 
14011 	/* tune up the balancing interval */
14012 	if ((env.flags & LBF_ALL_PINNED &&
14013 	     sd->balance_interval < MAX_PINNED_INTERVAL) ||
14014 	    sd->balance_interval < sd->max_interval)
14015 		sd->balance_interval *= 2;
14016 out:
14017 	if (need_unlock)
14018 		atomic_set_release(&sched_balance_running, 0);
14019 
14020 	return ld_moved;
14021 }
14022 
14023 static inline unsigned long
14024 get_sd_balance_interval(struct sched_domain *sd, int cpu_busy)
14025 {
14026 	unsigned long interval = sd->balance_interval;
14027 
14028 	if (cpu_busy)
14029 		interval *= sd->busy_factor;
14030 
14031 	/* scale ms to jiffies */
14032 	interval = msecs_to_jiffies(interval);
14033 
14034 	/*
14035 	 * Reduce likelihood of busy balancing at higher domains racing with
14036 	 * balancing at lower domains by preventing their balancing periods
14037 	 * from being multiples of each other.
14038 	 */
14039 	if (cpu_busy)
14040 		interval -= 1;
14041 
14042 	interval = clamp(interval, 1UL, max_load_balance_interval);
14043 
14044 	return interval;
14045 }
14046 
14047 static inline void
14048 update_next_balance(struct sched_domain *sd, unsigned long *next_balance)
14049 {
14050 	unsigned long interval, next;
14051 
14052 	/* used by idle balance, so cpu_busy = 0 */
14053 	interval = get_sd_balance_interval(sd, 0);
14054 	next = sd->last_balance + interval;
14055 
14056 	if (time_after(*next_balance, next))
14057 		*next_balance = next;
14058 }
14059 
14060 /*
14061  * active_load_balance_cpu_stop is run by the CPU stopper. It pushes
14062  * running tasks off the busiest CPU onto idle CPUs. It requires at
14063  * least 1 task to be running on each physical CPU where possible, and
14064  * avoids physical / logical imbalances.
14065  */
14066 static int __active_load_balance_cpu_stop(void *data, unsigned int lb_flags)
14067 {
14068 	struct rq *busiest_rq = data;
14069 	int busiest_cpu = cpu_of(busiest_rq);
14070 	int target_cpu = busiest_rq->push_cpu;
14071 	struct rq *target_rq = cpu_rq(target_cpu);
14072 	struct sched_domain *sd;
14073 	struct task_struct *p = NULL;
14074 	struct rq_flags rf;
14075 
14076 	rq_lock_irq(busiest_rq, &rf);
14077 	/*
14078 	 * Between queueing the stop-work and running it is a hole in which
14079 	 * CPUs can become inactive. We should not move tasks from or to
14080 	 * inactive CPUs.
14081 	 */
14082 	if (!cpu_active(busiest_cpu) || !cpu_active(target_cpu))
14083 		goto out_unlock;
14084 
14085 	/* Make sure the requested CPU hasn't gone down in the meantime: */
14086 	if (unlikely(busiest_cpu != smp_processor_id() ||
14087 		     !busiest_rq->active_balance))
14088 		goto out_unlock;
14089 
14090 	/* Is there any task to move? */
14091 	if (busiest_rq->nr_running <= 1)
14092 		goto out_unlock;
14093 
14094 	/*
14095 	 * This condition is "impossible", if it occurs
14096 	 * we need to fix it. Originally reported by
14097 	 * Bjorn Helgaas on a 128-CPU setup.
14098 	 */
14099 	WARN_ON_ONCE(busiest_rq == target_rq);
14100 
14101 	/* Search for an sd spanning us and the target CPU. */
14102 	rcu_read_lock();
14103 	for_each_domain(target_cpu, sd) {
14104 		if (cpumask_test_cpu(busiest_cpu, sched_domain_span(sd)))
14105 			break;
14106 	}
14107 
14108 	if (likely(sd)) {
14109 		struct lb_env env = {
14110 			.sd		= sd,
14111 			.dst_cpu	= target_cpu,
14112 			.dst_rq		= target_rq,
14113 			.src_cpu	= busiest_rq->cpu,
14114 			.src_rq		= busiest_rq,
14115 			.idle		= CPU_IDLE,
14116 			.flags		= LBF_ACTIVE_LB | lb_flags,
14117 		};
14118 
14119 		schedstat_inc(sd->alb_count);
14120 		update_rq_clock(busiest_rq);
14121 
14122 		p = detach_one_task(&env);
14123 		if (p) {
14124 			schedstat_inc(sd->alb_pushed);
14125 			/* Active balancing done, reset the failure counter. */
14126 			sd->nr_balance_failed = 0;
14127 		} else {
14128 			schedstat_inc(sd->alb_failed);
14129 		}
14130 	}
14131 	rcu_read_unlock();
14132 out_unlock:
14133 	busiest_rq->active_balance = 0;
14134 	rq_unlock(busiest_rq, &rf);
14135 
14136 	if (p)
14137 		attach_one_task(target_rq, p);
14138 
14139 	local_irq_enable();
14140 
14141 	return 0;
14142 }
14143 
14144 static int active_load_balance_cpu_stop(void *data)
14145 {
14146 	return __active_load_balance_cpu_stop(data, 0);
14147 }
14148 
14149 static int active_load_balance_llc_cpu_stop(void *data)
14150 {
14151 	return __active_load_balance_cpu_stop(data, LBF_ACTIVE_LB_LLC);
14152 }
14153 
14154 /*
14155  * Scale the max sched_balance_rq interval with the number of CPUs in the system.
14156  * This trades load-balance latency on larger machines for less cross talk.
14157  */
14158 void update_max_interval(void)
14159 {
14160 	max_load_balance_interval = HZ*num_online_cpus()/10;
14161 }
14162 
14163 static inline void update_newidle_stats(struct sched_domain *sd, unsigned int success)
14164 {
14165 	sd->newidle_call++;
14166 	sd->newidle_success += success;
14167 
14168 	if (sd->newidle_call >= 1024) {
14169 		u64 now = sched_clock();
14170 		s64 delta = now - sd->newidle_stamp;
14171 		sd->newidle_stamp = now;
14172 		int ratio = 0;
14173 
14174 		if (delta < 0)
14175 			delta = 0;
14176 
14177 		if (sched_feat(NI_RATE)) {
14178 			/*
14179 			 * ratio  delta   freq
14180 			 *
14181 			 * 1024 -  4  s -  128 Hz
14182 			 *  512 -  2  s -  256 Hz
14183 			 *  256 -  1  s -  512 Hz
14184 			 *  128 - .5  s - 1024 Hz
14185 			 *   64 - .25 s - 2048 Hz
14186 			 */
14187 			ratio = delta >> 22;
14188 		}
14189 
14190 		ratio += sd->newidle_success;
14191 
14192 		sd->newidle_ratio = min(1024, ratio);
14193 		sd->newidle_call /= 2;
14194 		sd->newidle_success /= 2;
14195 	}
14196 }
14197 
14198 static inline bool
14199 update_newidle_cost(struct sched_domain *sd, u64 cost, unsigned int success)
14200 {
14201 	unsigned long next_decay = sd->last_decay_max_lb_cost + HZ;
14202 	unsigned long now = jiffies;
14203 
14204 	if (cost)
14205 		update_newidle_stats(sd, success);
14206 
14207 	if (cost > sd->max_newidle_lb_cost) {
14208 		/*
14209 		 * Track max cost of a domain to make sure to not delay the
14210 		 * next wakeup on the CPU.
14211 		 */
14212 		sd->max_newidle_lb_cost = cost;
14213 		sd->last_decay_max_lb_cost = now;
14214 
14215 	} else if (time_after(now, next_decay)) {
14216 		/*
14217 		 * Decay the newidle max times by ~1% per second to ensure that
14218 		 * it is not outdated and the current max cost is actually
14219 		 * shorter.
14220 		 */
14221 		sd->max_newidle_lb_cost = (sd->max_newidle_lb_cost * 253) / 256;
14222 		sd->last_decay_max_lb_cost = now;
14223 		return true;
14224 	}
14225 
14226 	return false;
14227 }
14228 
14229 /*
14230  * It checks each scheduling domain to see if it is due to be balanced,
14231  * and initiates a balancing operation if so.
14232  *
14233  * Balancing parameters are set up in init_sched_domains.
14234  */
14235 static void sched_balance_domains(struct rq *rq, enum cpu_idle_type idle)
14236 {
14237 	int continue_balancing = 1;
14238 	int cpu = rq->cpu;
14239 	int busy = idle != CPU_IDLE && !sched_idle_rq(rq);
14240 	unsigned long interval;
14241 	struct sched_domain *sd;
14242 	/* Earliest time when we have to do rebalance again */
14243 	unsigned long next_balance = jiffies + 60*HZ;
14244 	int update_next_balance = 0;
14245 	int need_decay = 0;
14246 	u64 max_cost = 0;
14247 
14248 	rcu_read_lock();
14249 	for_each_domain(cpu, sd) {
14250 		/*
14251 		 * Decay the newidle max times here because this is a regular
14252 		 * visit to all the domains.
14253 		 */
14254 		need_decay = update_newidle_cost(sd, 0, 0);
14255 		max_cost += sd->max_newidle_lb_cost;
14256 
14257 		/*
14258 		 * Stop the load balance at this level. There is another
14259 		 * CPU in our sched group which is doing load balancing more
14260 		 * actively.
14261 		 */
14262 		if (!continue_balancing) {
14263 			if (need_decay)
14264 				continue;
14265 			break;
14266 		}
14267 
14268 		interval = get_sd_balance_interval(sd, busy);
14269 		if (time_after_eq(jiffies, sd->last_balance + interval)) {
14270 			if (sched_balance_rq(cpu, rq, sd, idle, &continue_balancing)) {
14271 				/*
14272 				 * The LBF_DST_PINNED logic could have changed
14273 				 * env->dst_cpu, so we can't know our idle
14274 				 * state even if we migrated tasks. Update it.
14275 				 */
14276 				idle = idle_cpu(cpu);
14277 				busy = !idle && !sched_idle_rq(rq);
14278 			}
14279 			sd->last_balance = jiffies;
14280 			interval = get_sd_balance_interval(sd, busy);
14281 		}
14282 		if (time_after(next_balance, sd->last_balance + interval)) {
14283 			next_balance = sd->last_balance + interval;
14284 			update_next_balance = 1;
14285 		}
14286 	}
14287 	if (need_decay) {
14288 		/*
14289 		 * Ensure the rq-wide value also decays but keep it at a
14290 		 * reasonable floor to avoid funnies with rq->avg_idle.
14291 		 */
14292 		rq->max_idle_balance_cost =
14293 			max((u64)sysctl_sched_migration_cost, max_cost);
14294 	}
14295 	rcu_read_unlock();
14296 
14297 	/*
14298 	 * next_balance will be updated only when there is a need.
14299 	 * When the cpu is attached to null domain for ex, it will not be
14300 	 * updated.
14301 	 */
14302 	if (likely(update_next_balance))
14303 		rq->next_balance = next_balance;
14304 
14305 }
14306 
14307 static inline int on_null_domain(struct rq *rq)
14308 {
14309 	return unlikely(!rcu_dereference_sched(rq->sd));
14310 }
14311 
14312 #ifdef CONFIG_NO_HZ_COMMON
14313 /*
14314  * NOHZ idle load balancing (ILB) details:
14315  *
14316  * - When one of the busy CPUs notices that there may be an idle rebalancing
14317  *   needed, they will kick the idle load balancer, which then does idle
14318  *   load balancing for all the idle CPUs.
14319  */
14320 static inline int find_new_ilb(void)
14321 {
14322 	struct cpumask *ilb_cpus;
14323 	int ilb_cpu, fallback = -1;
14324 
14325 	lockdep_assert_irqs_disabled();
14326 
14327 	/*
14328 	 * Reuse the per-CPU select_rq_mask, which is protected from concurrent
14329 	 * use on this CPU by having interrupts disabled.
14330 	 */
14331 	ilb_cpus = this_cpu_cpumask_var_ptr(select_rq_mask);
14332 	cpumask_and(ilb_cpus, nohz.idle_cpus_mask,
14333 		    housekeeping_cpumask(HK_TYPE_KERNEL_NOISE));
14334 
14335 	for_each_cpu(ilb_cpu, ilb_cpus) {
14336 		if (!idle_cpu(ilb_cpu)) {
14337 			/*
14338 			 * Once an idle fallback exists, a busy CPU proves that
14339 			 * this core cannot be fully idle. Skip its siblings.
14340 			 */
14341 			if (sched_smt_active() && fallback >= 0)
14342 				cpumask_andnot(ilb_cpus, ilb_cpus, cpu_smt_mask(ilb_cpu));
14343 			continue;
14344 		}
14345 
14346 		/*
14347 		 * Running the idle load balancer on an idle sibling of a busy
14348 		 * SMT core can reduce the capacity available to its sibling. Prefer
14349 		 * a CPU whose entire core is idle, but retain the first idle CPU as
14350 		 * a fallback so idle balancing can still make progress when no fully
14351 		 * idle core exists.
14352 		 */
14353 		if (sched_smt_active() && !is_core_idle(ilb_cpu)) {
14354 			if (fallback < 0)
14355 				fallback = ilb_cpu;
14356 
14357 			/*
14358 			 * The core is not idle, so there is no need to check
14359 			 * any of its other SMT siblings.
14360 			 */
14361 			cpumask_andnot(ilb_cpus, ilb_cpus,
14362 				       cpu_smt_mask(ilb_cpu));
14363 			continue;
14364 		}
14365 
14366 		return ilb_cpu;
14367 	}
14368 
14369 	return fallback;
14370 }
14371 
14372 /*
14373  * Kick a CPU to do the NOHZ balancing, if it is time for it, via a cross-CPU
14374  * SMP function call (IPI).
14375  *
14376  * Prefer a CPU on a fully idle core in the HK_TYPE_KERNEL_NOISE housekeeping
14377  * set. Fall back to the first idle CPU when no fully idle core exists.
14378  */
14379 static void kick_ilb(unsigned int flags)
14380 {
14381 	int ilb_cpu;
14382 
14383 	/*
14384 	 * Increase nohz.next_balance only when if full ilb is triggered but
14385 	 * not if we only update stats.
14386 	 */
14387 	if (flags & NOHZ_BALANCE_KICK)
14388 		nohz.next_balance = jiffies+1;
14389 
14390 	ilb_cpu = find_new_ilb();
14391 	if (ilb_cpu < 0)
14392 		return;
14393 
14394 	/*
14395 	 * Don't bother if no new NOHZ balance work items for ilb_cpu,
14396 	 * i.e. all bits in flags are already set in ilb_cpu.
14397 	 */
14398 	if ((atomic_read(nohz_flags(ilb_cpu)) & flags) == flags)
14399 		return;
14400 
14401 	/*
14402 	 * Access to rq::nohz_csd is serialized by NOHZ_KICK_MASK; he who sets
14403 	 * the first flag owns it; cleared by nohz_csd_func().
14404 	 */
14405 	flags = atomic_fetch_or(flags, nohz_flags(ilb_cpu));
14406 	if (flags & NOHZ_KICK_MASK)
14407 		return;
14408 
14409 	/*
14410 	 * This way we generate an IPI on the target CPU which
14411 	 * is idle, and the softirq performing NOHZ idle load balancing
14412 	 * will be run before returning from the IPI.
14413 	 */
14414 	smp_call_function_single_async(ilb_cpu, &cpu_rq(ilb_cpu)->nohz_csd);
14415 }
14416 
14417 /*
14418  * Current decision point for kicking the idle load balancer in the presence
14419  * of idle CPUs in the system.
14420  */
14421 static void nohz_balancer_kick(struct rq *rq)
14422 {
14423 	unsigned long now = jiffies;
14424 	struct sched_domain_shared *sds;
14425 	struct sched_domain *sd;
14426 	int nr_busy, i, cpu = rq->cpu;
14427 	unsigned int flags = 0;
14428 
14429 	if (unlikely(rq->idle_balance))
14430 		return;
14431 
14432 	/*
14433 	 * We may be recently in ticked or tickless idle mode. At the first
14434 	 * busy tick after returning from idle, we will update the busy stats.
14435 	 */
14436 	nohz_balance_exit_idle(rq);
14437 
14438 	if (READ_ONCE(nohz.has_blocked_load) &&
14439 	    time_after(now, READ_ONCE(nohz.next_blocked)))
14440 		flags = NOHZ_STATS_KICK;
14441 
14442 	/*
14443 	 * Most of the time system is not 100% busy. i.e nohz.nr_cpus > 0
14444 	 * Skip the read if time is not due.
14445 	 *
14446 	 * If none are in tickless mode, there maybe a narrow window
14447 	 * (28 jiffies, HZ=1000) where flags maybe set and kick_ilb called.
14448 	 * But idle load balancing is not done as find_new_ilb fails.
14449 	 * That's very rare. So read nohz.nr_cpus only if time is due.
14450 	 */
14451 	if (time_before(now, nohz.next_balance))
14452 		goto out;
14453 
14454 	/*
14455 	 * None are in tickless mode and hence no need for NOHZ idle load
14456 	 * balancing
14457 	 */
14458 	if (unlikely(cpumask_empty(nohz.idle_cpus_mask)))
14459 		return;
14460 
14461 	if (rq->nr_running >= 2) {
14462 		flags = NOHZ_STATS_KICK | NOHZ_BALANCE_KICK;
14463 		goto out;
14464 	}
14465 
14466 	sd = rcu_dereference_all(rq->sd);
14467 	if (sd) {
14468 		/*
14469 		 * If there's a runnable CFS task and the current CPU has reduced
14470 		 * capacity, kick the ILB to see if there's a better CPU to run on:
14471 		 */
14472 		if (rq->cfs.h_nr_runnable >= 1 && check_cpu_capacity(rq, sd)) {
14473 			flags |= NOHZ_STATS_KICK | NOHZ_BALANCE_KICK;
14474 			goto out;
14475 		}
14476 	}
14477 
14478 	sd = rcu_dereference_all(per_cpu(sd_asym_packing, cpu));
14479 	if (sd) {
14480 		/*
14481 		 * When ASYM_PACKING; see if there's a more preferred CPU
14482 		 * currently idle; in which case, kick the ILB to move tasks
14483 		 * around.
14484 		 *
14485 		 * When balancing between cores, all the SMT siblings of the
14486 		 * preferred CPU must be idle.
14487 		 */
14488 		for_each_cpu_and(i, sched_domain_span(sd), nohz.idle_cpus_mask) {
14489 			if (sched_asym(sd, i, cpu)) {
14490 				flags |= NOHZ_STATS_KICK | NOHZ_BALANCE_KICK;
14491 				goto out;
14492 			}
14493 		}
14494 	}
14495 
14496 	sd = rcu_dereference_all(per_cpu(sd_asym_cpucapacity, cpu));
14497 	if (sd) {
14498 		/*
14499 		 * When ASYM_CPUCAPACITY; see if there's a higher capacity CPU
14500 		 * to run the misfit task on.
14501 		 */
14502 		if (check_misfit_status(rq))
14503 			flags |= NOHZ_STATS_KICK | NOHZ_BALANCE_KICK;
14504 
14505 		/*
14506 		 * For asymmetric systems, we do not want to nicely balance
14507 		 * cache use, instead we want to embrace asymmetry and only
14508 		 * ensure tasks have enough CPU capacity.
14509 		 *
14510 		 * Skip the LLC logic because it's not relevant in that case.
14511 		 */
14512 		goto out;
14513 	}
14514 
14515 	sds = rcu_dereference_all(per_cpu(sd_balance_shared, cpu));
14516 	if (sds) {
14517 		/*
14518 		 * If there is an imbalance between LLC domains (IOW we could
14519 		 * increase the overall cache utilization), we need a less-loaded LLC
14520 		 * domain to pull some load from. Likewise, we may need to spread
14521 		 * load within the current LLC domain (e.g. packed SMT cores but
14522 		 * other CPUs are idle). We can't really know from here how busy
14523 		 * the others are - so just get a NOHZ balance going if it looks
14524 		 * like this LLC domain has tasks we could move.
14525 		 */
14526 		nr_busy = atomic_read(&sds->nr_busy_cpus);
14527 		if (nr_busy > 1)
14528 			flags |= NOHZ_STATS_KICK | NOHZ_BALANCE_KICK;
14529 	}
14530 out:
14531 	if (READ_ONCE(nohz.needs_update))
14532 		flags |= NOHZ_NEXT_KICK;
14533 
14534 	if (flags)
14535 		kick_ilb(flags);
14536 }
14537 
14538 static void set_cpu_sd_state_busy(int cpu)
14539 {
14540 	struct sched_domain *sd;
14541 	sd = rcu_dereference_all(per_cpu(sd_llc, cpu));
14542 
14543 	/*
14544 	 * sd->nohz_idle only pairs with nr_busy_cpus on sd->shared; if this
14545 	 * domain has no shared object there is nothing to clear or account.
14546 	 */
14547 	if (!sd || !sd->shared || !sd->nohz_idle)
14548 		return;
14549 	sd->nohz_idle = 0;
14550 
14551 	atomic_inc(&sd->shared->nr_busy_cpus);
14552 }
14553 
14554 void nohz_balance_exit_idle(struct rq *rq)
14555 {
14556 	WARN_ON_ONCE(rq != this_rq());
14557 
14558 	if (likely(!rq->nohz_tick_stopped))
14559 		return;
14560 
14561 	rq->nohz_tick_stopped = 0;
14562 	cpumask_clear_cpu(rq->cpu, nohz.idle_cpus_mask);
14563 
14564 	set_cpu_sd_state_busy(rq->cpu);
14565 }
14566 
14567 static void set_cpu_sd_state_idle(int cpu)
14568 {
14569 	struct sched_domain *sd;
14570 	sd = rcu_dereference_all(per_cpu(sd_llc, cpu));
14571 
14572 	/* See set_cpu_sd_state_busy(): nohz_idle is only used with sd->shared. */
14573 	if (!sd || !sd->shared || sd->nohz_idle)
14574 		return;
14575 	sd->nohz_idle = 1;
14576 
14577 	atomic_dec(&sd->shared->nr_busy_cpus);
14578 }
14579 
14580 /*
14581  * This routine will record that the CPU is going idle with tick stopped.
14582  * This info will be used in performing idle load balancing in the future.
14583  */
14584 void nohz_balance_enter_idle(int cpu)
14585 {
14586 	struct rq *rq = cpu_rq(cpu);
14587 
14588 	WARN_ON_ONCE(cpu != smp_processor_id());
14589 
14590 	/* If this CPU is going down, then nothing needs to be done: */
14591 	if (!cpu_active(cpu))
14592 		return;
14593 
14594 	/*
14595 	 * Can be set safely without rq->lock held
14596 	 * If a clear happens, it will have evaluated last additions because
14597 	 * rq->lock is held during the check and the clear
14598 	 */
14599 	rq->has_blocked_load = 1;
14600 
14601 	/*
14602 	 * The tick is still stopped but load could have been added in the
14603 	 * meantime. We set the nohz.has_blocked_load flag to trig a check of the
14604 	 * *_avg. The CPU is already part of nohz.idle_cpus_mask so the clear
14605 	 * of nohz.has_blocked_load can only happen after checking the new load
14606 	 */
14607 	if (rq->nohz_tick_stopped)
14608 		goto out;
14609 
14610 	/* If we're a completely isolated CPU, we don't play: */
14611 	if (on_null_domain(rq))
14612 		return;
14613 
14614 	rq->nohz_tick_stopped = 1;
14615 
14616 	cpumask_set_cpu(cpu, nohz.idle_cpus_mask);
14617 
14618 	/*
14619 	 * Ensures that if nohz_idle_balance() fails to observe our
14620 	 * @idle_cpus_mask store, it must observe the @has_blocked_load
14621 	 * and @needs_update stores.
14622 	 */
14623 	smp_mb__after_atomic();
14624 
14625 	set_cpu_sd_state_idle(cpu);
14626 
14627 	WRITE_ONCE(nohz.needs_update, 1);
14628 out:
14629 	/*
14630 	 * Each time a cpu enter idle, we assume that it has blocked load and
14631 	 * enable the periodic update of the load of idle CPUs
14632 	 */
14633 	WRITE_ONCE(nohz.has_blocked_load, 1);
14634 }
14635 
14636 static bool update_nohz_stats(struct rq *rq)
14637 {
14638 	unsigned int cpu = rq->cpu;
14639 
14640 	if (!rq->has_blocked_load)
14641 		return false;
14642 
14643 	if (!cpumask_test_cpu(cpu, nohz.idle_cpus_mask))
14644 		return false;
14645 
14646 	if (!time_after(jiffies, READ_ONCE(rq->last_blocked_load_update_tick)))
14647 		return true;
14648 
14649 	sched_balance_update_blocked_averages(cpu);
14650 
14651 	return rq->has_blocked_load;
14652 }
14653 
14654 /*
14655  * Internal function that runs load balance for all idle CPUs. The load balance
14656  * can be a simple update of blocked load or a complete load balance with
14657  * tasks movement depending of flags.
14658  */
14659 static void _nohz_idle_balance(struct rq *this_rq, unsigned int flags)
14660 {
14661 	/* Earliest time when we have to do rebalance again */
14662 	unsigned long now = jiffies;
14663 	unsigned long next_balance = now + 60*HZ;
14664 	bool has_blocked_load = false;
14665 	int update_next_balance = 0;
14666 	int this_cpu = this_rq->cpu;
14667 	int balance_cpu;
14668 	struct rq *rq;
14669 
14670 	WARN_ON_ONCE((flags & NOHZ_KICK_MASK) == NOHZ_BALANCE_KICK);
14671 
14672 	/*
14673 	 * We assume there will be no idle load after this update and clear
14674 	 * the has_blocked_load flag. If a cpu enters idle in the mean time, it will
14675 	 * set the has_blocked_load flag and trigger another update of idle load.
14676 	 * Because a cpu that becomes idle, is added to idle_cpus_mask before
14677 	 * setting the flag, we are sure to not clear the state and not
14678 	 * check the load of an idle cpu.
14679 	 *
14680 	 * Same applies to idle_cpus_mask vs needs_update.
14681 	 */
14682 	if (flags & NOHZ_STATS_KICK)
14683 		WRITE_ONCE(nohz.has_blocked_load, 0);
14684 	if (flags & NOHZ_NEXT_KICK)
14685 		WRITE_ONCE(nohz.needs_update, 0);
14686 
14687 	/*
14688 	 * Ensures that if we miss the CPU, we must see the has_blocked_load
14689 	 * store from nohz_balance_enter_idle().
14690 	 */
14691 	smp_mb();
14692 
14693 	/*
14694 	 * Start with the next CPU after this_cpu so we will end with this_cpu and let a
14695 	 * chance for other idle cpu to pull load.
14696 	 */
14697 	for_each_cpu_wrap(balance_cpu,  nohz.idle_cpus_mask, this_cpu+1) {
14698 		if (!idle_cpu(balance_cpu))
14699 			continue;
14700 
14701 		/*
14702 		 * If this CPU gets work to do, stop the load balancing
14703 		 * work being done for other CPUs. Next load
14704 		 * balancing owner will pick it up.
14705 		 */
14706 		if (!idle_cpu(this_cpu) && need_resched()) {
14707 			if (flags & NOHZ_STATS_KICK)
14708 				has_blocked_load = true;
14709 			if (flags & NOHZ_NEXT_KICK)
14710 				WRITE_ONCE(nohz.needs_update, 1);
14711 			goto abort;
14712 		}
14713 
14714 		rq = cpu_rq(balance_cpu);
14715 
14716 		if (flags & NOHZ_STATS_KICK)
14717 			has_blocked_load |= update_nohz_stats(rq);
14718 
14719 		/*
14720 		 * If time for next balance is due,
14721 		 * do the balance.
14722 		 */
14723 		if (time_after_eq(jiffies, rq->next_balance)) {
14724 			struct rq_flags rf;
14725 
14726 			rq_lock_irqsave(rq, &rf);
14727 			update_rq_clock(rq);
14728 			rq_unlock_irqrestore(rq, &rf);
14729 
14730 			if (flags & NOHZ_BALANCE_KICK)
14731 				sched_balance_domains(rq, CPU_IDLE);
14732 		}
14733 
14734 		if (time_after(next_balance, rq->next_balance)) {
14735 			next_balance = rq->next_balance;
14736 			update_next_balance = 1;
14737 		}
14738 	}
14739 
14740 	/*
14741 	 * next_balance will be updated only when there is a need.
14742 	 * When the CPU is attached to null domain for ex, it will not be
14743 	 * updated.
14744 	 */
14745 	if (likely(update_next_balance))
14746 		nohz.next_balance = next_balance;
14747 
14748 	if (flags & NOHZ_STATS_KICK)
14749 		WRITE_ONCE(nohz.next_blocked,
14750 			   now + msecs_to_jiffies(LOAD_AVG_PERIOD));
14751 
14752 abort:
14753 	/* There is still blocked load, enable periodic update */
14754 	if (has_blocked_load)
14755 		WRITE_ONCE(nohz.has_blocked_load, 1);
14756 }
14757 
14758 /*
14759  * In CONFIG_NO_HZ_COMMON case, the idle balance kickee will do the
14760  * rebalancing for all the CPUs for whom scheduler ticks are stopped.
14761  */
14762 static bool nohz_idle_balance(struct rq *this_rq, enum cpu_idle_type idle)
14763 {
14764 	unsigned int flags = this_rq->nohz_idle_balance;
14765 
14766 	if (!flags)
14767 		return false;
14768 
14769 	this_rq->nohz_idle_balance = 0;
14770 
14771 	if (idle != CPU_IDLE)
14772 		return false;
14773 
14774 	_nohz_idle_balance(this_rq, flags);
14775 
14776 	return true;
14777 }
14778 
14779 /*
14780  * Check if we need to directly run the ILB for updating blocked load before
14781  * entering idle state. Here we run ILB directly without issuing IPIs.
14782  *
14783  * Note that when this function is called, the tick may not yet be stopped on
14784  * this CPU yet. nohz.idle_cpus_mask is updated only when tick is stopped and
14785  * cleared on the next busy tick. In other words, nohz.idle_cpus_mask updates
14786  * don't align with CPUs enter/exit idle to avoid bottlenecks due to high idle
14787  * entry/exit rate (usec). So it is possible that _nohz_idle_balance() is
14788  * called from this function on (this) CPU that's not yet in the mask. That's
14789  * OK because the goal of nohz_run_idle_balance() is to run ILB only for
14790  * updating the blocked load of already idle CPUs without waking up one of
14791  * those idle CPUs and outside the preempt disable / IRQ off phase of the local
14792  * cpu about to enter idle, because it can take a long time.
14793  */
14794 void nohz_run_idle_balance(int cpu)
14795 {
14796 	unsigned int flags;
14797 
14798 	flags = atomic_fetch_andnot(NOHZ_NEWILB_KICK, nohz_flags(cpu));
14799 
14800 	/*
14801 	 * Update the blocked load only if no SCHED_SOFTIRQ is about to happen
14802 	 * (i.e. NOHZ_STATS_KICK set) and will do the same.
14803 	 */
14804 	if ((flags == NOHZ_NEWILB_KICK) && !need_resched())
14805 		_nohz_idle_balance(cpu_rq(cpu), NOHZ_STATS_KICK);
14806 }
14807 
14808 static void nohz_newidle_balance(struct rq *this_rq)
14809 {
14810 	int this_cpu = this_rq->cpu;
14811 
14812 	/* Will wake up very soon. No time for doing anything else*/
14813 	if (this_rq->avg_idle < sysctl_sched_migration_cost)
14814 		return;
14815 
14816 	/* Don't need to update blocked load of idle CPUs*/
14817 	if (!READ_ONCE(nohz.has_blocked_load) ||
14818 	    time_before(jiffies, READ_ONCE(nohz.next_blocked)))
14819 		return;
14820 
14821 	/*
14822 	 * Set the need to trigger ILB in order to update blocked load
14823 	 * before entering idle state.
14824 	 */
14825 	atomic_or(NOHZ_NEWILB_KICK, nohz_flags(this_cpu));
14826 }
14827 
14828 #else /* !CONFIG_NO_HZ_COMMON: */
14829 static inline void nohz_balancer_kick(struct rq *rq) { }
14830 
14831 static inline bool nohz_idle_balance(struct rq *this_rq, enum cpu_idle_type idle)
14832 {
14833 	return false;
14834 }
14835 
14836 static inline void nohz_newidle_balance(struct rq *this_rq) { }
14837 #endif /* !CONFIG_NO_HZ_COMMON */
14838 
14839 /*
14840  * sched_balance_newidle is called by schedule() if this_cpu is about to become
14841  * idle. Attempts to pull tasks from other CPUs.
14842  *
14843  * Returns:
14844  *   < 0 - we released the lock and there are !fair tasks present
14845  *     0 - failed, no new tasks
14846  *   > 0 - success, new (fair) tasks present
14847  */
14848 static int sched_balance_newidle(struct rq *this_rq, struct rq_flags *rf)
14849 	__must_hold(__rq_lockp(this_rq))
14850 {
14851 	unsigned long next_balance = jiffies + HZ;
14852 	int this_cpu = this_rq->cpu;
14853 	int continue_balancing = 1;
14854 	u64 t0, t1, curr_cost = 0;
14855 	struct sched_domain *sd;
14856 	int pulled_task = 0;
14857 
14858 	update_misfit_status(NULL, this_rq);
14859 
14860 	/*
14861 	 * There is a task waiting to run. No need to search for one.
14862 	 * Return 0; the task will be enqueued when switching to idle.
14863 	 */
14864 	if (this_rq->ttwu_pending)
14865 		return 0;
14866 
14867 	/*
14868 	 * We must set idle_stamp _before_ calling sched_balance_rq()
14869 	 * for CPU_NEWLY_IDLE, such that we measure the this duration
14870 	 * as idle time.
14871 	 */
14872 	this_rq->idle_stamp = rq_clock(this_rq);
14873 
14874 	/*
14875 	 * Do not pull tasks towards !active CPUs...
14876 	 */
14877 	if (!cpu_active(this_cpu))
14878 		return 0;
14879 
14880 	/*
14881 	 * This is OK, because current is on_cpu, which avoids it being picked
14882 	 * for load-balance and preemption/IRQs are still disabled avoiding
14883 	 * further scheduler activity on it and we're being very careful to
14884 	 * re-start the picking loop.
14885 	 */
14886 	rq_unpin_lock(this_rq, rf);
14887 
14888 	sd = rcu_dereference_sched_domain(this_rq->sd);
14889 	if (!sd)
14890 		goto out;
14891 
14892 	if (!get_rd_overloaded(this_rq->rd) ||
14893 	    this_rq->avg_idle < sd->max_newidle_lb_cost) {
14894 
14895 		update_next_balance(sd, &next_balance);
14896 		goto out;
14897 	}
14898 
14899 	/*
14900 	 * Include sched_balance_update_blocked_averages() in the cost
14901 	 * calculation because it can be quite costly -- this ensures we skip
14902 	 * it when avg_idle gets to be very low.
14903 	 */
14904 	t0 = sched_clock_cpu(this_cpu);
14905 	__sched_balance_update_blocked_averages(this_rq);
14906 
14907 	rq_modified_begin(this_rq, &fair_sched_class);
14908 	raw_spin_rq_unlock(this_rq);
14909 
14910 	for_each_domain(this_cpu, sd) {
14911 		u64 domain_cost;
14912 
14913 		update_next_balance(sd, &next_balance);
14914 
14915 		if (this_rq->avg_idle < curr_cost + sd->max_newidle_lb_cost)
14916 			break;
14917 
14918 		if (sd->flags & SD_BALANCE_NEWIDLE) {
14919 			unsigned int weight = 1;
14920 
14921 			if (sched_feat(NI_RANDOM) && sd->newidle_ratio < 1024) {
14922 				/*
14923 				 * Throw a 1k sided dice; and only run
14924 				 * newidle_balance according to the success
14925 				 * rate.
14926 				 */
14927 				u32 d1k = sched_rng() % 1024;
14928 				weight = 1 + sd->newidle_ratio;
14929 				if (d1k > weight) {
14930 					update_newidle_stats(sd, 0);
14931 					continue;
14932 				}
14933 				weight = (1024 + weight/2) / weight;
14934 			}
14935 
14936 			pulled_task = sched_balance_rq(this_cpu, this_rq,
14937 						   sd, CPU_NEWLY_IDLE,
14938 						   &continue_balancing);
14939 
14940 			t1 = sched_clock_cpu(this_cpu);
14941 			domain_cost = t1 - t0;
14942 			curr_cost += domain_cost;
14943 			t0 = t1;
14944 
14945 			/*
14946 			 * Track max cost of a domain to make sure to not delay the
14947 			 * next wakeup on the CPU.
14948 			 */
14949 			update_newidle_cost(sd, domain_cost, weight * !!pulled_task);
14950 		}
14951 
14952 		/*
14953 		 * Stop searching for tasks to pull if there are
14954 		 * now runnable tasks on this rq.
14955 		 */
14956 		if (pulled_task || !continue_balancing)
14957 			break;
14958 	}
14959 
14960 	raw_spin_rq_lock(this_rq);
14961 
14962 	if (curr_cost > this_rq->max_idle_balance_cost)
14963 		this_rq->max_idle_balance_cost = curr_cost;
14964 
14965 	/*
14966 	 * While browsing the domains, we released the rq lock, a task could
14967 	 * have been enqueued in the meantime. Since we're not going idle,
14968 	 * pretend we pulled a task.
14969 	 */
14970 	if (this_rq->cfs.h_nr_queued && !pulled_task)
14971 		pulled_task = 1;
14972 
14973 	/* If a higher prio class was modified, restart the pick */
14974 	if (rq_modified_above(this_rq, &fair_sched_class))
14975 		pulled_task = -1;
14976 
14977 out:
14978 	/* Move the next balance forward */
14979 	if (time_after(this_rq->next_balance, next_balance))
14980 		this_rq->next_balance = next_balance;
14981 
14982 	if (pulled_task)
14983 		this_rq->idle_stamp = 0;
14984 	else
14985 		nohz_newidle_balance(this_rq);
14986 
14987 	rq_repin_lock(this_rq, rf);
14988 
14989 	return pulled_task;
14990 }
14991 
14992 /*
14993  * This softirq handler is triggered via SCHED_SOFTIRQ from two places:
14994  *
14995  * - directly from the local sched_tick() for periodic load balancing
14996  *
14997  * - indirectly from a remote sched_tick() for NOHZ idle balancing
14998  *   through the SMP cross-call nohz_csd_func()
14999  */
15000 static __latent_entropy void sched_balance_softirq(void)
15001 {
15002 	struct rq *this_rq = this_rq();
15003 	enum cpu_idle_type idle = this_rq->idle_balance;
15004 	/*
15005 	 * If this CPU has a pending NOHZ_BALANCE_KICK, then do the
15006 	 * balancing on behalf of the other idle CPUs whose ticks are
15007 	 * stopped. Do nohz_idle_balance *before* sched_balance_domains to
15008 	 * give the idle CPUs a chance to load balance. Else we may
15009 	 * load balance only within the local sched_domain hierarchy
15010 	 * and abort nohz_idle_balance altogether if we pull some load.
15011 	 */
15012 	if (nohz_idle_balance(this_rq, idle))
15013 		return;
15014 
15015 	/* normal load balance */
15016 	sched_balance_update_blocked_averages(this_rq->cpu);
15017 	sched_balance_domains(this_rq, idle);
15018 }
15019 
15020 /*
15021  * Trigger the SCHED_SOFTIRQ if it is time to do periodic load balancing.
15022  */
15023 void sched_balance_trigger(struct rq *rq)
15024 {
15025 	/*
15026 	 * Don't need to rebalance while attached to NULL domain or
15027 	 * runqueue CPU is not active
15028 	 */
15029 	if (unlikely(on_null_domain(rq) || !cpu_active(cpu_of(rq))))
15030 		return;
15031 
15032 	if (time_after_eq(jiffies, rq->next_balance))
15033 		raise_softirq(SCHED_SOFTIRQ);
15034 
15035 	nohz_balancer_kick(rq);
15036 }
15037 
15038 static void rq_online_fair(struct rq *rq)
15039 {
15040 	update_sysctl();
15041 
15042 	update_runtime_enabled(rq);
15043 }
15044 
15045 static void rq_offline_fair(struct rq *rq)
15046 {
15047 	update_sysctl();
15048 
15049 	/* Ensure any throttled groups are reachable by pick_next_task */
15050 	unthrottle_offline_cfs_rqs(rq);
15051 
15052 	/* Ensure that we remove rq contribution to group share: */
15053 	clear_tg_offline_cfs_rqs(rq);
15054 }
15055 
15056 #ifdef CONFIG_SCHED_CORE
15057 static inline bool
15058 __entity_slice_used(struct sched_entity *se, int min_nr_tasks)
15059 {
15060 	u64 rtime = se->sum_exec_runtime - se->prev_sum_exec_runtime;
15061 	u64 slice = se->slice;
15062 
15063 	return (rtime * min_nr_tasks > slice);
15064 }
15065 
15066 #define MIN_NR_TASKS_DURING_FORCEIDLE	2
15067 static inline void task_tick_core(struct rq *rq, struct task_struct *curr)
15068 {
15069 	if (!sched_core_enabled(rq))
15070 		return;
15071 
15072 	/*
15073 	 * If runqueue has only one task which used up its slice and
15074 	 * if the sibling is forced idle, then trigger schedule to
15075 	 * give forced idle task a chance.
15076 	 *
15077 	 * __entity_slice_used() considers only this active rq and it gets the
15078 	 * whole slice. But during force idle, we have siblings acting
15079 	 * like a single runqueue and hence we need to consider runnable
15080 	 * tasks on this CPU and the forced idle CPU. Ideally, we should
15081 	 * go through the forced idle rq, but that would be a perf hit.
15082 	 * We can assume that the forced idle CPU has at least
15083 	 * MIN_NR_TASKS_DURING_FORCEIDLE - 1 tasks and use that to check
15084 	 * if we need to give up the CPU.
15085 	 */
15086 	if (rq->core->core_forceidle_count && rq->cfs.h_nr_queued == 1 &&
15087 	    __entity_slice_used(&curr->se, MIN_NR_TASKS_DURING_FORCEIDLE))
15088 		resched_curr(rq);
15089 }
15090 
15091 /*
15092  * Consider any infeasible weight scenario. Take for instance two tasks,
15093  * each bound to their respective sibling, one with weight 1 and one with
15094  * weight 2. Then the lower weight task will run ahead of the higher weight
15095  * task without bound.
15096  *
15097  * This utterly destroys the concept of a shared time base.
15098  *
15099  * Remember; all this is about a proportionally fair scheduling, where each
15100  * tasks receives:
15101  *
15102  *              w_i
15103  *   dt_i = ---------- dt                                     (1)
15104  *          \Sum_j w_j
15105  *
15106  * which we do by tracking a virtual time, s_i:
15107  *
15108  *          1
15109  *   s_i = --- d[t]_i                                         (2)
15110  *         w_i
15111  *
15112  * Where d[t] is a delta of discrete time, while dt is an infinitesimal.
15113  * The immediate corollary is that the ideal schedule S, where (2) to use
15114  * an infinitesimal delta, is:
15115  *
15116  *           1
15117  *   S = ---------- dt                                        (3)
15118  *       \Sum_i w_i
15119  *
15120  * From which we can define the lag, or deviation from the ideal, as:
15121  *
15122  *   lag(i) = S - s_i                                         (4)
15123  *
15124  * And since the one and only purpose is to approximate S, we get that:
15125  *
15126  *   \Sum_i w_i lag(i) := 0                                   (5)
15127  *
15128  * If this were not so, we no longer converge to S, and we can no longer
15129  * claim our scheduler has any of the properties we derive from S. This is
15130  * exactly what you did above, you broke it!
15131  *
15132  *
15133  * Let's continue for a while though; to see if there is anything useful to
15134  * be learned. We can combine (1)-(3) or (4)-(5) and express S in s_i:
15135  *
15136  *       \Sum_i w_i s_i
15137  *   S = --------------                                       (6)
15138  *         \Sum_i w_i
15139  *
15140  * Which gives us a way to compute S, given our s_i. Now, if you've read
15141  * our code, you know that we do not in fact do this, the reason for this
15142  * is two-fold. Firstly, computing S in that way requires a 64bit division
15143  * for every time we'd use it (see 12), and secondly, this only describes
15144  * the steady-state, it doesn't handle dynamics.
15145  *
15146  * Anyway, in (6):  s_i -> x + (s_i - x), to get:
15147  *
15148  *           \Sum_i w_i (s_i - x)
15149  *   S - x = --------------------                             (7)
15150  *              \Sum_i w_i
15151  *
15152  * Which shows that S and s_i transform alike (which makes perfect sense
15153  * given that S is basically the (weighted) average of s_i).
15154  *
15155  * So the thing to remember is that the above is strictly UP. It is
15156  * possible to generalize to multiple runqueues -- however it gets really
15157  * yuck when you have to add affinity support, as illustrated by our very
15158  * first counter-example.
15159  *
15160  * Luckily I think we can avoid needing a full multi-queue variant for
15161  * core-scheduling (or load-balancing). The crucial observation is that we
15162  * only actually need this comparison in the presence of forced-idle; only
15163  * then do we need to tell if the stalled rq has higher priority over the
15164  * other.
15165  *
15166  * [XXX assumes SMT2; better consider the more general case, I suspect
15167  * it'll work out because our comparison is always between 2 rqs and the
15168  * answer is only interesting if one of them is forced-idle]
15169  *
15170  * And (under assumption of SMT2) when there is forced-idle, there is only
15171  * a single queue, so everything works like normal.
15172  *
15173  * Let, for our runqueue 'k':
15174  *
15175  *   T_k = \Sum_i w_i s_i
15176  *   W_k = \Sum_i w_i      ; for all i of k                  (8)
15177  *
15178  * Then we can write (6) like:
15179  *
15180  *         T_k
15181  *   S_k = ---                                               (9)
15182  *         W_k
15183  *
15184  * From which immediately follows that:
15185  *
15186  *           T_k + T_l
15187  *   S_k+l = ---------                                       (10)
15188  *           W_k + W_l
15189  *
15190  * On which we can define a combined lag:
15191  *
15192  *   lag_k+l(i) := S_k+l - s_i                               (11)
15193  *
15194  * And that gives us the tools to compare tasks across a combined runqueue.
15195  *
15196  *
15197  * Combined this gives the following:
15198  *
15199  *  a) when a runqueue enters force-idle, sync it against it's sibling rq(s)
15200  *     using (7); this only requires storing single 'time'-stamps.
15201  *
15202  *  b) when comparing tasks between 2 runqueues of which one is forced-idle,
15203  *     compare the combined lag, per (11).
15204  *
15205  * Now, of course cgroups (I so hate them) make this more interesting in
15206  * that a) seems to suggest we need to iterate all cgroup on a CPU at such
15207  * boundaries, but I think we can avoid that. The force-idle is for the
15208  * whole CPU, all it's rqs. So we can mark it in the root and lazily
15209  * propagate downward on demand.
15210  */
15211 
15212 /*
15213  * So this sync is basically a relative reset of S to 0.
15214  *
15215  * So with 2 queues, when one goes idle, we drop them both to 0 and one
15216  * then increases due to not being idle, and the idle one builds up lag to
15217  * get re-elected. So far so simple, right?
15218  *
15219  * When there's 3, we can have the situation where 2 run and one is idle,
15220  * we sync to 0 and let the idle one build up lag to get re-election. Now
15221  * suppose another one also drops idle. At this point dropping all to 0
15222  * again would destroy the built-up lag from the queue that was already
15223  * idle, not good.
15224  *
15225  * So instead of syncing everything, we can:
15226  *
15227  *   less := !((s64)(s_a - s_b) <= 0)
15228  *
15229  *   (v_a - S_a) - (v_b - S_b) == v_a - v_b - S_a + S_b
15230  *                             == v_a - (v_b - S_a + S_b)
15231  *
15232  * IOW, we can recast the (lag) comparison to a one-sided difference.
15233  * So if then, instead of syncing the whole queue, sync the idle queue
15234  * against the active queue with S_a + S_b at the point where we sync.
15235  *
15236  * (XXX consider the implication of living in a cyclic group: N / 2^n N)
15237  *
15238  * This gives us means of syncing single queues against the active queue,
15239  * and for already idle queues to preserve their build-up lag.
15240  *
15241  * Of course, then we get the situation where there's 2 active and one
15242  * going idle, who do we pick to sync against? Theory would have us sync
15243  * against the combined S, but as we've already demonstrated, there is no
15244  * such thing in infeasible weight scenarios.
15245  *
15246  * One thing I've considered; and this is where that core_active rudiment
15247  * came from, is having active queues sync up between themselves after
15248  * every tick. This limits the observed divergence due to the work
15249  * conservancy.
15250  *
15251  * On top of that, we can improve upon things by employing (10) here.
15252  */
15253 
15254 /*
15255  * se_fi_update - Update the cfs_rq->zero_vruntime_fi in a CFS hierarchy if needed.
15256  */
15257 static void se_fi_update(const struct sched_entity *se, unsigned int fi_seq,
15258 			 bool forceidle)
15259 {
15260 	for_each_sched_entity(se) {
15261 		struct cfs_rq *cfs_rq = cfs_rq_of(se);
15262 
15263 		if (forceidle) {
15264 			if (cfs_rq->forceidle_seq == fi_seq)
15265 				break;
15266 			cfs_rq->forceidle_seq = fi_seq;
15267 		}
15268 
15269 		cfs_rq->zero_vruntime_fi = cfs_rq->zero_vruntime;
15270 	}
15271 }
15272 
15273 void task_vruntime_update(struct rq *rq, struct task_struct *p, bool in_fi)
15274 {
15275 	struct sched_entity *se = &p->se;
15276 
15277 	if (p->sched_class != &fair_sched_class)
15278 		return;
15279 
15280 	se_fi_update(se, rq->core->core_forceidle_seq, in_fi);
15281 }
15282 
15283 bool cfs_prio_less(const struct task_struct *a, const struct task_struct *b,
15284 			bool in_fi)
15285 {
15286 	struct rq *rq = task_rq(a);
15287 	const struct sched_entity *sea = &a->se;
15288 	const struct sched_entity *seb = &b->se;
15289 	struct cfs_rq *cfs_rqa;
15290 	struct cfs_rq *cfs_rqb;
15291 	s64 delta;
15292 
15293 	WARN_ON_ONCE(task_rq(b)->core != rq->core);
15294 
15295 	cfs_rqa = &task_rq(a)->cfs;
15296 	cfs_rqb = &task_rq(b)->cfs;
15297 
15298 	/*
15299 	 * Find delta after normalizing se's vruntime with its cfs_rq's
15300 	 * zero_vruntime_fi, which would have been updated in prior calls
15301 	 * to se_fi_update().
15302 	 */
15303 	delta = vruntime_op(sea->vruntime, "-", seb->vruntime) +
15304 		vruntime_op(cfs_rqb->zero_vruntime_fi, "-", cfs_rqa->zero_vruntime_fi);
15305 
15306 	return delta > 0;
15307 }
15308 
15309 static int task_is_throttled_fair(struct task_struct *p, int cpu)
15310 {
15311 	struct cfs_rq *cfs_rq;
15312 
15313 #ifdef CONFIG_FAIR_GROUP_SCHED
15314 	cfs_rq = tg_cfs_rq(task_group(p), cpu);
15315 #else
15316 	cfs_rq = &cpu_rq(cpu)->cfs;
15317 #endif
15318 	return throttled_hierarchy(cfs_rq);
15319 }
15320 #else /* !CONFIG_SCHED_CORE: */
15321 static inline void task_tick_core(struct rq *rq, struct task_struct *curr) {}
15322 #endif /* !CONFIG_SCHED_CORE */
15323 
15324 /*
15325  * scheduler tick hitting a task of our scheduling class.
15326  *
15327  * NOTE: This function can be called remotely by the tick offload that
15328  * goes along full dynticks. Therefore no local assumption can be made
15329  * and everything must be accessed through the @rq and @curr passed in
15330  * parameters.
15331  */
15332 static void task_tick_fair(struct rq *rq, struct task_struct *curr, int queued)
15333 {
15334 	struct sched_entity *se = &curr->se;
15335 
15336 	if (se->on_rq) {
15337 		unsigned long weight = NICE_0_LOAD;
15338 		struct cfs_rq *cfs_rq;
15339 
15340 		for_each_sched_entity(se) {
15341 			cfs_rq = cfs_rq_of(se);
15342 			entity_tick(cfs_rq, se, queued);
15343 
15344 			weight = __calc_prop_weight(cfs_rq, se, weight);
15345 		}
15346 
15347 		se = &curr->se;
15348 		reweight_eevdf(cfs_rq, se, weight, se->on_rq);
15349 	}
15350 
15351 	if (queued)
15352 		return;
15353 
15354 	if (static_branch_unlikely(&sched_numa_balancing))
15355 		task_tick_numa(rq, curr);
15356 
15357 	task_tick_cache(rq, curr);
15358 
15359 	update_misfit_status(curr, rq);
15360 	check_update_overutilized_status(task_rq(curr));
15361 
15362 	task_tick_core(rq, curr);
15363 }
15364 
15365 /*
15366  * called on fork with the child task as argument from the parent's context
15367  *  - child not yet on the tasklist
15368  *  - preemption disabled
15369  */
15370 static void task_fork_fair(struct task_struct *p)
15371 {
15372 	set_task_max_allowed_capacity(p);
15373 }
15374 
15375 /*
15376  * Priority of the task has changed. Check to see if we preempt
15377  * the current task.
15378  */
15379 static void
15380 prio_changed_fair(struct rq *rq, struct task_struct *p, u64 oldprio)
15381 {
15382 	if (!task_on_rq_queued(p))
15383 		return;
15384 
15385 	if (p->prio == oldprio)
15386 		return;
15387 
15388 	if (rq->cfs.h_nr_queued == 1)
15389 		return;
15390 
15391 	/*
15392 	 * Reschedule if we are currently running on this runqueue and
15393 	 * our priority decreased, or if we are not currently running on
15394 	 * this runqueue and our priority is higher than the current's
15395 	 */
15396 	if (task_current_donor(rq, p)) {
15397 		if (p->prio > oldprio)
15398 			resched_curr(rq);
15399 	} else {
15400 		wakeup_preempt(rq, p, 0);
15401 	}
15402 }
15403 
15404 #ifdef CONFIG_FAIR_GROUP_SCHED
15405 /*
15406  * Propagate the changes of the sched_entity across the tg tree to make it
15407  * visible to the root
15408  */
15409 static void propagate_entity_cfs_rq(struct sched_entity *se)
15410 {
15411 	struct cfs_rq *cfs_rq = cfs_rq_of(se);
15412 
15413 	/*
15414 	 * If a task gets attached to this cfs_rq and before being queued,
15415 	 * it gets migrated to another CPU due to reasons like affinity
15416 	 * change, make sure this cfs_rq stays on leaf cfs_rq list to have
15417 	 * that removed load decayed or it can cause faireness problem.
15418 	 */
15419 	if (!cfs_rq_pelt_clock_throttled(cfs_rq))
15420 		list_add_leaf_cfs_rq(cfs_rq);
15421 
15422 	/* Start to propagate at parent */
15423 	se = se->parent;
15424 
15425 	for_each_sched_entity(se) {
15426 		cfs_rq = cfs_rq_of(se);
15427 
15428 		update_load_avg(cfs_rq, se, UPDATE_TG);
15429 
15430 		if (!cfs_rq_pelt_clock_throttled(cfs_rq))
15431 			list_add_leaf_cfs_rq(cfs_rq);
15432 	}
15433 
15434 	assert_list_leaf_cfs_rq(rq_of(cfs_rq));
15435 }
15436 #else /* !CONFIG_FAIR_GROUP_SCHED: */
15437 static void propagate_entity_cfs_rq(struct sched_entity *se) { }
15438 #endif /* !CONFIG_FAIR_GROUP_SCHED */
15439 
15440 static void detach_entity_cfs_rq(struct sched_entity *se)
15441 {
15442 	struct cfs_rq *cfs_rq = cfs_rq_of(se);
15443 
15444 	/*
15445 	 * In case the task sched_avg hasn't been attached:
15446 	 * - A forked task which hasn't been woken up by wake_up_new_task().
15447 	 * - A task which has been woken up by try_to_wake_up() but is
15448 	 *   waiting for actually being woken up by sched_ttwu_pending().
15449 	 */
15450 	if (!se->avg.last_update_time)
15451 		return;
15452 
15453 	/* Catch up with the cfs_rq and remove our load when we leave */
15454 	update_load_avg(cfs_rq, se, 0);
15455 	detach_entity_load_avg(cfs_rq, se);
15456 	update_tg_load_avg(cfs_rq);
15457 	propagate_entity_cfs_rq(se);
15458 }
15459 
15460 static void attach_entity_cfs_rq(struct sched_entity *se)
15461 {
15462 	struct cfs_rq *cfs_rq = cfs_rq_of(se);
15463 
15464 	/* Synchronize entity with its cfs_rq */
15465 	update_load_avg(cfs_rq, se, sched_feat(ATTACH_AGE_LOAD) ? 0 : SKIP_AGE_LOAD);
15466 	attach_entity_load_avg(cfs_rq, se);
15467 	update_tg_load_avg(cfs_rq);
15468 	propagate_entity_cfs_rq(se);
15469 }
15470 
15471 static void detach_task_cfs_rq(struct task_struct *p)
15472 {
15473 	struct sched_entity *se = &p->se;
15474 
15475 	detach_entity_cfs_rq(se);
15476 }
15477 
15478 static void attach_task_cfs_rq(struct task_struct *p)
15479 {
15480 	struct sched_entity *se = &p->se;
15481 
15482 	attach_entity_cfs_rq(se);
15483 }
15484 
15485 static void switching_from_fair(struct rq *rq, struct task_struct *p)
15486 {
15487 	if (p->se.sched_delayed)
15488 		dequeue_task(rq, p, DEQUEUE_SLEEP | DEQUEUE_DELAYED | DEQUEUE_NOCLOCK);
15489 }
15490 
15491 static void switched_from_fair(struct rq *rq, struct task_struct *p)
15492 {
15493 	detach_task_cfs_rq(p);
15494 }
15495 
15496 static void switched_to_fair(struct rq *rq, struct task_struct *p)
15497 {
15498 	WARN_ON_ONCE(p->se.sched_delayed);
15499 
15500 	attach_task_cfs_rq(p);
15501 
15502 	set_task_max_allowed_capacity(p);
15503 
15504 	if (task_on_rq_queued(p)) {
15505 		/*
15506 		 * We were most likely switched from sched_rt, so
15507 		 * kick off the schedule if running, otherwise just see
15508 		 * if we can still preempt the current task.
15509 		 */
15510 		if (task_current_donor(rq, p))
15511 			resched_curr(rq);
15512 		else
15513 			wakeup_preempt(rq, p, 0);
15514 	}
15515 }
15516 
15517 static void set_next_task_fair(struct rq *rq, struct task_struct *p, bool first)
15518 {
15519 	struct sched_entity *se = &p->se;
15520 	bool throttled = false;
15521 	struct cfs_rq *cfs_rq = &rq->cfs;
15522 	unsigned long weight = NICE_0_LOAD;
15523 	bool on_rq = se->on_rq;
15524 
15525 	clear_buddies(cfs_rq, se);
15526 
15527 	if (on_rq)
15528 		__dequeue_entity(cfs_rq, se);
15529 
15530 	for_each_sched_entity(se) {
15531 		cfs_rq = cfs_rq_of(se);
15532 
15533 		if (!IS_ENABLED(CONFIG_FAIR_GROUP_SCHED) ||
15534 		    !first || !cfs_rq->h_curr)
15535 			set_next_entity(cfs_rq, se);
15536 
15537 		/* ensure bandwidth has been allocated on our new cfs_rq */
15538 		throttled |= account_cfs_rq_runtime(cfs_rq, 0);
15539 
15540 		if (on_rq)
15541 			weight = __calc_prop_weight(cfs_rq, se, weight);
15542 	}
15543 
15544 	if (throttled)
15545 		task_throttle_setup_work(p);
15546 
15547 	se = &p->se;
15548 	cfs_rq->curr = se;
15549 
15550 	if (on_rq) {
15551 		reweight_eevdf(cfs_rq, se, weight, se->on_rq);
15552 		if (first)
15553 			set_protect_slice(cfs_rq, se);
15554 	}
15555 
15556 	if (task_on_rq_queued(p)) {
15557 		/*
15558 		 * Move the next running task to the front of the list, so our
15559 		 * cfs_tasks list becomes MRU one.
15560 		 */
15561 		list_move(&se->group_node, &rq->cfs_tasks);
15562 	}
15563 	if (!first)
15564 		return;
15565 
15566 	WARN_ON_ONCE(se->sched_delayed);
15567 
15568 	if (hrtick_enabled_fair(rq))
15569 		hrtick_start_fair(rq, p);
15570 
15571 	update_misfit_status(p, rq);
15572 	sched_fair_update_stop_tick(rq, p);
15573 }
15574 
15575 void init_cfs_rq(struct cfs_rq *cfs_rq)
15576 {
15577 	cfs_rq->tasks_timeline = RB_ROOT_CACHED;
15578 	cfs_rq->zero_vruntime = (u64)(-(1LL << 20));
15579 	raw_spin_lock_init(&cfs_rq->removed.lock);
15580 }
15581 
15582 #ifdef CONFIG_FAIR_GROUP_SCHED
15583 static void task_change_group_fair(struct task_struct *p)
15584 {
15585 	/*
15586 	 * We couldn't detach or attach a forked task which
15587 	 * hasn't been woken up by wake_up_new_task().
15588 	 */
15589 	if (READ_ONCE(p->__state) == TASK_NEW)
15590 		return;
15591 
15592 	detach_task_cfs_rq(p);
15593 
15594 	/* Tell se's cfs_rq has been changed -- migrated */
15595 	p->se.avg.last_update_time = 0;
15596 	set_task_rq(p, task_cpu(p));
15597 	attach_task_cfs_rq(p);
15598 }
15599 
15600 void free_fair_sched_group(struct task_group *tg)
15601 {
15602 	free_percpu(tg->cfs_rq);
15603 }
15604 
15605 int alloc_fair_sched_group(struct task_group *tg, struct task_group *parent)
15606 {
15607 	struct cfs_tg_state __percpu *state;
15608 	struct sched_entity *se;
15609 	struct cfs_rq *cfs_rq;
15610 	int i;
15611 
15612 	state = alloc_percpu_gfp(struct cfs_tg_state, GFP_KERNEL);
15613 	if (!state)
15614 		goto err;
15615 
15616 	tg->cfs_rq = &state->cfs_rq;
15617 	tg->shares = NICE_0_LOAD;
15618 
15619 	init_cfs_bandwidth(tg_cfs_bandwidth(tg), tg_cfs_bandwidth(parent));
15620 
15621 	for_each_possible_cpu(i) {
15622 		cfs_rq = tg_cfs_rq(tg, i);
15623 		if (!cfs_rq)
15624 			goto err;
15625 
15626 		se = tg_se(tg, i);
15627 		init_cfs_rq(cfs_rq);
15628 		init_tg_cfs_entry(tg, cfs_rq, se, i, tg_se(parent, i));
15629 		init_entity_runnable_average(se);
15630 	}
15631 
15632 	return 1;
15633 
15634 err:
15635 	return 0;
15636 }
15637 
15638 void online_fair_sched_group(struct task_group *tg)
15639 {
15640 	struct sched_entity *se;
15641 	struct rq_flags rf;
15642 	struct rq *rq;
15643 	int i;
15644 
15645 	for_each_possible_cpu(i) {
15646 		rq = cpu_rq(i);
15647 		se = tg_se(tg, i);
15648 		rq_lock_irq(rq, &rf);
15649 		update_rq_clock(rq);
15650 		attach_entity_cfs_rq(se);
15651 		sync_throttle(tg, i);
15652 		rq_unlock_irq(rq, &rf);
15653 	}
15654 }
15655 
15656 void unregister_fair_sched_group(struct task_group *tg)
15657 {
15658 	int cpu;
15659 
15660 	destroy_cfs_bandwidth(tg_cfs_bandwidth(tg));
15661 
15662 	for_each_possible_cpu(cpu) {
15663 		struct cfs_rq *cfs_rq = tg_cfs_rq(tg, cpu);
15664 		struct sched_entity *se = tg_se(tg, cpu);
15665 		struct rq *rq = cpu_rq(cpu);
15666 
15667 		if (se)
15668 			remove_entity_load_avg(se);
15669 
15670 		/*
15671 		 * Only empty task groups can be destroyed; so we can speculatively
15672 		 * check on_list without danger of it being re-added.
15673 		 */
15674 		if (cfs_rq->on_list) {
15675 			guard(rq_lock_irqsave)(rq);
15676 			list_del_leaf_cfs_rq(cfs_rq);
15677 		}
15678 	}
15679 }
15680 
15681 void init_tg_cfs_entry(struct task_group *tg, struct cfs_rq *cfs_rq,
15682 			struct sched_entity *se, int cpu,
15683 			struct sched_entity *parent)
15684 {
15685 	struct rq *rq = cpu_rq(cpu);
15686 
15687 	cfs_rq->tg = tg;
15688 	cfs_rq->rq = rq;
15689 	init_cfs_rq_runtime(cfs_rq);
15690 
15691 	/* se could be NULL for root_task_group */
15692 	if (!se)
15693 		return;
15694 
15695 	if (!parent) {
15696 		se->cfs_rq = &rq->cfs;
15697 		se->depth = 0;
15698 	} else {
15699 		se->cfs_rq = parent->my_q;
15700 		se->depth = parent->depth + 1;
15701 	}
15702 
15703 	se->my_q = cfs_rq;
15704 	/* guarantee group entities always have weight */
15705 	update_load_set(&se->load, NICE_0_LOAD);
15706 	se->parent = parent;
15707 }
15708 
15709 static DEFINE_MUTEX(shares_mutex);
15710 
15711 static int __sched_group_set_shares(struct task_group *tg, unsigned long shares)
15712 {
15713 	int i;
15714 
15715 	lockdep_assert_held(&shares_mutex);
15716 
15717 	/*
15718 	 * We can't change the weight of the root cgroup.
15719 	 */
15720 	if (is_root_task_group(tg))
15721 		return -EINVAL;
15722 
15723 	shares = clamp(shares, scale_load(MIN_SHARES), scale_load(MAX_SHARES));
15724 
15725 	if (tg->shares == shares)
15726 		return 0;
15727 
15728 	tg->shares = shares;
15729 	for_each_possible_cpu(i) {
15730 		struct rq *rq = cpu_rq(i);
15731 		struct sched_entity *se = tg_se(tg, i);
15732 		struct rq_flags rf;
15733 
15734 		/* Propagate contribution to hierarchy */
15735 		rq_lock_irqsave(rq, &rf);
15736 		update_rq_clock(rq);
15737 		for_each_sched_entity(se) {
15738 			update_load_avg(cfs_rq_of(se), se, UPDATE_TG);
15739 			update_cfs_group(se);
15740 		}
15741 		rq_unlock_irqrestore(rq, &rf);
15742 	}
15743 
15744 	return 0;
15745 }
15746 
15747 int sched_group_set_shares(struct task_group *tg, unsigned long shares)
15748 {
15749 	int ret;
15750 
15751 	mutex_lock(&shares_mutex);
15752 	if (tg_is_idle(tg))
15753 		ret = -EINVAL;
15754 	else
15755 		ret = __sched_group_set_shares(tg, shares);
15756 	mutex_unlock(&shares_mutex);
15757 
15758 	return ret;
15759 }
15760 
15761 int sched_group_set_idle(struct task_group *tg, long idle)
15762 {
15763 	int i;
15764 
15765 	if (tg == &root_task_group)
15766 		return -EINVAL;
15767 
15768 	if (idle < 0 || idle > 1)
15769 		return -EINVAL;
15770 
15771 	mutex_lock(&shares_mutex);
15772 
15773 	if (tg->idle == idle) {
15774 		mutex_unlock(&shares_mutex);
15775 		return 0;
15776 	}
15777 
15778 	tg->idle = idle;
15779 
15780 	for_each_possible_cpu(i) {
15781 		struct rq *rq = cpu_rq(i);
15782 		struct sched_entity *se = tg_se(tg, i);
15783 		struct cfs_rq *grp_cfs_rq = tg_cfs_rq(tg, i);
15784 		bool was_idle = cfs_rq_is_idle(grp_cfs_rq);
15785 		long idle_task_delta;
15786 		struct rq_flags rf;
15787 
15788 		rq_lock_irqsave(rq, &rf);
15789 
15790 		grp_cfs_rq->idle = idle;
15791 		if (WARN_ON_ONCE(was_idle == cfs_rq_is_idle(grp_cfs_rq)))
15792 			goto next_cpu;
15793 
15794 		idle_task_delta = grp_cfs_rq->h_nr_queued -
15795 				  grp_cfs_rq->h_nr_idle;
15796 		if (!cfs_rq_is_idle(grp_cfs_rq))
15797 			idle_task_delta *= -1;
15798 
15799 		for_each_sched_entity(se) {
15800 			struct cfs_rq *cfs_rq = cfs_rq_of(se);
15801 
15802 			if (!se->on_rq)
15803 				break;
15804 
15805 			cfs_rq->h_nr_idle += idle_task_delta;
15806 
15807 			/* Already accounted at parent level and above. */
15808 			if (cfs_rq_is_idle(cfs_rq))
15809 				break;
15810 		}
15811 
15812 next_cpu:
15813 		rq_unlock_irqrestore(rq, &rf);
15814 	}
15815 
15816 	/* Idle groups have minimum weight. */
15817 	if (tg_is_idle(tg))
15818 		__sched_group_set_shares(tg, scale_load(WEIGHT_IDLEPRIO));
15819 	else
15820 		__sched_group_set_shares(tg, NICE_0_LOAD);
15821 
15822 	mutex_unlock(&shares_mutex);
15823 	return 0;
15824 }
15825 
15826 #endif /* CONFIG_FAIR_GROUP_SCHED */
15827 
15828 
15829 static unsigned int get_rr_interval_fair(struct rq *rq, struct task_struct *task)
15830 {
15831 	struct sched_entity *se = &task->se;
15832 	unsigned int rr_interval = 0;
15833 
15834 	/*
15835 	 * Time slice is 0 for SCHED_OTHER tasks that are on an otherwise
15836 	 * idle runqueue:
15837 	 */
15838 	if (rq->cfs.load.weight)
15839 		rr_interval = NS_TO_JIFFIES(se->slice);
15840 
15841 	return rr_interval;
15842 }
15843 
15844 /*
15845  * All the scheduling class methods:
15846  */
15847 DEFINE_SCHED_CLASS(fair) = {
15848 	.enqueue_task		= enqueue_task_fair,
15849 	.dequeue_task		= dequeue_task_fair,
15850 	.yield_task		= yield_task_fair,
15851 	.yield_to_task		= yield_to_task_fair,
15852 
15853 	.wakeup_preempt		= wakeup_preempt_fair,
15854 
15855 	.pick_task		= pick_task_fair,
15856 	.put_prev_task		= put_prev_task_fair,
15857 	.set_next_task          = set_next_task_fair,
15858 
15859 	.select_task_rq		= select_task_rq_fair,
15860 	.migrate_task_rq	= migrate_task_rq_fair,
15861 
15862 	.rq_online		= rq_online_fair,
15863 	.rq_offline		= rq_offline_fair,
15864 
15865 	.task_dead		= task_dead_fair,
15866 	.set_cpus_allowed	= set_cpus_allowed_fair,
15867 
15868 	.task_tick		= task_tick_fair,
15869 	.task_fork		= task_fork_fair,
15870 
15871 	.reweight_task		= reweight_task_fair,
15872 	.prio_changed		= prio_changed_fair,
15873 	.switching_from		= switching_from_fair,
15874 	.switched_from		= switched_from_fair,
15875 	.switched_to		= switched_to_fair,
15876 
15877 	.get_rr_interval	= get_rr_interval_fair,
15878 
15879 	.update_curr		= update_curr_fair,
15880 
15881 #ifdef CONFIG_FAIR_GROUP_SCHED
15882 	.task_change_group	= task_change_group_fair,
15883 #endif
15884 
15885 #ifdef CONFIG_SCHED_CORE
15886 	.task_is_throttled	= task_is_throttled_fair,
15887 #endif
15888 
15889 #ifdef CONFIG_UCLAMP_TASK
15890 	.uclamp_enabled		= 1,
15891 #endif
15892 };
15893 
15894 void print_cfs_stats(struct seq_file *m, int cpu)
15895 {
15896 	struct cfs_rq *cfs_rq, *pos;
15897 
15898 	rcu_read_lock();
15899 	for_each_leaf_cfs_rq_safe(cpu_rq(cpu), cfs_rq, pos)
15900 		print_cfs_rq(m, cpu, cfs_rq);
15901 	rcu_read_unlock();
15902 }
15903 
15904 #ifdef CONFIG_NUMA_BALANCING
15905 void show_numa_stats(struct task_struct *p, struct seq_file *m)
15906 {
15907 	int node;
15908 	unsigned long tsf = 0, tpf = 0, gsf = 0, gpf = 0;
15909 	struct numa_group *ng;
15910 
15911 	rcu_read_lock();
15912 	ng = rcu_dereference_all(p->numa_group);
15913 	for_each_online_node(node) {
15914 		if (p->numa_faults) {
15915 			tsf = p->numa_faults[task_faults_idx(NUMA_MEM, node, 0)];
15916 			tpf = p->numa_faults[task_faults_idx(NUMA_MEM, node, 1)];
15917 		}
15918 		if (ng) {
15919 			gsf = ng->faults[task_faults_idx(NUMA_MEM, node, 0)];
15920 			gpf = ng->faults[task_faults_idx(NUMA_MEM, node, 1)];
15921 		}
15922 		print_numa_stats(m, node, tsf, tpf, gsf, gpf);
15923 	}
15924 	rcu_read_unlock();
15925 }
15926 #endif /* CONFIG_NUMA_BALANCING */
15927 
15928 __init void init_sched_fair_class(void)
15929 {
15930 	int i;
15931 
15932 	for_each_possible_cpu(i) {
15933 		zalloc_cpumask_var_node(&per_cpu(load_balance_mask, i), GFP_KERNEL, cpu_to_node(i));
15934 		zalloc_cpumask_var_node(&per_cpu(select_rq_mask,    i), GFP_KERNEL, cpu_to_node(i));
15935 		zalloc_cpumask_var_node(&per_cpu(should_we_balance_tmpmask, i),
15936 					GFP_KERNEL, cpu_to_node(i));
15937 
15938 #ifdef CONFIG_CFS_BANDWIDTH
15939 		INIT_CSD(&cpu_rq(i)->cfsb_csd, __cfsb_csd_unthrottle, cpu_rq(i));
15940 		INIT_LIST_HEAD(&cpu_rq(i)->cfsb_csd_list);
15941 #endif
15942 	}
15943 
15944 	open_softirq(SCHED_SOFTIRQ, sched_balance_softirq);
15945 
15946 #ifdef CONFIG_NO_HZ_COMMON
15947 	nohz.next_balance = jiffies;
15948 	nohz.next_blocked = jiffies;
15949 	zalloc_cpumask_var(&nohz.idle_cpus_mask, GFP_NOWAIT);
15950 #endif
15951 }
15952