1 /* SPDX-License-Identifier: GPL-2.0-or-later */
2
3 #ifndef __CPUSET_INTERNAL_H
4 #define __CPUSET_INTERNAL_H
5
6 #include <linux/cgroup.h>
7 #include <linux/cpu.h>
8 #include <linux/cpumask.h>
9 #include <linux/cpuset.h>
10 #include <linux/spinlock.h>
11 #include <linux/union_find.h>
12 #include <linux/sched/isolation.h>
13
14 /* See "Frequency meter" comments, below. */
15
16 struct fmeter {
17 int cnt; /* unprocessed events count */
18 int val; /* most recent output value */
19 time64_t time; /* clock (secs) when val computed */
20 spinlock_t lock; /* guards read or write of above */
21 };
22
23 /*
24 * Invalid partition error code
25 */
26 enum prs_errcode {
27 PERR_NONE = 0,
28 PERR_INVCPUS,
29 PERR_INVPARENT,
30 PERR_NOTPART,
31 PERR_NOTEXCL,
32 PERR_NOCPUS,
33 PERR_HOTPLUG,
34 PERR_CPUSEMPTY,
35 PERR_HKEEPING,
36 PERR_ACCESS,
37 PERR_REMOTE,
38 };
39
40 /* bits in struct cpuset flags field */
41 typedef enum {
42 CS_CPU_EXCLUSIVE,
43 CS_MEM_EXCLUSIVE,
44 CS_MEM_HARDWALL,
45 CS_MEMORY_MIGRATE,
46 CS_SCHED_LOAD_BALANCE,
47 CS_SPREAD_PAGE,
48 CS_SPREAD_SLAB,
49 } cpuset_flagbits_t;
50
51 /* The various types of files and directories in a cpuset file system */
52
53 typedef enum {
54 FILE_MEMORY_MIGRATE,
55 FILE_CPULIST,
56 FILE_MEMLIST,
57 FILE_EFFECTIVE_CPULIST,
58 FILE_EFFECTIVE_MEMLIST,
59 FILE_SUBPARTS_CPULIST,
60 FILE_EXCLUSIVE_CPULIST,
61 FILE_EFFECTIVE_XCPULIST,
62 FILE_ISOLATED_CPULIST,
63 FILE_CPU_EXCLUSIVE,
64 FILE_MEM_EXCLUSIVE,
65 FILE_MEM_HARDWALL,
66 FILE_SCHED_LOAD_BALANCE,
67 FILE_PARTITION_ROOT,
68 FILE_SCHED_RELAX_DOMAIN_LEVEL,
69 FILE_MEMORY_PRESSURE_ENABLED,
70 FILE_MEMORY_PRESSURE,
71 FILE_SPREAD_PAGE,
72 FILE_SPREAD_SLAB,
73 } cpuset_filetype_t;
74
75 struct cpuset {
76 struct cgroup_subsys_state css;
77
78 unsigned long flags; /* "unsigned long" so bitops work */
79
80 /*
81 * On default hierarchy:
82 *
83 * The user-configured masks can only be changed by writing to
84 * cpuset.cpus and cpuset.mems, and won't be limited by the
85 * parent masks.
86 *
87 * The effective masks is the real masks that apply to the tasks
88 * in the cpuset. They may be changed if the configured masks are
89 * changed or hotplug happens.
90 *
91 * effective_mask == configured_mask & parent's effective_mask,
92 * and if it ends up empty, it will inherit the parent's mask.
93 *
94 *
95 * On legacy hierarchy:
96 *
97 * The user-configured masks are always the same with effective masks.
98 */
99
100 /* user-configured CPUs and Memory Nodes allow to tasks */
101 cpumask_var_t cpus_allowed;
102 nodemask_t mems_allowed;
103
104 /* effective CPUs and Memory Nodes allow to tasks */
105 cpumask_var_t effective_cpus;
106 nodemask_t effective_mems;
107
108 /*
109 * Exclusive CPUs dedicated to current cgroup (default hierarchy only)
110 *
111 * The effective_cpus of a valid partition root comes solely from its
112 * effective_xcpus and some of the effective_xcpus may be distributed
113 * to sub-partitions below & hence excluded from its effective_cpus.
114 * For a valid partition root, its effective_cpus have no relationship
115 * with cpus_allowed unless its exclusive_cpus isn't set.
116 *
117 * This value will only be set if either exclusive_cpus is set or
118 * when this cpuset becomes a local partition root.
119 */
120 cpumask_var_t effective_xcpus;
121
122 /*
123 * Exclusive CPUs as requested by the user (default hierarchy only)
124 *
125 * Its value is independent of cpus_allowed and designates the set of
126 * CPUs that can be granted to the current cpuset or its children when
127 * it becomes a valid partition root. The effective set of exclusive
128 * CPUs granted (effective_xcpus) depends on whether those exclusive
129 * CPUs are passed down by its ancestors and not yet taken up by
130 * another sibling partition root along the way.
131 *
132 * If its value isn't set, it defaults to cpus_allowed.
133 */
134 cpumask_var_t exclusive_cpus;
135
136 /*
137 * This is old Memory Nodes tasks took on.
138 *
139 * - top_cpuset.old_mems_allowed is initialized to mems_allowed.
140 * - A new cpuset's old_mems_allowed is initialized when some
141 * task is moved into it.
142 * - old_mems_allowed is used in cpuset_migrate_mm() when we change
143 * cpuset.mems_allowed and have tasks' nodemask updated, and
144 * then old_mems_allowed is updated to mems_allowed.
145 */
146 nodemask_t old_mems_allowed;
147
148 /*
149 * For linking impacted cpusets during an attach operation.
150 */
151 struct llist_node attach_node;
152
153 /* partition root state */
154 int partition_root_state;
155
156 /*
157 * Whether cpuset is a remote partition.
158 * It used to be a list anchoring all remote partitions — we can switch back
159 * to a list if we need to iterate over the remote partitions.
160 */
161 bool remote_partition;
162
163 /*
164 * number of SCHED_DEADLINE tasks attached to this cpuset, so that we
165 * know when to rebuild associated root domain bandwidth information.
166 */
167 atomic_t nr_deadline_tasks;
168 int nr_migrate_dl_tasks;
169 /* DL bandwidth that needs destination reservation for this attach. */
170 u64 sum_migrate_dl_bw;
171 /*
172 * CPU used for temporary DL bandwidth allocation during attach;
173 * -1 if no DL bandwidth was allocated in the current attach.
174 */
175 int dl_bw_cpu;
176
177 /* Invalid partition error code, not lock protected */
178 enum prs_errcode prs_err;
179
180 /* Handle for cpuset.cpus.partition */
181 struct cgroup_file partition_file;
182
183 #ifdef CONFIG_CPUSETS_V1
184 struct fmeter fmeter; /* memory_pressure filter */
185
186 /* for custom sched domain */
187 int relax_domain_level;
188
189 /* Used to merge intersecting subsets for generate_sched_domains */
190 struct uf_node node;
191 #endif
192 };
193
194 extern struct cpuset top_cpuset;
195
css_cs(struct cgroup_subsys_state * css)196 static inline struct cpuset *css_cs(struct cgroup_subsys_state *css)
197 {
198 return css ? container_of(css, struct cpuset, css) : NULL;
199 }
200
201 /* Retrieve the cpuset for a task */
task_cs(struct task_struct * task)202 static inline struct cpuset *task_cs(struct task_struct *task)
203 {
204 return css_cs(task_css(task, cpuset_cgrp_id));
205 }
206
parent_cs(struct cpuset * cs)207 static inline struct cpuset *parent_cs(struct cpuset *cs)
208 {
209 return css_cs(cs->css.parent);
210 }
211
212 /* convenient tests for these bits */
is_cpuset_online(struct cpuset * cs)213 static inline bool is_cpuset_online(struct cpuset *cs)
214 {
215 return css_is_online(&cs->css) && !css_is_dying(&cs->css);
216 }
217
is_cpu_exclusive(const struct cpuset * cs)218 static inline int is_cpu_exclusive(const struct cpuset *cs)
219 {
220 return test_bit(CS_CPU_EXCLUSIVE, &cs->flags);
221 }
222
is_mem_exclusive(const struct cpuset * cs)223 static inline int is_mem_exclusive(const struct cpuset *cs)
224 {
225 return test_bit(CS_MEM_EXCLUSIVE, &cs->flags);
226 }
227
is_mem_hardwall(const struct cpuset * cs)228 static inline int is_mem_hardwall(const struct cpuset *cs)
229 {
230 return test_bit(CS_MEM_HARDWALL, &cs->flags);
231 }
232
is_sched_load_balance(const struct cpuset * cs)233 static inline int is_sched_load_balance(const struct cpuset *cs)
234 {
235 return test_bit(CS_SCHED_LOAD_BALANCE, &cs->flags);
236 }
237
is_memory_migrate(const struct cpuset * cs)238 static inline int is_memory_migrate(const struct cpuset *cs)
239 {
240 return test_bit(CS_MEMORY_MIGRATE, &cs->flags);
241 }
242
is_spread_page(const struct cpuset * cs)243 static inline int is_spread_page(const struct cpuset *cs)
244 {
245 return test_bit(CS_SPREAD_PAGE, &cs->flags);
246 }
247
is_spread_slab(const struct cpuset * cs)248 static inline int is_spread_slab(const struct cpuset *cs)
249 {
250 return test_bit(CS_SPREAD_SLAB, &cs->flags);
251 }
252
253 /*
254 * Helper routine for generate_sched_domains().
255 * Do cpusets a, b have overlapping effective cpus_allowed masks?
256 */
cpusets_overlap(struct cpuset * a,struct cpuset * b)257 static inline int cpusets_overlap(struct cpuset *a, struct cpuset *b)
258 {
259 return cpumask_intersects(a->effective_cpus, b->effective_cpus);
260 }
261
nr_cpusets(void)262 static inline int nr_cpusets(void)
263 {
264 /* jump label reference count + the top-level cpuset */
265 return static_key_count(&cpusets_enabled_key.key) + 1;
266 }
267
cpuset_is_populated(struct cpuset * cs)268 static inline bool cpuset_is_populated(struct cpuset *cs)
269 {
270 lockdep_assert_cpuset_lock_held();
271 return cgroup_is_populated(cs->css.cgroup);
272 }
273
274 /**
275 * cpuset_for_each_child - traverse online children of a cpuset
276 * @child_cs: loop cursor pointing to the current child
277 * @pos_css: used for iteration
278 * @parent_cs: target cpuset to walk children of
279 *
280 * Walk @child_cs through the online children of @parent_cs. Must be used
281 * with RCU read locked.
282 */
283 #define cpuset_for_each_child(child_cs, pos_css, parent_cs) \
284 css_for_each_child((pos_css), &(parent_cs)->css) \
285 if (is_cpuset_online(((child_cs) = css_cs((pos_css)))))
286
287 /**
288 * cpuset_for_each_descendant_pre - pre-order walk of a cpuset's descendants
289 * @des_cs: loop cursor pointing to the current descendant
290 * @pos_css: used for iteration
291 * @root_cs: target cpuset to walk ancestor of
292 *
293 * Walk @des_cs through the online descendants of @root_cs. Must be used
294 * with RCU read locked. The caller may modify @pos_css by calling
295 * css_rightmost_descendant() to skip subtree. @root_cs is included in the
296 * iteration and the first node to be visited.
297 */
298 #define cpuset_for_each_descendant_pre(des_cs, pos_css, root_cs) \
299 css_for_each_descendant_pre((pos_css), &(root_cs)->css) \
300 if (is_cpuset_online(((des_cs) = css_cs((pos_css)))))
301
302 void rebuild_sched_domains_locked(void);
303 void cpuset_callback_lock_irq(void);
304 void cpuset_callback_unlock_irq(void);
305 void cpuset_update_tasks_cpumask(struct cpuset *cs, struct cpumask *new_cpus);
306 void cpuset_update_tasks_nodemask(struct cpuset *cs);
307 int cpuset_update_flag(cpuset_flagbits_t bit, struct cpuset *cs, int turning_on);
308 ssize_t cpuset_write_resmask(struct kernfs_open_file *of,
309 char *buf, size_t nbytes, loff_t off);
310 int cpuset_common_seq_show(struct seq_file *sf, void *v);
311 void cpuset_full_lock(void);
312 void cpuset_full_unlock(void);
313
314 /*
315 * cpuset-v1.c
316 */
317 #ifdef CONFIG_CPUSETS_V1
318 extern struct cftype cpuset1_files[];
319 void cpuset1_update_task_spread_flags(struct cpuset *cs,
320 struct task_struct *tsk);
321 void cpuset1_update_tasks_flags(struct cpuset *cs);
322 void cpuset1_hotplug_update_tasks(struct cpuset *cs,
323 struct cpumask *new_cpus, nodemask_t *new_mems,
324 bool cpus_updated, bool mems_updated);
325 int cpuset1_validate_change(struct cpuset *cur, struct cpuset *trial);
326 bool cpuset1_cpus_excl_conflict(struct cpuset *cs1, struct cpuset *cs2);
327 void cpuset1_init(struct cpuset *cs);
328 void cpuset1_online_css(struct cgroup_subsys_state *css);
329 int cpuset1_generate_sched_domains(cpumask_var_t **domains,
330 struct sched_domain_attr **attributes);
331
332 #else
cpuset1_update_task_spread_flags(struct cpuset * cs,struct task_struct * tsk)333 static inline void cpuset1_update_task_spread_flags(struct cpuset *cs,
334 struct task_struct *tsk) {}
cpuset1_update_tasks_flags(struct cpuset * cs)335 static inline void cpuset1_update_tasks_flags(struct cpuset *cs) {}
cpuset1_hotplug_update_tasks(struct cpuset * cs,struct cpumask * new_cpus,nodemask_t * new_mems,bool cpus_updated,bool mems_updated)336 static inline void cpuset1_hotplug_update_tasks(struct cpuset *cs,
337 struct cpumask *new_cpus, nodemask_t *new_mems,
338 bool cpus_updated, bool mems_updated) {}
cpuset1_validate_change(struct cpuset * cur,struct cpuset * trial)339 static inline int cpuset1_validate_change(struct cpuset *cur,
340 struct cpuset *trial) { return 0; }
cpuset1_cpus_excl_conflict(struct cpuset * cs1,struct cpuset * cs2)341 static inline bool cpuset1_cpus_excl_conflict(struct cpuset *cs1,
342 struct cpuset *cs2) { return false; }
cpuset1_init(struct cpuset * cs)343 static inline void cpuset1_init(struct cpuset *cs) {}
cpuset1_online_css(struct cgroup_subsys_state * css)344 static inline void cpuset1_online_css(struct cgroup_subsys_state *css) {}
cpuset1_generate_sched_domains(cpumask_var_t ** domains,struct sched_domain_attr ** attributes)345 static inline int cpuset1_generate_sched_domains(cpumask_var_t **domains,
346 struct sched_domain_attr **attributes) { return 0; };
347
348 #endif /* CONFIG_CPUSETS_V1 */
349
350 #endif /* __CPUSET_INTERNAL_H */
351