1 /* SPDX-License-Identifier: GPL-2.0 */
2 /*
3 * Copyright (c) 2022 Meta Platforms, Inc. and affiliates.
4 * Copyright (c) 2022 Tejun Heo <tj@kernel.org>
5 * Copyright (c) 2022 David Vernet <dvernet@meta.com>
6 */
7 #ifndef __SCX_COMMON_BPF_H
8 #define __SCX_COMMON_BPF_H
9
10 /*
11 * The generated kfunc prototypes in vmlinux.h are missing address space
12 * attributes which cause build failures. For now, suppress the generated
13 * prototypes. See https://github.com/sched-ext/scx/issues/1111.
14 */
15 #define BPF_NO_KFUNC_PROTOTYPES
16
17 #ifdef LSP
18 #define __bpf__
19 #include "../vmlinux.h"
20 #else
21 #include "vmlinux.h"
22 #endif
23
24 #include <bpf/bpf_helpers.h>
25 #include <bpf/bpf_tracing.h>
26 #include <asm-generic/errno.h>
27 #include "user_exit_info.bpf.h"
28 #include "enum_defs.autogen.h"
29 #include "bpf_arena_common.bpf.h"
30
31 #define PF_IDLE 0x00000002 /* I am an IDLE thread */
32 #define PF_IO_WORKER 0x00000010 /* Task is an IO worker */
33 #define PF_WQ_WORKER 0x00000020 /* I'm a workqueue worker */
34 #define PF_KCOMPACTD 0x00010000 /* I am kcompactd */
35 #define PF_KSWAPD 0x00020000 /* I am kswapd */
36 #define PF_KTHREAD 0x00200000 /* I am a kernel thread */
37 #define PF_EXITING 0x00000004
38 #define CLOCK_MONOTONIC 1
39
40 #ifndef NR_CPUS
41 #define NR_CPUS 1024
42 #endif
43
44 #ifndef NUMA_NO_NODE
45 #define NUMA_NO_NODE (-1)
46 #endif
47
48 extern int LINUX_KERNEL_VERSION __kconfig;
49 extern const char CONFIG_CC_VERSION_TEXT[64] __kconfig __weak;
50 extern const char CONFIG_LOCALVERSION[64] __kconfig __weak;
51
52 /*
53 * Earlier versions of clang/pahole lost upper 32bits in 64bit enums which can
54 * lead to really confusing misbehaviors. Let's trigger a build failure.
55 */
___vmlinux_h_sanity_check___(void)56 static inline void ___vmlinux_h_sanity_check___(void)
57 {
58 _Static_assert(SCX_DSQ_FLAG_BUILTIN,
59 "bpftool generated vmlinux.h is missing high bits for 64bit enums, upgrade clang and pahole");
60 }
61
62 s32 scx_bpf_create_dsq(u64 dsq_id, s32 node) __ksym;
63 s32 scx_bpf_select_cpu_dfl(struct task_struct *p, s32 prev_cpu, u64 wake_flags, bool *is_idle) __ksym;
64 s32 __scx_bpf_select_cpu_and(struct task_struct *p, const struct cpumask *cpus_allowed,
65 struct scx_bpf_select_cpu_and_args *args) __ksym __weak;
66 bool __scx_bpf_dsq_insert_vtime(struct task_struct *p, struct scx_bpf_dsq_insert_vtime_args *args) __ksym __weak;
67 u32 scx_bpf_dispatch_nr_slots(void) __ksym;
68 void scx_bpf_dispatch_cancel(void) __ksym;
69 void scx_bpf_kick_cpu(s32 cpu, u64 flags) __ksym;
70 s32 scx_bpf_dsq_nr_queued(u64 dsq_id) __ksym;
71 void scx_bpf_destroy_dsq(u64 dsq_id) __ksym;
72 struct task_struct *scx_bpf_dsq_peek(u64 dsq_id) __ksym __weak;
73 int bpf_iter_scx_dsq_new(struct bpf_iter_scx_dsq *it, u64 dsq_id, u64 flags) __ksym __weak;
74 struct task_struct *bpf_iter_scx_dsq_next(struct bpf_iter_scx_dsq *it) __ksym __weak;
75 void bpf_iter_scx_dsq_destroy(struct bpf_iter_scx_dsq *it) __ksym __weak;
76 void scx_bpf_exit_bstr(s64 exit_code, char *fmt, unsigned long long *data, u32 data__sz) __ksym __weak;
77 void scx_bpf_error_bstr(char *fmt, unsigned long long *data, u32 data_len) __ksym;
78 void scx_bpf_dump_bstr(char *fmt, unsigned long long *data, u32 data_len) __ksym __weak;
79 u32 scx_bpf_cpuperf_cap(s32 cpu) __ksym __weak;
80 u32 scx_bpf_cpuperf_cur(s32 cpu) __ksym __weak;
81 void scx_bpf_cpuperf_set(s32 cpu, u32 perf) __ksym __weak;
82 u32 scx_bpf_nr_node_ids(void) __ksym __weak;
83 u32 scx_bpf_nr_cpu_ids(void) __ksym __weak;
84 int scx_bpf_cpu_node(s32 cpu) __ksym __weak;
85 const struct cpumask *scx_bpf_get_possible_cpumask(void) __ksym __weak;
86 const struct cpumask *scx_bpf_get_online_cpumask(void) __ksym __weak;
87 void scx_bpf_put_cpumask(const struct cpumask *cpumask) __ksym __weak;
88 const struct cpumask *scx_bpf_get_idle_cpumask_node(int node) __ksym __weak;
89 const struct cpumask *scx_bpf_get_idle_cpumask(void) __ksym;
90 const struct cpumask *scx_bpf_get_idle_smtmask_node(int node) __ksym __weak;
91 const struct cpumask *scx_bpf_get_idle_smtmask(void) __ksym;
92 void scx_bpf_put_idle_cpumask(const struct cpumask *cpumask) __ksym;
93 bool scx_bpf_test_and_clear_cpu_idle(s32 cpu) __ksym;
94 s32 scx_bpf_pick_idle_cpu_node(const cpumask_t *cpus_allowed, int node, u64 flags) __ksym __weak;
95 s32 scx_bpf_pick_idle_cpu(const cpumask_t *cpus_allowed, u64 flags) __ksym;
96 s32 scx_bpf_pick_any_cpu_node(const cpumask_t *cpus_allowed, int node, u64 flags) __ksym __weak;
97 s32 scx_bpf_pick_any_cpu(const cpumask_t *cpus_allowed, u64 flags) __ksym;
98 bool scx_bpf_task_running(const struct task_struct *p) __ksym;
99 s32 scx_bpf_task_cpu(const struct task_struct *p) __ksym;
100 struct rq *scx_bpf_locked_rq(void) __ksym;
101 struct task_struct *scx_bpf_cpu_curr(s32 cpu) __ksym __weak;
102 struct task_struct *scx_bpf_tid_to_task(u64 tid) __ksym __weak;
103 u64 scx_bpf_now(void) __ksym __weak;
104 void scx_bpf_events(struct scx_event_stats *events, size_t events__sz) __ksym __weak;
105 s32 scx_bpf_cpu_to_cid(s32 cpu) __ksym __weak;
106 s32 scx_bpf_cid_to_cpu(s32 cid) __ksym __weak;
107 void scx_bpf_cid_topo(s32 cid, struct scx_cid_topo *out) __ksym __weak;
108 void scx_bpf_kick_cid(s32 cid, u64 flags) __ksym __weak;
109 s32 scx_bpf_task_cid(const struct task_struct *p) __ksym __weak;
110 s32 scx_bpf_this_cid(void) __ksym __weak;
111 struct task_struct *scx_bpf_cid_curr(s32 cid) __ksym __weak;
112 u32 scx_bpf_nr_cids(void) __ksym __weak;
113 u32 scx_bpf_nr_online_cids(void) __ksym __weak;
114 u32 scx_bpf_cidperf_cap(s32 cid) __ksym __weak;
115 u32 scx_bpf_cidperf_cur(s32 cid) __ksym __weak;
116 s32 scx_bpf_cidperf_set(s32 cid, u32 perf) __ksym __weak;
117
118 /* sub-scheduler cap control, scx_bpf_sub_caps() cgroup_id 0 == self */
119 s32 scx_bpf_sub_grant(u64 cgroup_id, u64 caps, const struct scx_cmask __arena *cmask__arena, struct scx_cmask __arena *denied_out__arena__nullable) __ksym __weak;
120 void scx_bpf_sub_revoke(u64 cgroup_id, u64 caps, const struct scx_cmask __arena *cmask__arena) __ksym __weak;
121 s32 scx_bpf_sub_caps(u64 cgroup_id, u64 caps, struct scx_cmask __arena *out__arena) __ksym __weak;
122 s32 scx_bpf_sub_kill_bstr(u64 cgroup_id, char *fmt, unsigned long long *data, u32 data__sz) __ksym __weak;
123
124 /*
125 * Use the following as @it__iter when calling scx_bpf_dsq_move[_vtime]() from
126 * within bpf_for_each() loops.
127 */
128 #define BPF_FOR_EACH_ITER (&___it)
129
130 #define scx_read_event(e, name) \
131 (bpf_core_field_exists((e)->name) ? (e)->name : 0)
132
133 static inline __attribute__((format(printf, 1, 2)))
___scx_bpf_bstr_format_checker(const char * fmt,...)134 void ___scx_bpf_bstr_format_checker(const char *fmt, ...) {}
135
136 #define SCX_STRINGIFY(x) #x
137 #define SCX_TOSTRING(x) SCX_STRINGIFY(x)
138
139 /*
140 * Helper macro for initializing the fmt and variadic argument inputs to both
141 * bstr exit kfuncs. Callers to this function should use ___fmt and ___param to
142 * refer to the initialized list of inputs to the bstr kfunc.
143 */
144 #define scx_bpf_bstr_preamble(fmt, args...) \
145 static char ___fmt[] = fmt; \
146 /* \
147 * Note that __param[] must have at least one \
148 * element to keep the verifier happy. \
149 */ \
150 unsigned long long ___param[___bpf_narg(args) ?: 1] = {}; \
151 \
152 _Pragma("GCC diagnostic push") \
153 _Pragma("GCC diagnostic ignored \"-Wint-conversion\"") \
154 ___bpf_fill(___param, args); \
155 _Pragma("GCC diagnostic pop")
156
157 /*
158 * scx_bpf_exit() wraps the scx_bpf_exit_bstr() kfunc with variadic arguments
159 * instead of an array of u64. Using this macro will cause the scheduler to
160 * exit cleanly with the specified exit code being passed to user space.
161 */
162 #define scx_bpf_exit(code, fmt, args...) \
163 ({ \
164 scx_bpf_bstr_preamble(fmt, args) \
165 scx_bpf_exit_bstr(code, ___fmt, ___param, sizeof(___param)); \
166 ___scx_bpf_bstr_format_checker(fmt, ##args); \
167 })
168
169 /*
170 * scx_bpf_sub_kill() wraps the scx_bpf_sub_kill_bstr() kfunc with variadic
171 * arguments instead of an array of u64. It kills the direct child sub-scheduler
172 * @cgid, passing the formatted reason to its user space, and evaluates to the
173 * kfunc's return value. On a kernel without sub-scheduler support the kfunc is
174 * absent and it returns -EOPNOTSUPP.
175 */
176 #define scx_bpf_sub_kill(cgid, fmt, args...) \
177 ({ \
178 scx_bpf_bstr_preamble(fmt, args) \
179 ___scx_bpf_bstr_format_checker(fmt, ##args); \
180 bpf_ksym_exists(scx_bpf_sub_kill_bstr) ? \
181 scx_bpf_sub_kill_bstr((cgid), ___fmt, ___param, \
182 sizeof(___param)) : -EOPNOTSUPP; \
183 })
184
185 /*
186 * scx_bpf_error() wraps the scx_bpf_error_bstr() kfunc with variadic arguments
187 * instead of an array of u64. Invoking this macro will cause the scheduler to
188 * exit in an erroneous state, with diagnostic information being passed to the
189 * user. It appends the file and line number to aid debugging.
190 */
191 #define scx_bpf_error(fmt, args...) \
192 ({ \
193 scx_bpf_bstr_preamble( \
194 __FILE__ ":" SCX_TOSTRING(__LINE__) ": " fmt, ##args) \
195 scx_bpf_error_bstr(___fmt, ___param, sizeof(___param)); \
196 ___scx_bpf_bstr_format_checker( \
197 __FILE__ ":" SCX_TOSTRING(__LINE__) ": " fmt, ##args); \
198 })
199
200 /*
201 * scx_bpf_dump() wraps the scx_bpf_dump_bstr() kfunc with variadic arguments
202 * instead of an array of u64. To be used from ops.dump() and friends.
203 */
204 #define scx_bpf_dump(fmt, args...) \
205 ({ \
206 scx_bpf_bstr_preamble(fmt, args) \
207 scx_bpf_dump_bstr(___fmt, ___param, sizeof(___param)); \
208 ___scx_bpf_bstr_format_checker(fmt, ##args); \
209 })
210
211 /*
212 * scx_bpf_dump_header() is a wrapper around scx_bpf_dump that adds a header
213 * of system information for debugging.
214 */
215 #define scx_bpf_dump_header() \
216 ({ \
217 scx_bpf_dump("kernel: %d.%d.%d %s\ncc: %s\n", \
218 LINUX_KERNEL_VERSION >> 16, \
219 LINUX_KERNEL_VERSION >> 8 & 0xFF, \
220 LINUX_KERNEL_VERSION & 0xFF, \
221 CONFIG_LOCALVERSION, \
222 CONFIG_CC_VERSION_TEXT); \
223 })
224
225 #define BPF_STRUCT_OPS(name, args...) \
226 SEC("struct_ops/"#name) \
227 BPF_PROG(name, ##args)
228
229 #define BPF_STRUCT_OPS_SLEEPABLE(name, args...) \
230 SEC("struct_ops.s/"#name) \
231 BPF_PROG(name, ##args)
232
233 /**
234 * RESIZABLE_ARRAY - Generates annotations for an array that may be resized
235 * @elfsec: the data section of the BPF program in which to place the array
236 * @arr: the name of the array
237 *
238 * libbpf has an API for setting map value sizes. Since data sections (i.e.
239 * bss, data, rodata) themselves are maps, a data section can be resized. If
240 * a data section has an array as its last element, the BTF info for that
241 * array will be adjusted so that length of the array is extended to meet the
242 * new length of the data section. This macro annotates an array to have an
243 * element count of one with the assumption that this array can be resized
244 * within the userspace program. It also annotates the section specifier so
245 * this array exists in a custom sub data section which can be resized
246 * independently.
247 *
248 * See RESIZE_ARRAY() for the userspace convenience macro for resizing an
249 * array declared with RESIZABLE_ARRAY().
250 */
251 #define RESIZABLE_ARRAY(elfsec, arr) arr[1] SEC("."#elfsec"."#arr)
252
253 /**
254 * MEMBER_VPTR - Obtain the verified pointer to a struct or array member
255 * @base: struct or array to index
256 * @member: dereferenced member (e.g. .field, [idx0][idx1], .field[idx0] ...)
257 *
258 * The verifier often gets confused by the instruction sequence the compiler
259 * generates for indexing struct fields or arrays. This macro forces the
260 * compiler to generate a code sequence which first calculates the byte offset,
261 * checks it against the struct or array size and add that byte offset to
262 * generate the pointer to the member to help the verifier.
263 *
264 * Ideally, we want to abort if the calculated offset is out-of-bounds. However,
265 * BPF currently doesn't support abort, so evaluate to %NULL instead. The caller
266 * must check for %NULL and take appropriate action to appease the verifier. To
267 * avoid confusing the verifier, it's best to check for %NULL and dereference
268 * immediately.
269 *
270 * vptr = MEMBER_VPTR(my_array, [i][j]);
271 * if (!vptr)
272 * return error;
273 * *vptr = new_value;
274 *
275 * sizeof(@base) should encompass the memory area to be accessed and thus can't
276 * be a pointer to the area. Use `MEMBER_VPTR(*ptr, .member)` instead of
277 * `MEMBER_VPTR(ptr, ->member)`.
278 */
279 #ifndef MEMBER_VPTR
280 #define MEMBER_VPTR(base, member) (typeof((base) member) *) \
281 ({ \
282 u64 __base = (u64)&(base); \
283 u64 __addr = (u64)&((base) member) - __base; \
284 _Static_assert(sizeof(base) >= sizeof((base) member), \
285 "@base is smaller than @member, is @base a pointer?"); \
286 asm volatile ( \
287 "if %0 <= %[max] goto +2\n" \
288 "%0 = 0\n" \
289 "goto +1\n" \
290 "%0 += %1\n" \
291 : "+r"(__addr) \
292 : "r"(__base), \
293 [max]"i"(sizeof(base) - sizeof((base) member))); \
294 __addr; \
295 })
296 #endif /* MEMBER_VPTR */
297
298 /**
299 * ARRAY_ELEM_PTR - Obtain the verified pointer to an array element
300 * @arr: array to index into
301 * @i: array index
302 * @n: number of elements in array
303 *
304 * Similar to MEMBER_VPTR() but is intended for use with arrays where the
305 * element count needs to be explicit.
306 * It can be used in cases where a global array is defined with an initial
307 * size but is intended to be be resized before loading the BPF program.
308 * Without this version of the macro, MEMBER_VPTR() will use the compile time
309 * size of the array to compute the max, which will result in rejection by
310 * the verifier.
311 */
312 #ifndef ARRAY_ELEM_PTR
313 #define ARRAY_ELEM_PTR(arr, i, n) (typeof(arr[i]) *) \
314 ({ \
315 u64 __base = (u64)arr; \
316 u64 __addr = (u64)&(arr[i]) - __base; \
317 asm volatile ( \
318 "if %0 <= %[max] goto +2\n" \
319 "%0 = 0\n" \
320 "goto +1\n" \
321 "%0 += %1\n" \
322 : "+r"(__addr) \
323 : "r"(__base), \
324 [max]"r"(sizeof(arr[0]) * ((n) - 1))); \
325 __addr; \
326 })
327 #endif /* ARRAY_ELEM_PTR */
328
329 /**
330 * __sink - Hide @expr's value from the compiler and BPF verifier
331 * @expr: The expression whose value should be opacified
332 *
333 * No-op at runtime. The empty inline assembly with a read-write constraint
334 * ("+g") has two effects at compile/verify time:
335 *
336 * 1. Compiler: treats @expr as both read and written, preventing dead-code
337 * elimination and keeping @expr (and any side effects that produced it)
338 * alive.
339 *
340 * 2. BPF verifier: forgets the precise value/range of @expr ("makes it
341 * imprecise"). The verifier normally tracks exact ranges for every register
342 * and stack slot. While useful, precision means each distinct value creates a
343 * separate verifier state. Inside loops this leads to state explosion - each
344 * iteration carries different precise values so states never merge and the
345 * verifier explores every iteration individually.
346 *
347 * Example - preventing loop state explosion::
348 *
349 * u32 nr_intersects = 0, nr_covered = 0;
350 * __sink(nr_intersects);
351 * __sink(nr_covered);
352 * bpf_for(i, 0, nr_nodes) {
353 * if (intersects(cpumask, node_mask[i]))
354 * nr_intersects++;
355 * if (covers(cpumask, node_mask[i]))
356 * nr_covered++;
357 * }
358 *
359 * Without __sink(), the verifier tracks every possible (nr_intersects,
360 * nr_covered) pair across iterations, causing "BPF program is too large". With
361 * __sink(), the values become unknown scalars so all iterations collapse into
362 * one reusable state.
363 *
364 * Example - keeping a reference alive::
365 *
366 * struct task_struct *t = bpf_task_acquire(task);
367 * __sink(t);
368 *
369 * Follows the convention from BPF selftests (bpf_misc.h).
370 */
371 #define __sink(expr) asm volatile ("" : "+g"(expr))
372
373 /*
374 * BPF declarations and helpers
375 */
376
377 /* list and rbtree */
378 #define __contains(name, node) __attribute__((btf_decl_tag("contains:" #name ":" #node)))
379 #define private(name) SEC(".data." #name) __hidden __attribute__((aligned(8)))
380
381 void *bpf_obj_new_impl(__u64 local_type_id, void *meta) __ksym;
382 void bpf_obj_drop_impl(void *kptr, void *meta) __ksym;
383
384 #define bpf_obj_new(type) ((type *)bpf_obj_new_impl(bpf_core_type_id_local(type), NULL))
385 #define bpf_obj_drop(kptr) bpf_obj_drop_impl(kptr, NULL)
386
387 int bpf_list_push_front_impl(struct bpf_list_head *head,
388 struct bpf_list_node *node,
389 void *meta, __u64 off) __ksym;
390 #define bpf_list_push_front(head, node) bpf_list_push_front_impl(head, node, NULL, 0)
391
392 int bpf_list_push_back_impl(struct bpf_list_head *head,
393 struct bpf_list_node *node,
394 void *meta, __u64 off) __ksym;
395 #define bpf_list_push_back(head, node) bpf_list_push_back_impl(head, node, NULL, 0)
396
397 struct bpf_list_node *bpf_list_pop_front(struct bpf_list_head *head) __ksym;
398 struct bpf_list_node *bpf_list_pop_back(struct bpf_list_head *head) __ksym;
399 struct bpf_rb_node *bpf_rbtree_remove(struct bpf_rb_root *root,
400 struct bpf_rb_node *node) __ksym;
401 int bpf_rbtree_add_impl(struct bpf_rb_root *root, struct bpf_rb_node *node,
402 bool (less)(struct bpf_rb_node *a, const struct bpf_rb_node *b),
403 void *meta, __u64 off) __ksym;
404 #define bpf_rbtree_add(head, node, less) bpf_rbtree_add_impl(head, node, less, NULL, 0)
405
406 struct bpf_rb_node *bpf_rbtree_first(struct bpf_rb_root *root) __ksym;
407
408 void *bpf_refcount_acquire_impl(void *kptr, void *meta) __ksym;
409 #define bpf_refcount_acquire(kptr) bpf_refcount_acquire_impl(kptr, NULL)
410
411 /* task */
412 struct task_struct *bpf_task_from_pid(s32 pid) __ksym;
413 struct task_struct *bpf_task_acquire(struct task_struct *p) __ksym;
414 void bpf_task_release(struct task_struct *p) __ksym;
415
416 /* cgroup */
417 struct cgroup *bpf_cgroup_ancestor(struct cgroup *cgrp, int level) __ksym;
418 struct cgroup *bpf_cgroup_acquire(struct cgroup *cgrp) __ksym;
419 void bpf_cgroup_release(struct cgroup *cgrp) __ksym;
420 struct cgroup *bpf_cgroup_from_id(u64 cgid) __ksym;
421
422 /* css iteration */
423 struct bpf_iter_css;
424 struct cgroup_subsys_state;
425 extern int bpf_iter_css_new(struct bpf_iter_css *it,
426 struct cgroup_subsys_state *start,
427 unsigned int flags) __weak __ksym;
428 extern struct cgroup_subsys_state *
429 bpf_iter_css_next(struct bpf_iter_css *it) __weak __ksym;
430 extern void bpf_iter_css_destroy(struct bpf_iter_css *it) __weak __ksym;
431
432 /* cpumask */
433 struct bpf_cpumask *bpf_cpumask_create(void) __ksym;
434 struct bpf_cpumask *bpf_cpumask_acquire(struct bpf_cpumask *cpumask) __ksym;
435 void bpf_cpumask_release(struct bpf_cpumask *cpumask) __ksym;
436 u32 bpf_cpumask_first(const struct cpumask *cpumask) __ksym;
437 u32 bpf_cpumask_first_zero(const struct cpumask *cpumask) __ksym;
438 void bpf_cpumask_set_cpu(u32 cpu, struct bpf_cpumask *cpumask) __ksym;
439 void bpf_cpumask_clear_cpu(u32 cpu, struct bpf_cpumask *cpumask) __ksym;
440 bool bpf_cpumask_test_cpu(u32 cpu, const struct cpumask *cpumask) __ksym;
441 bool bpf_cpumask_test_and_set_cpu(u32 cpu, struct bpf_cpumask *cpumask) __ksym;
442 bool bpf_cpumask_test_and_clear_cpu(u32 cpu, struct bpf_cpumask *cpumask) __ksym;
443 void bpf_cpumask_setall(struct bpf_cpumask *cpumask) __ksym;
444 void bpf_cpumask_clear(struct bpf_cpumask *cpumask) __ksym;
445 bool bpf_cpumask_and(struct bpf_cpumask *dst, const struct cpumask *src1,
446 const struct cpumask *src2) __ksym;
447 void bpf_cpumask_or(struct bpf_cpumask *dst, const struct cpumask *src1,
448 const struct cpumask *src2) __ksym;
449 void bpf_cpumask_xor(struct bpf_cpumask *dst, const struct cpumask *src1,
450 const struct cpumask *src2) __ksym;
451 bool bpf_cpumask_equal(const struct cpumask *src1, const struct cpumask *src2) __ksym;
452 bool bpf_cpumask_intersects(const struct cpumask *src1, const struct cpumask *src2) __ksym;
453 bool bpf_cpumask_subset(const struct cpumask *src1, const struct cpumask *src2) __ksym;
454 bool bpf_cpumask_empty(const struct cpumask *cpumask) __ksym;
455 bool bpf_cpumask_full(const struct cpumask *cpumask) __ksym;
456 void bpf_cpumask_copy(struct bpf_cpumask *dst, const struct cpumask *src) __ksym;
457 u32 bpf_cpumask_any_distribute(const struct cpumask *cpumask) __ksym;
458 u32 bpf_cpumask_any_and_distribute(const struct cpumask *src1,
459 const struct cpumask *src2) __ksym;
460 u32 bpf_cpumask_weight(const struct cpumask *cpumask) __ksym;
461
462 int bpf_iter_bits_new(struct bpf_iter_bits *it, const u64 *unsafe_ptr__ign, u32 nr_words) __ksym;
463 int *bpf_iter_bits_next(struct bpf_iter_bits *it) __ksym;
464 void bpf_iter_bits_destroy(struct bpf_iter_bits *it) __ksym;
465
466 #define def_iter_struct(name) \
467 struct bpf_iter_##name { \
468 struct bpf_iter_bits it; \
469 const struct cpumask *bitmap; \
470 };
471
472 #define def_iter_new(name) \
473 static inline int bpf_iter_##name##_new( \
474 struct bpf_iter_##name *it, const u64 *unsafe_ptr__ign, u32 nr_words) \
475 { \
476 it->bitmap = scx_bpf_get_##name##_cpumask(); \
477 return bpf_iter_bits_new(&it->it, (const u64 *)it->bitmap, \
478 sizeof(struct cpumask) / 8); \
479 }
480
481 #define def_iter_next(name) \
482 static inline int *bpf_iter_##name##_next(struct bpf_iter_##name *it) { \
483 return bpf_iter_bits_next(&it->it); \
484 }
485
486 #define def_iter_destroy(name) \
487 static inline void bpf_iter_##name##_destroy(struct bpf_iter_##name *it) { \
488 scx_bpf_put_cpumask(it->bitmap); \
489 bpf_iter_bits_destroy(&it->it); \
490 }
491 #define def_for_each_cpu(cpu, name) for_each_##name##_cpu(cpu)
492
493 /// Provides iterator for possible and online cpus.
494 ///
495 /// # Example
496 ///
497 /// ```
498 /// static inline void example_use() {
499 /// int *cpu;
500 ///
501 /// for_each_possible_cpu(cpu){
502 /// bpf_printk("CPU %d is possible", *cpu);
503 /// }
504 ///
505 /// for_each_online_cpu(cpu){
506 /// bpf_printk("CPU %d is online", *cpu);
507 /// }
508 /// }
509 /// ```
510 def_iter_struct(possible);
511 def_iter_new(possible);
512 def_iter_next(possible);
513 def_iter_destroy(possible);
514 #define for_each_possible_cpu(cpu) bpf_for_each(possible, cpu, NULL, 0)
515
516 def_iter_struct(online);
517 def_iter_new(online);
518 def_iter_next(online);
519 def_iter_destroy(online);
520 #define for_each_online_cpu(cpu) bpf_for_each(online, cpu, NULL, 0)
521
522 /*
523 * Access a cpumask in read-only mode (typically to check bits).
524 */
cast_mask(struct bpf_cpumask * mask)525 static __always_inline const struct cpumask *cast_mask(struct bpf_cpumask *mask)
526 {
527 return (const struct cpumask *)mask;
528 }
529
530 /*
531 * Return true if task @p cannot migrate to a different CPU, false
532 * otherwise.
533 */
is_migration_disabled(const struct task_struct * p)534 static inline bool is_migration_disabled(const struct task_struct *p)
535 {
536 /*
537 * Testing p->migration_disabled in a BPF code is tricky because the
538 * migration is _always_ disabled while running the BPF code.
539 * The prolog (__bpf_prog_enter) and epilog (__bpf_prog_exit) for BPF
540 * code execution disable and re-enable the migration of the current
541 * task, respectively. So, the _current_ task of the sched_ext ops is
542 * always migration-disabled. Moreover, p->migration_disabled could be
543 * two or greater when a sched_ext ops BPF code (e.g., ops.tick) is
544 * executed in the middle of the other BPF code execution.
545 *
546 * Therefore, we should decide that the _current_ task is
547 * migration-disabled only when its migration_disabled count is greater
548 * than one. In other words, when p->migration_disabled == 1, there is
549 * an ambiguity, so we should check if @p is the current task or not.
550 */
551 if (bpf_core_field_exists(p->migration_disabled)) {
552 if (p->migration_disabled == 1)
553 return bpf_get_current_task_btf() != p;
554 else
555 return p->migration_disabled;
556 }
557 return false;
558 }
559
560 /* rcu */
561 void bpf_rcu_read_lock(void) __ksym;
562 void bpf_rcu_read_unlock(void) __ksym;
563
564 /* resilient qspinlock */
565 int bpf_res_spin_lock(struct bpf_res_spin_lock *lock) __ksym __weak;
566 void bpf_res_spin_unlock(struct bpf_res_spin_lock *lock) __ksym __weak;
567
568 /*
569 * Time helpers, most of which are from jiffies.h.
570 */
571
572 /**
573 * time_delta - Calculate the delta between new and old time stamp
574 * @after: first comparable as u64
575 * @before: second comparable as u64
576 *
577 * Return: the time difference, which is >= 0
578 */
time_delta(u64 after,u64 before)579 static inline s64 time_delta(u64 after, u64 before)
580 {
581 return (s64)(after - before) > 0 ? (s64)(after - before) : 0;
582 }
583
584 /**
585 * time_after - returns true if the time a is after time b.
586 * @a: first comparable as u64
587 * @b: second comparable as u64
588 *
589 * Do this with "<0" and ">=0" to only test the sign of the result. A
590 * good compiler would generate better code (and a really good compiler
591 * wouldn't care). Gcc is currently neither.
592 *
593 * Return: %true is time a is after time b, otherwise %false.
594 */
time_after(u64 a,u64 b)595 static inline bool time_after(u64 a, u64 b)
596 {
597 return (s64)(b - a) < 0;
598 }
599
600 /**
601 * time_before - returns true if the time a is before time b.
602 * @a: first comparable as u64
603 * @b: second comparable as u64
604 *
605 * Return: %true is time a is before time b, otherwise %false.
606 */
time_before(u64 a,u64 b)607 static inline bool time_before(u64 a, u64 b)
608 {
609 return time_after(b, a);
610 }
611
612 /**
613 * time_after_eq - returns true if the time a is after or the same as time b.
614 * @a: first comparable as u64
615 * @b: second comparable as u64
616 *
617 * Return: %true is time a is after or the same as time b, otherwise %false.
618 */
time_after_eq(u64 a,u64 b)619 static inline bool time_after_eq(u64 a, u64 b)
620 {
621 return (s64)(a - b) >= 0;
622 }
623
624 /**
625 * time_before_eq - returns true if the time a is before or the same as time b.
626 * @a: first comparable as u64
627 * @b: second comparable as u64
628 *
629 * Return: %true is time a is before or the same as time b, otherwise %false.
630 */
time_before_eq(u64 a,u64 b)631 static inline bool time_before_eq(u64 a, u64 b)
632 {
633 return time_after_eq(b, a);
634 }
635
636 /**
637 * time_in_range - Calculate whether a is in the range of [b, c].
638 * @a: time to test
639 * @b: beginning of the range
640 * @c: end of the range
641 *
642 * Return: %true is time a is in the range [b, c], otherwise %false.
643 */
time_in_range(u64 a,u64 b,u64 c)644 static inline bool time_in_range(u64 a, u64 b, u64 c)
645 {
646 return time_after_eq(a, b) && time_before_eq(a, c);
647 }
648
649 /**
650 * time_in_range_open - Calculate whether a is in the range of [b, c).
651 * @a: time to test
652 * @b: beginning of the range
653 * @c: end of the range
654 *
655 * Return: %true is time a is in the range [b, c), otherwise %false.
656 */
time_in_range_open(u64 a,u64 b,u64 c)657 static inline bool time_in_range_open(u64 a, u64 b, u64 c)
658 {
659 return time_after_eq(a, b) && time_before(a, c);
660 }
661
662
663 /*
664 * Other helpers
665 */
666
667 /* useful compiler attributes */
668 #ifndef likely
669 #define likely(x) __builtin_expect(!!(x), 1)
670 #endif
671 #ifndef unlikely
672 #define unlikely(x) __builtin_expect(!!(x), 0)
673 #endif
674 #ifndef __maybe_unused
675 #define __maybe_unused __attribute__((__unused__))
676 #endif
677
678 /*
679 * READ/WRITE_ONCE() are from kernel (include/asm-generic/rwonce.h). They
680 * prevent compiler from caching, redoing or reordering reads or writes.
681 */
682 typedef __u8 __attribute__((__may_alias__)) __u8_alias_t;
683 typedef __u16 __attribute__((__may_alias__)) __u16_alias_t;
684 typedef __u32 __attribute__((__may_alias__)) __u32_alias_t;
685 typedef __u64 __attribute__((__may_alias__)) __u64_alias_t;
686
__read_once_size(const volatile void * p,void * res,int size)687 static __always_inline void __read_once_size(const volatile void *p, void *res, int size)
688 {
689 switch (size) {
690 case 1: *(__u8_alias_t *) res = *(volatile __u8_alias_t *) p; break;
691 case 2: *(__u16_alias_t *) res = *(volatile __u16_alias_t *) p; break;
692 case 4: *(__u32_alias_t *) res = *(volatile __u32_alias_t *) p; break;
693 case 8: *(__u64_alias_t *) res = *(volatile __u64_alias_t *) p; break;
694 default:
695 barrier();
696 __builtin_memcpy((void *)res, (const void *)p, size);
697 barrier();
698 }
699 }
700
__write_once_size(volatile void * p,void * res,int size)701 static __always_inline void __write_once_size(volatile void *p, void *res, int size)
702 {
703 switch (size) {
704 case 1: *(volatile __u8_alias_t *) p = *(__u8_alias_t *) res; break;
705 case 2: *(volatile __u16_alias_t *) p = *(__u16_alias_t *) res; break;
706 case 4: *(volatile __u32_alias_t *) p = *(__u32_alias_t *) res; break;
707 case 8: *(volatile __u64_alias_t *) p = *(__u64_alias_t *) res; break;
708 default:
709 barrier();
710 __builtin_memcpy((void *)p, (const void *)res, size);
711 barrier();
712 }
713 }
714
715 /*
716 * __unqual_typeof(x) - Declare an unqualified scalar type, leaving
717 * non-scalar types unchanged,
718 *
719 * Prefer C11 _Generic for better compile-times and simpler code. Note: 'char'
720 * is not type-compatible with 'signed char', and we define a separate case.
721 *
722 * This is copied verbatim from kernel's include/linux/compiler_types.h, but
723 * with default expression (for pointers) changed from (x) to (typeof(x)0).
724 *
725 * This is because LLVM has a bug where for lvalue (x), it does not get rid of
726 * an extra address_space qualifier, but does in case of rvalue (typeof(x)0).
727 * Hence, for pointers, we need to create an rvalue expression to get the
728 * desired type. See https://github.com/llvm/llvm-project/issues/53400.
729 */
730 #define __scalar_type_to_expr_cases(type) \
731 unsigned type : (unsigned type)0, signed type : (signed type)0
732
733 #define __unqual_typeof(x) \
734 typeof(_Generic((x), \
735 char: (char)0, \
736 __scalar_type_to_expr_cases(char), \
737 __scalar_type_to_expr_cases(short), \
738 __scalar_type_to_expr_cases(int), \
739 __scalar_type_to_expr_cases(long), \
740 __scalar_type_to_expr_cases(long long), \
741 default: (typeof(x))0))
742
743 #define READ_ONCE(x) \
744 ({ \
745 union { __unqual_typeof(x) __val; char __c[1]; } __u = \
746 { .__c = { 0 } }; \
747 __read_once_size((__unqual_typeof(x) *)&(x), __u.__c, sizeof(x)); \
748 __u.__val; \
749 })
750
751 #define WRITE_ONCE(x, val) \
752 ({ \
753 union { __unqual_typeof(x) __val; char __c[1]; } __u = \
754 { .__val = (val) }; \
755 __write_once_size((__unqual_typeof(x) *)&(x), __u.__c, sizeof(x)); \
756 __u.__val; \
757 })
758
759 /*
760 * __calc_avg - Calculate exponential weighted moving average (EWMA) with
761 * @old and @new values. @decay represents how large the @old value remains.
762 * With a larger @decay value, the moving average changes slowly, exhibiting
763 * fewer fluctuations.
764 */
765 #define __calc_avg(old, new, decay) ({ \
766 typeof(decay) thr = 1 << (decay); \
767 typeof(old) ret; \
768 if (((old) < thr) || ((new) < thr)) { \
769 if (((old) == 1) && ((new) == 0)) \
770 ret = 0; \
771 else \
772 ret = ((old) - ((old) >> 1)) + ((new) >> 1); \
773 } else { \
774 ret = ((old) - ((old) >> (decay))) + ((new) >> (decay)); \
775 } \
776 ret; \
777 })
778
779 /*
780 * log2_u32 - Compute the base 2 logarithm of a 32-bit exponential value.
781 * @v: The value for which we're computing the base 2 logarithm.
782 */
log2_u32(u32 v)783 static inline u32 log2_u32(u32 v)
784 {
785 u32 r;
786 u32 shift;
787
788 r = (v > 0xFFFF) << 4; v >>= r;
789 shift = (v > 0xFF) << 3; v >>= shift; r |= shift;
790 shift = (v > 0xF) << 2; v >>= shift; r |= shift;
791 shift = (v > 0x3) << 1; v >>= shift; r |= shift;
792 r |= (v >> 1);
793 return r;
794 }
795
796 /*
797 * log2_u64 - Compute the base 2 logarithm of a 64-bit exponential value.
798 * @v: The value for which we're computing the base 2 logarithm.
799 */
log2_u64(u64 v)800 static inline u32 log2_u64(u64 v)
801 {
802 u32 hi = v >> 32;
803 if (hi)
804 return log2_u32(hi) + 32 + 1;
805 else
806 return log2_u32(v) + 1;
807 }
808
809 /*
810 * sqrt_u64 - Calculate the square root of value @x using Newton's method.
811 */
__sqrt_u64(u64 x)812 static inline u64 __sqrt_u64(u64 x)
813 {
814 if (x == 0 || x == 1)
815 return x;
816
817 u64 r = ((1ULL << 32) > x) ? x : (1ULL << 32);
818
819 for (int i = 0; i < 8; ++i) {
820 u64 q = x / r;
821 if (r <= q)
822 break;
823 r = (r + q) >> 1;
824 }
825 return r;
826 }
827
828 /*
829 * ctzll -- Counts trailing zeros in an unsigned long long. If the input value
830 * is zero, the return value is undefined.
831 */
ctzll(u64 v)832 static inline int ctzll(u64 v)
833 {
834 #if (!defined(__BPF__) && defined(__SCX_TARGET_ARCH_x86)) || \
835 (defined(__BPF__) && defined(__clang_major__) && __clang_major__ >= 19)
836 /*
837 * Use the ctz builtin when: (1) building for native x86, or
838 * (2) building for BPF with clang >= 19 (BPF backend supports
839 * the intrinsic from clang 19 onward; earlier versions hit
840 * "unimplemented opcode" in the backend).
841 */
842 return __builtin_ctzll(v);
843 #else
844 /*
845 * If neither the target architecture nor the toolchains support ctzll,
846 * use software-based emulation. Let's use the De Bruijn sequence-based
847 * approach to find LSB fastly. See the details of De Bruijn sequence:
848 *
849 * https://en.wikipedia.org/wiki/De_Bruijn_sequence
850 * https://www.chessprogramming.org/BitScan#De_Bruijn_Multiplication
851 */
852 const int lookup_table[64] = {
853 0, 1, 48, 2, 57, 49, 28, 3, 61, 58, 50, 42, 38, 29, 17, 4,
854 62, 55, 59, 36, 53, 51, 43, 22, 45, 39, 33, 30, 24, 18, 12, 5,
855 63, 47, 56, 27, 60, 41, 37, 16, 54, 35, 52, 21, 44, 32, 23, 11,
856 46, 26, 40, 15, 34, 20, 31, 10, 25, 14, 19, 9, 13, 8, 7, 6,
857 };
858 const u64 DEBRUIJN_CONSTANT = 0x03f79d71b4cb0a89ULL;
859 unsigned int index;
860 u64 lowest_bit;
861 const int *lt;
862
863 if (v == 0)
864 return -1;
865
866 /*
867 * Isolate the least significant bit (LSB).
868 * For example, if v = 0b...10100, then v & -v = 0b...00100
869 */
870 lowest_bit = v & -v;
871
872 /*
873 * Each isolated bit produces a unique 6-bit value, guaranteed by the
874 * De Bruijn property. Calculate a unique index into the lookup table
875 * using the magic constant and a right shift.
876 *
877 * Multiplying by the 64-bit constant "spreads out" that 1-bit into a
878 * unique pattern in the top 6 bits. This uniqueness property is
879 * exactly what a De Bruijn sequence guarantees: Every possible 6-bit
880 * pattern (in top bits) occurs exactly once for each LSB position. So,
881 * the constant 0x03f79d71b4cb0a89ULL is carefully chosen to be a
882 * De Bruijn sequence, ensuring no collisions in the table index.
883 */
884 index = (lowest_bit * DEBRUIJN_CONSTANT) >> 58;
885
886 /*
887 * Lookup in a precomputed table. No collision is guaranteed by the
888 * De Bruijn property.
889 */
890 lt = MEMBER_VPTR(lookup_table, [index]);
891 return (lt)? *lt : -1;
892 #endif
893 }
894
895 /*
896 * Return a value proportionally scaled to the task's weight.
897 */
scale_by_task_weight(const struct task_struct * p,u64 value)898 static inline u64 scale_by_task_weight(const struct task_struct *p, u64 value)
899 {
900 return (value * p->scx.weight) / 100;
901 }
902
903 /*
904 * Return a value inversely proportional to the task's weight.
905 */
scale_by_task_weight_inverse(const struct task_struct * p,u64 value)906 static inline u64 scale_by_task_weight_inverse(const struct task_struct *p, u64 value)
907 {
908 return value * 100 / p->scx.weight;
909 }
910
911
912 /*
913 * Get a random u64 from the kernel's pseudo-random generator.
914 */
get_prandom_u64()915 static inline u64 get_prandom_u64()
916 {
917 return ((u64)bpf_get_prandom_u32() << 32) | bpf_get_prandom_u32();
918 }
919
920 /*
921 * Define the shadow structure to avoid a compilation error when
922 * vmlinux.h does not enable necessary kernel configs. The ___local
923 * suffix is a CO-RE convention that tells the loader to match this
924 * against the base struct rq in the kernel. The attribute
925 * preserve_access_index tells the compiler to generate a CO-RE
926 * relocation for these fields.
927 */
928 struct rq___local {
929 /*
930 * A monotonically increasing clock per CPU. It is rq->clock minus
931 * cumulative IRQ time and hypervisor steal time. Unlike rq->clock,
932 * it does not advance during IRQ processing or hypervisor preemption.
933 * It does advance during idle (the idle task counts as a running task
934 * for this purpose).
935 */
936 u64 clock_task;
937 /*
938 * Invariant version of clock_task scaled by CPU capacity and
939 * frequency. For example, clock_pelt advances 2x slower on a CPU
940 * with half the capacity.
941 *
942 * At idle exit, rq->clock_pelt jumps forward to resync with
943 * clock_task. The kernel's rq_clock_pelt() corrects for this jump
944 * by subtracting lost_idle_time, yielding a clock that appears
945 * continuous across idle transitions. scx_clock_pelt() mirrors
946 * rq_clock_pelt() by performing the same subtraction.
947 */
948 u64 clock_pelt;
949 /*
950 * Accumulates the magnitude of each clock_pelt jump at idle exit.
951 * Subtracting this from clock_pelt gives rq_clock_pelt(): a
952 * continuous, capacity-invariant clock suitable for both task
953 * execution time stamping and cross-idle measurements.
954 */
955 unsigned long lost_idle_time;
956 /*
957 * Shadow of paravirt_steal_clock() (the hypervisor's cumulative
958 * stolen time counter). Stays frozen while the hypervisor preempts
959 * the vCPU; catches up the next time update_rq_clock_task() is
960 * called. The delta is the stolen time not yet subtracted from
961 * clock_task.
962 *
963 * Unlike irqtime->total (a plain kernel-side field), the live stolen
964 * time counter lives in hypervisor-specific shared memory and has no
965 * kernel-side equivalent readable from BPF in a hypervisor-agnostic
966 * way. This field is therefore the only portable BPF-accessible
967 * approximation of cumulative steal time.
968 *
969 * Available only when CONFIG_PARAVIRT_TIME_ACCOUNTING is on.
970 */
971 u64 prev_steal_time_rq;
972 } __attribute__((preserve_access_index));
973
974 extern struct rq runqueues __ksym;
975
976 /*
977 * Define the shadow structure to avoid a compilation error when
978 * vmlinux.h does not enable necessary kernel configs.
979 */
980 struct irqtime___local {
981 /*
982 * Cumulative IRQ time counter for this CPU, in nanoseconds. Advances
983 * immediately at the exit of every hardirq and non-ksoftirqd softirq
984 * via irqtime_account_irq(). ksoftirqd time is counted as normal
985 * task time and is NOT included. NMI time is also NOT included.
986 *
987 * The companion field irqtime->sync (struct u64_stats_sync) protects
988 * against 64-bit tearing on 32-bit architectures. On 64-bit kernels,
989 * u64_stats_sync is an empty struct and all seqcount operations are
990 * no-ops, so a plain BPF_CORE_READ of this field is safe.
991 *
992 * Available only when CONFIG_IRQ_TIME_ACCOUNTING is on.
993 */
994 u64 total;
995 } __attribute__((preserve_access_index));
996
997 /*
998 * cpu_irqtime is a per-CPU variable defined only when
999 * CONFIG_IRQ_TIME_ACCOUNTING is on. Declare it as __weak so the BPF
1000 * loader sets its address to 0 (rather than failing) when the symbol
1001 * is absent from the running kernel.
1002 */
1003 extern struct irqtime___local cpu_irqtime __ksym __weak;
1004
get_current_rq(u32 cpu)1005 static inline struct rq___local *get_current_rq(u32 cpu)
1006 {
1007 /*
1008 * This is a workaround to get an rq pointer now that
1009 * scx_bpf_cpu_rq() has been removed.
1010 *
1011 * WARNING: The caller must hold the rq lock for @cpu. This is
1012 * guaranteed when called from scheduling callbacks (ops.running,
1013 * ops.stopping, ops.enqueue, ops.dequeue, ops.dispatch, etc.).
1014 * There is no runtime check available in BPF for kernel spinlock
1015 * state — correctness is enforced by calling context only.
1016 */
1017 return (void *)bpf_per_cpu_ptr(&runqueues, cpu);
1018 }
1019
scx_clock_task(u32 cpu)1020 static inline u64 scx_clock_task(u32 cpu)
1021 {
1022 struct rq___local *rq = get_current_rq(cpu);
1023
1024 /* Equivalent to the kernel's rq_clock_task(). */
1025 return rq ? rq->clock_task : 0;
1026 }
1027
scx_clock_pelt(u32 cpu)1028 static inline u64 scx_clock_pelt(u32 cpu)
1029 {
1030 struct rq___local *rq = get_current_rq(cpu);
1031
1032 /*
1033 * Equivalent to the kernel's rq_clock_pelt(): subtracts
1034 * lost_idle_time from clock_pelt to absorb the jump that occurs
1035 * when clock_pelt resyncs with clock_task at idle exit. The result
1036 * is a continuous, capacity-invariant clock safe for both task
1037 * execution time stamping and cross-idle measurements.
1038 */
1039 return rq ? (rq->clock_pelt - rq->lost_idle_time) : 0;
1040 }
1041
scx_clock_virt(u32 cpu)1042 static inline u64 scx_clock_virt(u32 cpu)
1043 {
1044 struct rq___local *rq;
1045
1046 /*
1047 * Check field existence before calling get_current_rq() so we avoid
1048 * the per_cpu lookup entirely on kernels built without
1049 * CONFIG_PARAVIRT_TIME_ACCOUNTING.
1050 */
1051 if (!bpf_core_field_exists(((struct rq___local *)0)->prev_steal_time_rq))
1052 return 0;
1053
1054 /* Lagging shadow of the kernel's paravirt_steal_clock(). */
1055 rq = get_current_rq(cpu);
1056 return rq ? BPF_CORE_READ(rq, prev_steal_time_rq) : 0;
1057 }
1058
scx_clock_irq(u32 cpu)1059 static inline u64 scx_clock_irq(u32 cpu)
1060 {
1061 struct irqtime___local *irqt;
1062
1063 /*
1064 * bpf_core_type_exists() resolves at load time: if struct irqtime is
1065 * absent from kernel BTF (CONFIG_IRQ_TIME_ACCOUNTING off), the loader
1066 * patches this into an unconditional return 0, making the
1067 * bpf_per_cpu_ptr() call below dead code that the verifier never sees.
1068 */
1069 if (!bpf_core_type_exists(struct irqtime___local))
1070 return 0;
1071
1072 /* Equivalent to the kernel's irq_time_read(). */
1073 irqt = bpf_per_cpu_ptr(&cpu_irqtime, cpu);
1074 return irqt ? BPF_CORE_READ(irqt, total) : 0;
1075 }
1076
1077 /* Abbreviated forms of <linux/overflow.h>'s struct_size() family. */
1078 #define flex_array_size(p, member, count) \
1079 ((count) * sizeof(*(p)->member))
1080
1081 #define struct_size(p, member, count) \
1082 (offsetof(typeof(*(p)), member) + flex_array_size(p, member, count))
1083
1084 #define struct_size_t(type, member, count) \
1085 struct_size((type *)NULL, member, count)
1086
1087 #include "compat.bpf.h"
1088 #include "enums.bpf.h"
1089 #include "cid.bpf.h"
1090
1091 #endif /* __SCX_COMMON_BPF_H */
1092