xref: /linux/kernel/bpf/verifier.c (revision b88822d2584f78cbdad1d8256510dcac108b5630)
1 // SPDX-License-Identifier: GPL-2.0-only
2 /* Copyright (c) 2011-2014 PLUMgrid, http://plumgrid.com
3  * Copyright (c) 2016 Facebook
4  * Copyright (c) 2018 Covalent IO, Inc. http://covalent.io
5  */
6 #include <uapi/linux/btf.h>
7 #include <linux/bpf-cgroup.h>
8 #include <linux/kernel.h>
9 #include <linux/types.h>
10 #include <linux/slab.h>
11 #include <linux/bpf.h>
12 #include <linux/btf.h>
13 #include <linux/bpf_verifier.h>
14 #include <linux/filter.h>
15 #include <net/netlink.h>
16 #include <linux/file.h>
17 #include <linux/vmalloc.h>
18 #include <linux/stringify.h>
19 #include <linux/bsearch.h>
20 #include <linux/sort.h>
21 #include <linux/perf_event.h>
22 #include <linux/ctype.h>
23 #include <linux/error-injection.h>
24 #include <linux/bpf_lsm.h>
25 #include <linux/security.h>
26 #include <linux/verification.h>
27 #include <linux/btf_ids.h>
28 #include <linux/poison.h>
29 #include <linux/module.h>
30 #include <linux/cpumask.h>
31 #include <linux/cnum.h>
32 #include <linux/bpf_mem_alloc.h>
33 #include <net/xdp.h>
34 #include <linux/trace_events.h>
35 #include <linux/kallsyms.h>
36 
37 #include "diagnostics.h"
38 #include "disasm.h"
39 
40 static const struct bpf_verifier_ops * const bpf_verifier_ops[] = {
41 #define BPF_PROG_TYPE(_id, _name, prog_ctx_type, kern_ctx_type) \
42 	[_id] = & _name ## _verifier_ops,
43 #define BPF_MAP_TYPE(_id, _ops)
44 #define BPF_LINK_TYPE(_id, _name)
45 #include <linux/bpf_types.h>
46 #undef BPF_PROG_TYPE
47 #undef BPF_MAP_TYPE
48 #undef BPF_LINK_TYPE
49 };
50 
51 enum bpf_features {
52 	BPF_FEAT_RDONLY_CAST_TO_VOID = 0,
53 	BPF_FEAT_STREAMS	     = 1,
54 	__MAX_BPF_FEAT,
55 };
56 
57 struct bpf_mem_alloc bpf_global_percpu_ma;
58 static bool bpf_global_percpu_ma_set;
59 
60 /* bpf_check() is a static code analyzer that walks eBPF program
61  * instruction by instruction and updates register/stack state.
62  * All paths of conditional branches are analyzed until 'bpf_exit' insn.
63  *
64  * The first pass is depth-first-search to check that the program is a DAG.
65  * It rejects the following programs:
66  * - larger than BPF_MAXINSNS insns
67  * - if loop is present (detected via back-edge)
68  * - unreachable insns exist (shouldn't be a forest. program = one function)
69  * - out of bounds or malformed jumps
70  * The second pass is all possible path descent from the 1st insn.
71  * Since it's analyzing all paths through the program, the length of the
72  * analysis is limited to 64k insn, which may be hit even if total number of
73  * insn is less then 4K, but there are too many branches that change stack/regs.
74  * Number of 'branches to be analyzed' is limited to 1k
75  *
76  * On entry to each instruction, each register has a type, and the instruction
77  * changes the types of the registers depending on instruction semantics.
78  * If instruction is BPF_MOV64_REG(BPF_REG_1, BPF_REG_5), then type of R5 is
79  * copied to R1.
80  *
81  * All registers are 64-bit.
82  * R0 - return register
83  * R1-R5 argument passing registers
84  * R6-R9 callee saved registers
85  * R10 - frame pointer read-only
86  *
87  * At the start of BPF program the register R1 contains a pointer to bpf_context
88  * and has type PTR_TO_CTX.
89  *
90  * Verifier tracks arithmetic operations on pointers in case:
91  *    BPF_MOV64_REG(BPF_REG_1, BPF_REG_10),
92  *    BPF_ALU64_IMM(BPF_ADD, BPF_REG_1, -20),
93  * 1st insn copies R10 (which has FRAME_PTR) type into R1
94  * and 2nd arithmetic instruction is pattern matched to recognize
95  * that it wants to construct a pointer to some element within stack.
96  * So after 2nd insn, the register R1 has type PTR_TO_STACK
97  * (and -20 constant is saved for further stack bounds checking).
98  * Meaning that this reg is a pointer to stack plus known immediate constant.
99  *
100  * Most of the time the registers have SCALAR_VALUE type, which
101  * means the register has some value, but it's not a valid pointer.
102  * (like pointer plus pointer becomes SCALAR_VALUE type)
103  *
104  * When verifier sees load or store instructions the type of base register
105  * can be: PTR_TO_MAP_VALUE, PTR_TO_CTX, PTR_TO_STACK, PTR_TO_SOCKET. These are
106  * four pointer types recognized by check_mem_access() function.
107  *
108  * PTR_TO_MAP_VALUE means that this register is pointing to 'map element value'
109  * and the range of [ptr, ptr + map's value_size) is accessible.
110  *
111  * registers used to pass values to function calls are checked against
112  * function argument constraints.
113  *
114  * ARG_PTR_TO_MAP_KEY is one of such argument constraints.
115  * It means that the register type passed to this function must be
116  * PTR_TO_STACK and it will be used inside the function as
117  * 'pointer to map element key'
118  *
119  * For example the argument constraints for bpf_map_lookup_elem():
120  *   .ret_type = RET_PTR_TO_MAP_VALUE_OR_NULL,
121  *   .arg1_type = ARG_CONST_MAP_PTR,
122  *   .arg2_type = ARG_PTR_TO_MAP_KEY,
123  *
124  * ret_type says that this function returns 'pointer to map elem value or null'
125  * function expects 1st argument to be a const pointer to 'struct bpf_map' and
126  * 2nd argument should be a pointer to stack, which will be used inside
127  * the helper function as a pointer to map element key.
128  *
129  * On the kernel side the helper function looks like:
130  * u64 bpf_map_lookup_elem(u64 r1, u64 r2, u64 r3, u64 r4, u64 r5)
131  * {
132  *    struct bpf_map *map = (struct bpf_map *) (unsigned long) r1;
133  *    void *key = (void *) (unsigned long) r2;
134  *    void *value;
135  *
136  *    here kernel can access 'key' and 'map' pointers safely, knowing that
137  *    [key, key + map->key_size) bytes are valid and were initialized on
138  *    the stack of eBPF program.
139  * }
140  *
141  * Corresponding eBPF program may look like:
142  *    BPF_MOV64_REG(BPF_REG_2, BPF_REG_10),  // after this insn R2 type is FRAME_PTR
143  *    BPF_ALU64_IMM(BPF_ADD, BPF_REG_2, -4), // after this insn R2 type is PTR_TO_STACK
144  *    BPF_LD_MAP_FD(BPF_REG_1, map_fd),      // after this insn R1 type is CONST_PTR_TO_MAP
145  *    BPF_RAW_INSN(BPF_JMP | BPF_CALL, 0, 0, 0, BPF_FUNC_map_lookup_elem),
146  * here verifier looks at prototype of map_lookup_elem() and sees:
147  * .arg1_type == ARG_CONST_MAP_PTR and R1->type == CONST_PTR_TO_MAP, which is ok,
148  * Now verifier knows that this map has key of R1->map_ptr->key_size bytes
149  *
150  * Then .arg2_type == ARG_PTR_TO_MAP_KEY and R2->type == PTR_TO_STACK, ok so far,
151  * Now verifier checks that [R2, R2 + map's key_size) are within stack limits
152  * and were initialized prior to this call.
153  * If it's ok, then verifier allows this BPF_CALL insn and looks at
154  * .ret_type which is RET_PTR_TO_MAP_VALUE_OR_NULL, so it sets
155  * R0->type = PTR_TO_MAP_VALUE_OR_NULL which means bpf_map_lookup_elem() function
156  * returns either pointer to map value or NULL.
157  *
158  * When type PTR_TO_MAP_VALUE_OR_NULL passes through 'if (reg != 0) goto +off'
159  * insn, the register holding that pointer in the true branch changes state to
160  * PTR_TO_MAP_VALUE and the same register changes state to CONST_IMM in the false
161  * branch. See check_cond_jmp_op().
162  *
163  * After the call R0 is set to return type of the function and registers R1-R5
164  * are set to NOT_INIT to indicate that they are no longer readable.
165  *
166  * The following reference types represent a potential reference to a kernel
167  * resource which, after first being allocated, must be checked and freed by
168  * the BPF program:
169  * - PTR_TO_SOCKET_OR_NULL, PTR_TO_SOCKET
170  *
171  * When the verifier sees a helper call return a reference type, it allocates a
172  * pointer id for the reference and stores it in the current function state.
173  * Similar to the way that PTR_TO_MAP_VALUE_OR_NULL is converted into
174  * PTR_TO_MAP_VALUE, PTR_TO_SOCKET_OR_NULL becomes PTR_TO_SOCKET when the type
175  * passes through a NULL-check conditional. For the branch wherein the state is
176  * changed to CONST_IMM, the verifier releases the reference.
177  *
178  * For each helper function that allocates a reference, such as
179  * bpf_sk_lookup_tcp(), there is a corresponding release function, such as
180  * bpf_sk_release(). When a reference type passes into the release function,
181  * the verifier also releases the reference. If any unchecked or unreleased
182  * reference remains at the end of the program, the verifier rejects it.
183  */
184 
185 /* verifier_state + insn_idx are pushed to stack when branch is encountered */
186 struct bpf_verifier_stack_elem {
187 	/* verifier state is 'st'
188 	 * before processing instruction 'insn_idx'
189 	 * and after processing instruction 'prev_insn_idx'
190 	 */
191 	struct bpf_verifier_state st;
192 	int insn_idx;
193 	int prev_insn_idx;
194 	struct bpf_verifier_stack_elem *next;
195 	/* length of verifier log at the time this state was pushed on stack */
196 	u32 log_pos;
197 	u64 diag_log_pos;
198 };
199 
200 #define BPF_COMPLEXITY_LIMIT_JMP_SEQ	8192
201 #define BPF_COMPLEXITY_LIMIT_STATES	64
202 
203 #define BPF_GLOBAL_PERCPU_MA_MAX_SIZE  512
204 
205 #define BPF_PRIV_STACK_MIN_SIZE		64
206 
207 static int acquire_reference(struct bpf_verifier_env *env, int insn_idx, int parent_id);
208 static int __release_reference_nomark(struct bpf_verifier_state *state, int id);
209 static int release_reference_nomark(struct bpf_verifier_env *env, int id);
210 static int release_reference(struct bpf_verifier_env *env, int id);
211 static void invalidate_non_owning_refs(struct bpf_verifier_env *env);
212 static void invalidate_rcu_protected_refs(struct bpf_verifier_env *env);
213 static bool in_rbtree_lock_required_cb(struct bpf_verifier_env *env);
214 static bool is_tracing_prog_type(enum bpf_prog_type type);
215 static int ref_set_non_owning(struct bpf_verifier_env *env,
216 			      struct bpf_reg_state *reg);
217 static bool is_trusted_reg(struct bpf_verifier_env *env, const struct bpf_reg_state *reg);
218 static inline bool in_sleepable_context(struct bpf_verifier_env *env);
219 static const char *non_sleepable_context_description(struct bpf_verifier_env *env);
220 static void scalar32_min_max_add(struct bpf_reg_state *dst_reg, struct bpf_reg_state *src_reg);
221 static void scalar_min_max_add(struct bpf_reg_state *dst_reg, struct bpf_reg_state *src_reg);
222 
223 static void bpf_map_ptr_store(struct bpf_insn_aux_data *aux,
224 			      struct bpf_map *map,
225 			      bool unpriv, bool poison)
226 {
227 	unpriv |= bpf_map_ptr_unpriv(aux);
228 	aux->map_ptr_state.unpriv = unpriv;
229 	aux->map_ptr_state.poison = poison;
230 	aux->map_ptr_state.map_ptr = map;
231 }
232 
233 static void bpf_map_key_store(struct bpf_insn_aux_data *aux, u64 state)
234 {
235 	bool poisoned = bpf_map_key_poisoned(aux);
236 
237 	aux->map_key_state = state | BPF_MAP_KEY_SEEN |
238 			     (poisoned ? BPF_MAP_KEY_POISON : 0ULL);
239 }
240 
241 static void update_ref_obj(struct ref_obj_desc *ref_obj, struct bpf_reg_state *reg)
242 {
243 	ref_obj->id = reg->id;
244 	ref_obj->parent_id = reg->parent_id;
245 	ref_obj->cnt++;
246 }
247 
248 static int validate_ref_obj(struct bpf_verifier_env *env, struct ref_obj_desc *ref_obj)
249 {
250 	if (ref_obj->cnt > 1) {
251 		verifier_bug(env, "function expects only one referenced object but got %d\n",
252 			     ref_obj->cnt);
253 		return -EFAULT;
254 	}
255 
256 	return 0;
257 }
258 
259 struct bpf_kfunc_meta {
260 	struct btf *btf;
261 	const struct btf_type *proto;
262 	const char *name;
263 	const u32 *flags;
264 	s32 id;
265 };
266 
267 struct btf *btf_vmlinux;
268 
269 typedef struct argno {
270 	int argno;
271 } argno_t;
272 
273 static argno_t argno_from_reg(u32 regno)
274 {
275 	return (argno_t){ .argno = regno };
276 }
277 
278 static argno_t argno_from_arg(u32 arg)
279 {
280 	return (argno_t){ .argno = -arg };
281 }
282 
283 static int reg_from_argno(argno_t a)
284 {
285 	if (a.argno >= 0)
286 		return a.argno;
287 	if (a.argno >= -MAX_BPF_FUNC_REG_ARGS)
288 		return -a.argno;
289 	return -1;
290 }
291 
292 static int arg_from_argno(argno_t a)
293 {
294 	if (a.argno < 0)
295 		return -a.argno;
296 	return -1;
297 }
298 
299 static int arg_idx_from_argno(argno_t a)
300 {
301 	return arg_from_argno(a) - 1;
302 }
303 
304 static const char *btf_type_name(const struct btf *btf, u32 id)
305 {
306 	return btf_name_by_offset(btf, btf_type_by_id(btf, id)->name_off);
307 }
308 
309 static DEFINE_MUTEX(bpf_verifier_lock);
310 static DEFINE_MUTEX(btf_vmlinux_lock);
311 static DEFINE_MUTEX(bpf_percpu_ma_lock);
312 
313 __printf(2, 3) static void verbose(void *private_data, const char *fmt, ...)
314 {
315 	struct bpf_verifier_env *env = private_data;
316 	va_list args;
317 
318 	if (!bpf_verifier_log_needed(&env->log))
319 		return;
320 
321 	va_start(args, fmt);
322 	bpf_verifier_vlog(&env->log, fmt, args);
323 	va_end(args);
324 }
325 
326 static void verbose_invalid_scalar(struct bpf_verifier_env *env,
327 				   struct bpf_reg_state *reg,
328 				   struct bpf_retval_range range, const char *ctx,
329 				   const char *reg_name)
330 {
331 	bool unknown = true;
332 
333 	verbose(env, "%s the register %s has", ctx, reg_name);
334 	if (reg_smin(reg) > S64_MIN) {
335 		verbose(env, " smin=%lld", reg_smin(reg));
336 		unknown = false;
337 	}
338 	if (reg_smax(reg) < S64_MAX) {
339 		verbose(env, " smax=%lld", reg_smax(reg));
340 		unknown = false;
341 	}
342 	if (unknown)
343 		verbose(env, " unknown scalar value");
344 	verbose(env, " should have been in [%d, %d]\n", range.minval, range.maxval);
345 }
346 
347 static bool reg_not_null(struct bpf_verifier_env *env, const struct bpf_reg_state *reg)
348 {
349 	enum bpf_reg_type type;
350 
351 	type = reg->type;
352 	if (type_may_be_null(type))
353 		return false;
354 
355 	/*
356 	 * The types below guarantee a non-NULL base, an unbounded offset can
357 	 * still wrap base + offset to zero.
358 	 */
359 	if (reg_smin(reg) <= -BPF_MAX_VAR_OFF || reg_smax(reg) >= BPF_MAX_VAR_OFF)
360 		return false;
361 
362 	type = base_type(type);
363 	return type == PTR_TO_SOCKET ||
364 		type == PTR_TO_TCP_SOCK ||
365 		type == PTR_TO_XDP_SOCK ||
366 		type == PTR_TO_BUF ||
367 		type == PTR_TO_MAP_VALUE ||
368 		type == PTR_TO_MAP_KEY ||
369 		type == PTR_TO_SOCK_COMMON ||
370 		(type == PTR_TO_BTF_ID && is_trusted_reg(env, reg)) ||
371 		(type == PTR_TO_MEM && !(reg->type & PTR_UNTRUSTED)) ||
372 		type == CONST_PTR_TO_MAP;
373 }
374 
375 static struct btf_record *reg_btf_record(const struct bpf_reg_state *reg)
376 {
377 	struct btf_record *rec = NULL;
378 	struct btf_struct_meta *meta;
379 
380 	if (reg->type == PTR_TO_MAP_VALUE) {
381 		rec = reg->map_ptr->record;
382 	} else if (type_is_ptr_alloc_obj(reg->type)) {
383 		meta = btf_find_struct_meta(reg->btf, reg->btf_id);
384 		if (meta)
385 			rec = meta->record;
386 	}
387 	return rec;
388 }
389 
390 bool bpf_subprog_is_global(const struct bpf_verifier_env *env, int subprog)
391 {
392 	struct bpf_func_info_aux *aux = env->prog->aux->func_info_aux;
393 
394 	return aux && aux[subprog].linkage == BTF_FUNC_GLOBAL;
395 }
396 
397 static bool subprog_returns_void(struct bpf_verifier_env *env, int subprog)
398 {
399 	const struct btf_type *type, *func, *func_proto;
400 	const struct btf *btf = env->prog->aux->btf;
401 	u32 btf_id;
402 
403 	btf_id = env->prog->aux->func_info[subprog].type_id;
404 
405 	func = btf_type_by_id(btf, btf_id);
406 	if (verifier_bug_if(!func, env, "btf_id %u not found", btf_id))
407 		return false;
408 
409 	func_proto = btf_type_by_id(btf, func->type);
410 	if (!func_proto)
411 		return false;
412 
413 	type = btf_type_skip_modifiers(btf, func_proto->type, NULL);
414 	if (!type)
415 		return false;
416 
417 	return btf_type_is_void(type);
418 }
419 
420 const char *bpf_subprog_name(const struct bpf_verifier_env *env, int subprog)
421 {
422 	struct bpf_func_info *info;
423 
424 	if (!env->prog->aux->func_info)
425 		return "";
426 
427 	info = &env->prog->aux->func_info[subprog];
428 	return btf_type_name(env->prog->aux->btf, info->type_id);
429 }
430 
431 void bpf_mark_subprog_exc_cb(struct bpf_verifier_env *env, int subprog)
432 {
433 	struct bpf_subprog_info *info = subprog_info(env, subprog);
434 
435 	info->is_cb = true;
436 	info->is_async_cb = true;
437 	info->is_exception_cb = true;
438 }
439 
440 static bool subprog_is_exc_cb(struct bpf_verifier_env *env, int subprog)
441 {
442 	return subprog_info(env, subprog)->is_exception_cb;
443 }
444 
445 static bool reg_may_point_to_spin_lock(const struct bpf_reg_state *reg)
446 {
447 	return btf_record_has_field(reg_btf_record(reg), BPF_SPIN_LOCK | BPF_RES_SPIN_LOCK);
448 }
449 
450 static bool type_is_rdonly_mem(u32 type)
451 {
452 	return type & MEM_RDONLY;
453 }
454 
455 static bool is_acquire_function(enum bpf_func_id func_id,
456 				const struct bpf_map *map)
457 {
458 	enum bpf_map_type map_type = map ? map->map_type : BPF_MAP_TYPE_UNSPEC;
459 
460 	if (func_id == BPF_FUNC_sk_lookup_tcp ||
461 	    func_id == BPF_FUNC_sk_lookup_udp ||
462 	    func_id == BPF_FUNC_skc_lookup_tcp ||
463 	    func_id == BPF_FUNC_ringbuf_reserve ||
464 	    func_id == BPF_FUNC_kptr_xchg)
465 		return true;
466 
467 	if (func_id == BPF_FUNC_map_lookup_elem &&
468 	    (map_type == BPF_MAP_TYPE_SOCKMAP ||
469 	     map_type == BPF_MAP_TYPE_SOCKHASH))
470 		return true;
471 
472 	return false;
473 }
474 
475 static bool is_ptr_cast_function(enum bpf_func_id func_id)
476 {
477 	return func_id == BPF_FUNC_tcp_sock ||
478 		func_id == BPF_FUNC_sk_fullsock ||
479 		func_id == BPF_FUNC_skc_to_tcp_sock ||
480 		func_id == BPF_FUNC_skc_to_tcp6_sock ||
481 		func_id == BPF_FUNC_skc_to_udp6_sock ||
482 		func_id == BPF_FUNC_skc_to_mptcp_sock ||
483 		func_id == BPF_FUNC_skc_to_tcp_timewait_sock ||
484 		func_id == BPF_FUNC_skc_to_tcp_request_sock;
485 }
486 
487 static bool is_sync_callback_calling_kfunc(u32 btf_id);
488 static bool is_async_callback_calling_kfunc(u32 btf_id);
489 static bool is_callback_calling_kfunc(u32 btf_id);
490 
491 static bool is_bpf_wq_set_callback_kfunc(u32 btf_id);
492 static bool is_task_work_add_kfunc(u32 func_id);
493 
494 static bool is_sync_callback_calling_function(enum bpf_func_id func_id)
495 {
496 	return func_id == BPF_FUNC_for_each_map_elem ||
497 	       func_id == BPF_FUNC_find_vma ||
498 	       func_id == BPF_FUNC_loop ||
499 	       func_id == BPF_FUNC_user_ringbuf_drain;
500 }
501 
502 static bool is_async_callback_calling_function(enum bpf_func_id func_id)
503 {
504 	return func_id == BPF_FUNC_timer_set_callback;
505 }
506 
507 static bool is_callback_calling_function(enum bpf_func_id func_id)
508 {
509 	return is_sync_callback_calling_function(func_id) ||
510 	       is_async_callback_calling_function(func_id);
511 }
512 
513 bool bpf_is_sync_callback_calling_insn(struct bpf_insn *insn)
514 {
515 	return (bpf_helper_call(insn) && is_sync_callback_calling_function(insn->imm)) ||
516 	       (bpf_pseudo_kfunc_call(insn) && is_sync_callback_calling_kfunc(insn->imm));
517 }
518 
519 bool bpf_is_async_callback_calling_insn(struct bpf_insn *insn)
520 {
521 	return (bpf_helper_call(insn) && is_async_callback_calling_function(insn->imm)) ||
522 	       (bpf_pseudo_kfunc_call(insn) && is_async_callback_calling_kfunc(insn->imm));
523 }
524 
525 static bool is_async_cb_sleepable(struct bpf_verifier_env *env, struct bpf_insn *insn)
526 {
527 	/* bpf_timer callbacks are never sleepable. */
528 	if (bpf_helper_call(insn) && insn->imm == BPF_FUNC_timer_set_callback)
529 		return false;
530 
531 	/* bpf_wq and bpf_task_work callbacks are always sleepable. */
532 	if (bpf_pseudo_kfunc_call(insn) && insn->off == 0 &&
533 	    (is_bpf_wq_set_callback_kfunc(insn->imm) || is_task_work_add_kfunc(insn->imm)))
534 		return true;
535 
536 	verifier_bug(env, "unhandled async callback in is_async_cb_sleepable");
537 	return false;
538 }
539 
540 bool bpf_is_may_goto_insn(struct bpf_insn *insn)
541 {
542 	return insn->code == (BPF_JMP | BPF_JCOND) && insn->src_reg == BPF_MAY_GOTO;
543 }
544 
545 static bool is_spi_bounds_valid(struct bpf_func_state *state, int spi, int nr_slots)
546 {
547        int allocated_slots = state->allocated_stack / BPF_REG_SIZE;
548 
549        /* We need to check that slots between [spi - nr_slots + 1, spi] are
550 	* within [0, allocated_stack).
551 	*
552 	* Please note that the spi grows downwards. For example, a dynptr
553 	* takes the size of two stack slots; the first slot will be at
554 	* spi and the second slot will be at spi - 1.
555 	*/
556        return spi - nr_slots + 1 >= 0 && spi < allocated_slots;
557 }
558 
559 static int stack_slot_obj_get_spi(struct bpf_verifier_env *env, struct bpf_reg_state *reg,
560 			          const char *obj_kind, int nr_slots)
561 {
562 	int off, spi;
563 
564 	if (!tnum_is_const(reg->var_off)) {
565 		verbose(env, "%s has to be at a constant offset\n", obj_kind);
566 		return -EINVAL;
567 	}
568 
569 	off = reg->var_off.value;
570 	if (off >= 0 || off % BPF_REG_SIZE) {
571 		verbose(env, "cannot pass in %s at an offset=%d\n", obj_kind, off);
572 		return -EINVAL;
573 	}
574 
575 	spi = bpf_get_spi(off);
576 	if (spi + 1 < nr_slots) {
577 		verbose(env, "cannot pass in %s at an offset=%d\n", obj_kind, off);
578 		return -EINVAL;
579 	}
580 
581 	if (!is_spi_bounds_valid(bpf_func(env, reg), spi, nr_slots))
582 		return -ERANGE;
583 	return spi;
584 }
585 
586 static int dynptr_get_spi(struct bpf_verifier_env *env, struct bpf_reg_state *reg)
587 {
588 	return stack_slot_obj_get_spi(env, reg, "dynptr", BPF_DYNPTR_NR_SLOTS);
589 }
590 
591 static int iter_get_spi(struct bpf_verifier_env *env, struct bpf_reg_state *reg, int nr_slots)
592 {
593 	return stack_slot_obj_get_spi(env, reg, "iter", nr_slots);
594 }
595 
596 static int irq_flag_get_spi(struct bpf_verifier_env *env, struct bpf_reg_state *reg)
597 {
598 	return stack_slot_obj_get_spi(env, reg, "irq_flag", 1);
599 }
600 
601 static enum bpf_dynptr_type arg_to_dynptr_type(enum bpf_arg_type arg_type)
602 {
603 	switch (arg_type & DYNPTR_TYPE_FLAG_MASK) {
604 	case DYNPTR_TYPE_LOCAL:
605 		return BPF_DYNPTR_TYPE_LOCAL;
606 	case DYNPTR_TYPE_RINGBUF:
607 		return BPF_DYNPTR_TYPE_RINGBUF;
608 	case DYNPTR_TYPE_SKB:
609 		return BPF_DYNPTR_TYPE_SKB;
610 	case DYNPTR_TYPE_XDP:
611 		return BPF_DYNPTR_TYPE_XDP;
612 	case DYNPTR_TYPE_SKB_META:
613 		return BPF_DYNPTR_TYPE_SKB_META;
614 	case DYNPTR_TYPE_FILE:
615 		return BPF_DYNPTR_TYPE_FILE;
616 	default:
617 		return BPF_DYNPTR_TYPE_INVALID;
618 	}
619 }
620 
621 static enum bpf_type_flag get_dynptr_type_flag(enum bpf_dynptr_type type)
622 {
623 	switch (type) {
624 	case BPF_DYNPTR_TYPE_LOCAL:
625 		return DYNPTR_TYPE_LOCAL;
626 	case BPF_DYNPTR_TYPE_RINGBUF:
627 		return DYNPTR_TYPE_RINGBUF;
628 	case BPF_DYNPTR_TYPE_SKB:
629 		return DYNPTR_TYPE_SKB;
630 	case BPF_DYNPTR_TYPE_XDP:
631 		return DYNPTR_TYPE_XDP;
632 	case BPF_DYNPTR_TYPE_SKB_META:
633 		return DYNPTR_TYPE_SKB_META;
634 	case BPF_DYNPTR_TYPE_FILE:
635 		return DYNPTR_TYPE_FILE;
636 	default:
637 		return 0;
638 	}
639 }
640 
641 static bool dynptr_type_referenced(enum bpf_dynptr_type type)
642 {
643 	return type == BPF_DYNPTR_TYPE_RINGBUF || type == BPF_DYNPTR_TYPE_FILE;
644 }
645 
646 static void __mark_dynptr_reg(struct bpf_reg_state *reg,
647 			      enum bpf_dynptr_type type,
648 			      bool first_slot, int id, int parent_id);
649 
650 static void mark_dynptr_stack_regs(struct bpf_verifier_env *env,
651 				   struct bpf_reg_state *sreg1,
652 				   struct bpf_reg_state *sreg2,
653 				   enum bpf_dynptr_type type, int parent_id)
654 {
655 	int id = ++env->id_gen;
656 
657 	__mark_dynptr_reg(sreg1, type, true, id, parent_id);
658 	__mark_dynptr_reg(sreg2, type, false, id, parent_id);
659 }
660 
661 static void mark_dynptr_cb_reg(struct bpf_verifier_env *env,
662 			       struct bpf_reg_state *reg,
663 			       enum bpf_dynptr_type type)
664 {
665 	__mark_dynptr_reg(reg, type, true, ++env->id_gen, 0);
666 }
667 
668 static int destroy_if_dynptr_stack_slot(struct bpf_verifier_env *env,
669 				        struct bpf_func_state *state, int spi);
670 
671 static int mark_stack_slots_dynptr(struct bpf_verifier_env *env, struct bpf_reg_state *reg,
672 				   enum bpf_arg_type arg_type, int insn_idx,
673 				   struct ref_obj_desc *ref_obj, struct bpf_dynptr_desc *dynptr)
674 {
675 	struct bpf_func_state *state = bpf_func(env, reg);
676 	int spi, i, err, parent_id = 0;
677 	enum bpf_dynptr_type type;
678 
679 	spi = dynptr_get_spi(env, reg);
680 	if (spi < 0)
681 		return spi;
682 
683 	/* We cannot assume both spi and spi - 1 belong to the same dynptr,
684 	 * hence we need to call destroy_if_dynptr_stack_slot twice for both,
685 	 * to ensure that for the following example:
686 	 *	[d1][d1][d2][d2]
687 	 * spi    3   2   1   0
688 	 * So marking spi = 2 should lead to destruction of both d1 and d2. In
689 	 * case they do belong to same dynptr, second call won't see slot_type
690 	 * as STACK_DYNPTR and will simply skip destruction.
691 	 */
692 	err = destroy_if_dynptr_stack_slot(env, state, spi);
693 	if (err)
694 		return err;
695 	err = destroy_if_dynptr_stack_slot(env, state, spi - 1);
696 	if (err)
697 		return err;
698 
699 	for (i = 0; i < BPF_REG_SIZE; i++) {
700 		state->stack[spi].slot_type[i] = STACK_DYNPTR;
701 		state->stack[spi - 1].slot_type[i] = STACK_DYNPTR;
702 	}
703 
704 	type = arg_to_dynptr_type(arg_type);
705 	if (type == BPF_DYNPTR_TYPE_INVALID)
706 		return -EINVAL;
707 
708 	if (dynptr->type == BPF_DYNPTR_TYPE_INVALID) { /* dynptr constructors */
709 		err = validate_ref_obj(env, ref_obj);
710 		if (err)
711 			return err;
712 
713 		/* Track parent's id if the parent is a referenced object */
714 		parent_id = ref_obj->id;
715 
716 		if (dynptr_type_referenced(type)) {
717 			int id;
718 
719 			/*
720 			 * Create an intermediate reference that tracks the referenced
721 			 * object for the referenced dynptr. Freeing a referenced dynptr
722 			 * through helpers/kfuncs will invalidate all clones.
723 			 */
724 			id = acquire_reference(env, insn_idx, parent_id);
725 			if (id < 0)
726 				return id;
727 
728 			parent_id = id;
729 		}
730 	} else { /* bpf_dynptr_clone() */
731 		parent_id = dynptr->parent_id;
732 	}
733 
734 	mark_dynptr_stack_regs(env, &state->stack[spi].spilled_ptr,
735 			       &state->stack[spi - 1].spilled_ptr, type, parent_id);
736 
737 	return 0;
738 }
739 
740 static void invalidate_dynptr(struct bpf_verifier_env *env, struct bpf_stack_state *stack)
741 {
742 	int i;
743 
744 	for (i = 0; i < BPF_REG_SIZE; i++) {
745 		stack[0].slot_type[i] = STACK_INVALID;
746 		stack[1].slot_type[i] = STACK_INVALID;
747 	}
748 
749 	bpf_mark_reg_not_init(env, &stack[0].spilled_ptr);
750 	bpf_mark_reg_not_init(env, &stack[1].spilled_ptr);
751 }
752 
753 static int unmark_stack_slots_dynptr(struct bpf_verifier_env *env, struct bpf_reg_state *reg)
754 {
755 	struct bpf_func_state *state = bpf_func(env, reg);
756 	int spi;
757 
758 	spi = dynptr_get_spi(env, reg);
759 	if (spi < 0)
760 		return spi;
761 
762 	/*
763 	 * For referenced dynptr, release the parent ref which cascades to
764 	 * all clones and derived slices. For non-referenced dynptr, only
765 	 * the dynptr and slices derived from it will be invalidated.
766 	 */
767 	reg = &state->stack[spi].spilled_ptr;
768 	return release_reference(env, dynptr_type_referenced(reg->dynptr.type)
769 				      ? reg->parent_id
770 				      : reg->id);
771 }
772 
773 static void __mark_reg_unknown(const struct bpf_verifier_env *env,
774 			       struct bpf_reg_state *reg);
775 
776 static void mark_reg_invalid(const struct bpf_verifier_env *env, struct bpf_reg_state *reg)
777 {
778 	if (!env->allow_ptr_leaks)
779 		bpf_mark_reg_not_init(env, reg);
780 	else
781 		__mark_reg_unknown(env, reg);
782 }
783 
784 static int dynptr_ref_cnt(struct bpf_verifier_env *env, int v_parent_id)
785 {
786 	struct bpf_stack_state *stack;
787 	struct bpf_func_state *state;
788 	struct bpf_reg_state *reg;
789 	int ref_cnt = 0;
790 
791 	bpf_for_each_reg_in_vstate_mask(env->cur_state, state, reg, stack, 1 << STACK_DYNPTR, ({
792 		if (!stack || stack->slot_type[0] != STACK_DYNPTR)
793 			continue;
794 		if (!stack->spilled_ptr.dynptr.first_slot)
795 			continue;
796 		if (stack->spilled_ptr.parent_id == v_parent_id)
797 			ref_cnt++;
798 	}));
799 
800 	return ref_cnt;
801 }
802 
803 static int destroy_if_dynptr_stack_slot(struct bpf_verifier_env *env,
804 				        struct bpf_func_state *state, int spi)
805 {
806 	int err = 0;
807 
808 	/* We always ensure that STACK_DYNPTR is never set partially,
809 	 * hence just checking for slot_type[0] is enough. This is
810 	 * different for STACK_SPILL, where it may be only set for
811 	 * 1 byte, so code has to use is_spilled_reg.
812 	 */
813 	if (state->stack[spi].slot_type[0] != STACK_DYNPTR)
814 		return 0;
815 
816 	/* Reposition spi to first slot */
817 	if (!state->stack[spi].spilled_ptr.dynptr.first_slot)
818 		spi = spi + 1;
819 
820 	/*
821 	 * A referenced dynptr can be overwritten only if there is at
822 	 * least one other dynptr sharing the same virtual ref parent,
823 	 * ensuring the reference can still be properly released.
824 	 */
825 	if (dynptr_type_referenced(state->stack[spi].spilled_ptr.dynptr.type) &&
826 	    dynptr_ref_cnt(env, state->stack[spi].spilled_ptr.parent_id) <= 1) {
827 		verbose(env, "cannot overwrite referenced dynptr\n");
828 		bpf_diag_res(
829 			env, env->insn_idx, "referenced dynptr overwrite",
830 			"This stack slot contains a dynptr that owns or protects a referenced resource. Overwriting the last dynptr for that resource would lose the verifier-tracked release path.",
831 			"Release or clone the dynptr so another live dynptr still tracks the referenced resource before overwriting this stack slot.");
832 		return -EINVAL;
833 	}
834 
835 	/* Invalidate the dynptr and any derived slices */
836 	err = release_reference(env, state->stack[spi].spilled_ptr.id);
837 	if (!err) {
838 		mark_stack_slot_scratched(env, spi);
839 		mark_stack_slot_scratched(env, spi - 1);
840 	}
841 
842 	return err;
843 }
844 
845 static bool is_dynptr_reg_valid_uninit(struct bpf_verifier_env *env, struct bpf_reg_state *reg)
846 {
847 	int spi;
848 
849 	if (reg->type == CONST_PTR_TO_DYNPTR)
850 		return false;
851 
852 	spi = dynptr_get_spi(env, reg);
853 
854 	/* -ERANGE (i.e. spi not falling into allocated stack slots) isn't an
855 	 * error because this just means the stack state hasn't been updated yet.
856 	 * We will do check_mem_access to check and update stack bounds later.
857 	 */
858 	if (spi < 0 && spi != -ERANGE)
859 		return false;
860 
861 	/* We don't need to check if the stack slots are marked by previous
862 	 * dynptr initializations because we allow overwriting existing unreferenced
863 	 * STACK_DYNPTR slots, see mark_stack_slots_dynptr which calls
864 	 * destroy_if_dynptr_stack_slot to ensure dynptr objects at the slots we are
865 	 * touching are completely destructed before we reinitialize them for a new
866 	 * one. For referenced ones, destroy_if_dynptr_stack_slot returns an error early
867 	 * instead of delaying it until the end where the user will get "Unreleased
868 	 * reference" error.
869 	 */
870 	return true;
871 }
872 
873 static bool is_dynptr_reg_valid_init(struct bpf_verifier_env *env, struct bpf_reg_state *reg)
874 {
875 	struct bpf_func_state *state = bpf_func(env, reg);
876 	int i, spi;
877 
878 	/* This already represents first slot of initialized bpf_dynptr.
879 	 *
880 	 * CONST_PTR_TO_DYNPTR already has fixed and var_off as 0 due to
881 	 * check_func_arg_reg_off's logic, so we don't need to check its
882 	 * offset and alignment.
883 	 */
884 	if (reg->type == CONST_PTR_TO_DYNPTR)
885 		return true;
886 
887 	spi = dynptr_get_spi(env, reg);
888 	if (spi < 0)
889 		return false;
890 	if (!state->stack[spi].spilled_ptr.dynptr.first_slot)
891 		return false;
892 
893 	for (i = 0; i < BPF_REG_SIZE; i++) {
894 		if (state->stack[spi].slot_type[i] != STACK_DYNPTR ||
895 		    state->stack[spi - 1].slot_type[i] != STACK_DYNPTR)
896 			return false;
897 	}
898 
899 	return true;
900 }
901 
902 static enum bpf_dynptr_type dynptr_reg_type(struct bpf_verifier_env *env, struct bpf_reg_state *reg)
903 {
904 	struct bpf_func_state *state;
905 	int spi;
906 
907 	if (reg->type == CONST_PTR_TO_DYNPTR)
908 		return reg->dynptr.type;
909 
910 	spi = dynptr_get_spi(env, reg);
911 	if (spi < 0)
912 		return BPF_DYNPTR_TYPE_INVALID;
913 	state = bpf_func(env, reg);
914 	return state->stack[spi].spilled_ptr.dynptr.type;
915 }
916 
917 static bool is_dynptr_type_expected(struct bpf_verifier_env *env, struct bpf_reg_state *reg,
918 				    enum bpf_arg_type arg_type)
919 {
920 	/* ARG_PTR_TO_DYNPTR takes any type of dynptr */
921 	if (arg_type == ARG_PTR_TO_DYNPTR)
922 		return true;
923 
924 	return dynptr_reg_type(env, reg) == arg_to_dynptr_type(arg_type);
925 }
926 
927 static void __mark_reg_known_zero(struct bpf_reg_state *reg);
928 
929 static bool in_rcu_cs(struct bpf_verifier_env *env);
930 
931 static bool is_kfunc_rcu_protected(struct bpf_call_arg_meta *meta);
932 
933 static int mark_stack_slots_iter(struct bpf_verifier_env *env,
934 				 struct bpf_call_arg_meta *meta,
935 				 struct bpf_reg_state *reg, int insn_idx,
936 				 struct btf *btf, u32 btf_id, int nr_slots)
937 {
938 	struct bpf_func_state *state = bpf_func(env, reg);
939 	int spi, i, j, id;
940 
941 	spi = iter_get_spi(env, reg, nr_slots);
942 	if (spi < 0)
943 		return spi;
944 
945 	id = acquire_reference(env, insn_idx, 0);
946 	if (id < 0)
947 		return id;
948 
949 	for (i = 0; i < nr_slots; i++) {
950 		struct bpf_stack_state *slot = &state->stack[spi - i];
951 		struct bpf_reg_state *st = &slot->spilled_ptr;
952 
953 		__mark_reg_known_zero(st);
954 		st->type = PTR_TO_STACK; /* we don't have dedicated reg type */
955 		if (is_kfunc_rcu_protected(meta)) {
956 			if (in_rcu_cs(env))
957 				st->type |= MEM_RCU;
958 			else
959 				st->type |= PTR_UNTRUSTED;
960 		}
961 		st->id = i == 0 ? id : 0;
962 		st->iter.btf = btf;
963 		st->iter.btf_id = btf_id;
964 		st->iter.state = BPF_ITER_STATE_ACTIVE;
965 		st->iter.depth = 0;
966 
967 		for (j = 0; j < BPF_REG_SIZE; j++)
968 			slot->slot_type[j] = STACK_ITER;
969 
970 		mark_stack_slot_scratched(env, spi - i);
971 	}
972 
973 	return 0;
974 }
975 
976 static int unmark_stack_slots_iter(struct bpf_verifier_env *env,
977 				   struct bpf_reg_state *reg, int nr_slots)
978 {
979 	struct bpf_func_state *state = bpf_func(env, reg);
980 	int spi, i, j;
981 
982 	spi = iter_get_spi(env, reg, nr_slots);
983 	if (spi < 0)
984 		return spi;
985 
986 	for (i = 0; i < nr_slots; i++) {
987 		struct bpf_stack_state *slot = &state->stack[spi - i];
988 		struct bpf_reg_state *st = &slot->spilled_ptr;
989 
990 		if (i == 0)
991 			WARN_ON_ONCE(release_reference(env, st->id));
992 
993 		bpf_mark_reg_not_init(env, st);
994 
995 		for (j = 0; j < BPF_REG_SIZE; j++)
996 			slot->slot_type[j] = STACK_INVALID;
997 
998 		mark_stack_slot_scratched(env, spi - i);
999 	}
1000 
1001 	return 0;
1002 }
1003 
1004 static bool is_iter_reg_valid_uninit(struct bpf_verifier_env *env,
1005 				     struct bpf_reg_state *reg, int nr_slots)
1006 {
1007 	struct bpf_func_state *state = bpf_func(env, reg);
1008 	int spi, i, j;
1009 
1010 	/* For -ERANGE (i.e. spi not falling into allocated stack slots), we
1011 	 * will do check_mem_access to check and update stack bounds later, so
1012 	 * return true for that case.
1013 	 */
1014 	spi = iter_get_spi(env, reg, nr_slots);
1015 	if (spi == -ERANGE)
1016 		return true;
1017 	if (spi < 0)
1018 		return false;
1019 
1020 	for (i = 0; i < nr_slots; i++) {
1021 		struct bpf_stack_state *slot = &state->stack[spi - i];
1022 
1023 		for (j = 0; j < BPF_REG_SIZE; j++)
1024 			if (slot->slot_type[j] == STACK_ITER)
1025 				return false;
1026 	}
1027 
1028 	return true;
1029 }
1030 
1031 static int is_iter_reg_valid_init(struct bpf_verifier_env *env, struct bpf_reg_state *reg,
1032 				   struct btf *btf, u32 btf_id, int nr_slots)
1033 {
1034 	struct bpf_func_state *state = bpf_func(env, reg);
1035 	int spi, i, j;
1036 
1037 	spi = iter_get_spi(env, reg, nr_slots);
1038 	if (spi < 0)
1039 		return -EINVAL;
1040 
1041 	for (i = 0; i < nr_slots; i++) {
1042 		struct bpf_stack_state *slot = &state->stack[spi - i];
1043 		struct bpf_reg_state *st = &slot->spilled_ptr;
1044 
1045 		if (st->type & PTR_UNTRUSTED)
1046 			return -EPROTO;
1047 		/* only main (first) slot has id set */
1048 		if (i == 0 && !st->id)
1049 			return -EINVAL;
1050 		if (i != 0 && st->id)
1051 			return -EINVAL;
1052 		if (st->iter.btf != btf || st->iter.btf_id != btf_id)
1053 			return -EINVAL;
1054 
1055 		for (j = 0; j < BPF_REG_SIZE; j++)
1056 			if (slot->slot_type[j] != STACK_ITER)
1057 				return -EINVAL;
1058 	}
1059 
1060 	return 0;
1061 }
1062 
1063 static int acquire_irq_state(struct bpf_verifier_env *env, int insn_idx);
1064 static int release_irq_state(struct bpf_verifier_env *env, int id);
1065 
1066 static int mark_stack_slot_irq_flag(struct bpf_verifier_env *env,
1067 				     struct bpf_call_arg_meta *meta,
1068 				     struct bpf_reg_state *reg, int insn_idx,
1069 				     int kfunc_class)
1070 {
1071 	struct bpf_func_state *state = bpf_func(env, reg);
1072 	struct bpf_stack_state *slot;
1073 	struct bpf_reg_state *st;
1074 	int spi, i, id;
1075 
1076 	spi = irq_flag_get_spi(env, reg);
1077 	if (spi < 0)
1078 		return spi;
1079 
1080 	id = acquire_irq_state(env, insn_idx);
1081 	if (id < 0)
1082 		return id;
1083 
1084 	slot = &state->stack[spi];
1085 	st = &slot->spilled_ptr;
1086 
1087 	__mark_reg_known_zero(st);
1088 	st->type = PTR_TO_STACK; /* we don't have dedicated reg type */
1089 	st->id = id;
1090 	st->irq.kfunc_class = kfunc_class;
1091 
1092 	for (i = 0; i < BPF_REG_SIZE; i++)
1093 		slot->slot_type[i] = STACK_IRQ_FLAG;
1094 
1095 	mark_stack_slot_scratched(env, spi);
1096 	return 0;
1097 }
1098 
1099 static int unmark_stack_slot_irq_flag(struct bpf_verifier_env *env, struct bpf_reg_state *reg,
1100 				      int kfunc_class)
1101 {
1102 	struct bpf_func_state *state = bpf_func(env, reg);
1103 	struct bpf_stack_state *slot;
1104 	struct bpf_reg_state *st;
1105 	int spi, i, err;
1106 
1107 	spi = irq_flag_get_spi(env, reg);
1108 	if (spi < 0)
1109 		return spi;
1110 
1111 	slot = &state->stack[spi];
1112 	st = &slot->spilled_ptr;
1113 
1114 	if (st->irq.kfunc_class != kfunc_class) {
1115 		const char *flag_kfunc = st->irq.kfunc_class == IRQ_NATIVE_KFUNC ? "native" : "lock";
1116 		const char *used_kfunc = kfunc_class == IRQ_NATIVE_KFUNC ? "native" : "lock";
1117 		const char *reason;
1118 
1119 		verbose(env, "irq flag acquired by %s kfuncs cannot be restored with %s kfuncs\n",
1120 			flag_kfunc, used_kfunc);
1121 		reason = bpf_diag_fmt(env,
1122 				      "This IRQ flag was saved by %s IRQ kfuncs, but the restore call "
1123 			"belongs to the %s IRQ kfunc family. Save and restore operations "
1124 			"must use the same family.",
1125 			flag_kfunc, used_kfunc);
1126 		bpf_diag_irq(env, env->insn_idx, "IRQ flag restore mismatch", reason,
1127 			     "Restore the flag with the matching IRQ restore kfunc for the save "
1128 			     "operation that created it.",
1129 			     bpf_diag_irq_depth(env->cur_state));
1130 		return -EINVAL;
1131 	}
1132 
1133 	err = release_irq_state(env, st->id);
1134 	WARN_ON_ONCE(err && err != -EACCES);
1135 	if (err) {
1136 		int insn_idx = 0;
1137 
1138 		for (int i = 0; i < env->cur_state->acquired_refs; i++) {
1139 			if (env->cur_state->refs[i].id == env->cur_state->active_irq_id) {
1140 				insn_idx = env->cur_state->refs[i].insn_idx;
1141 				break;
1142 			}
1143 		}
1144 
1145 		verbose(env, "cannot restore irq state out of order, expected id=%d acquired at insn_idx=%d\n",
1146 			env->cur_state->active_irq_id, insn_idx);
1147 		bpf_diag_irq(env, env->insn_idx, "IRQ flag restore out of order",
1148 			     "IRQ-disabled regions must be restored in last-in, first-out order, "
1149 			     "but this restore does not match the currently active IRQ flag.",
1150 			     "Restore nested IRQ flags in the reverse order they were saved.",
1151 			     bpf_diag_irq_depth(env->cur_state));
1152 		return err;
1153 	}
1154 
1155 	bpf_mark_reg_not_init(env, st);
1156 
1157 	for (i = 0; i < BPF_REG_SIZE; i++)
1158 		slot->slot_type[i] = STACK_INVALID;
1159 
1160 	mark_stack_slot_scratched(env, spi);
1161 	return 0;
1162 }
1163 
1164 static bool is_irq_flag_reg_valid_uninit(struct bpf_verifier_env *env, struct bpf_reg_state *reg)
1165 {
1166 	struct bpf_func_state *state = bpf_func(env, reg);
1167 	struct bpf_stack_state *slot;
1168 	int spi, i;
1169 
1170 	/* For -ERANGE (i.e. spi not falling into allocated stack slots), we
1171 	 * will do check_mem_access to check and update stack bounds later, so
1172 	 * return true for that case.
1173 	 */
1174 	spi = irq_flag_get_spi(env, reg);
1175 	if (spi == -ERANGE)
1176 		return true;
1177 	if (spi < 0)
1178 		return false;
1179 
1180 	slot = &state->stack[spi];
1181 
1182 	for (i = 0; i < BPF_REG_SIZE; i++)
1183 		if (slot->slot_type[i] == STACK_IRQ_FLAG)
1184 			return false;
1185 	return true;
1186 }
1187 
1188 static int is_irq_flag_reg_valid_init(struct bpf_verifier_env *env, struct bpf_reg_state *reg)
1189 {
1190 	struct bpf_func_state *state = bpf_func(env, reg);
1191 	struct bpf_stack_state *slot;
1192 	struct bpf_reg_state *st;
1193 	int spi, i;
1194 
1195 	spi = irq_flag_get_spi(env, reg);
1196 	if (spi < 0)
1197 		return -EINVAL;
1198 
1199 	slot = &state->stack[spi];
1200 	st = &slot->spilled_ptr;
1201 
1202 	if (!st->id)
1203 		return -EINVAL;
1204 
1205 	for (i = 0; i < BPF_REG_SIZE; i++)
1206 		if (slot->slot_type[i] != STACK_IRQ_FLAG)
1207 			return -EINVAL;
1208 	return 0;
1209 }
1210 
1211 /* Check if given stack slot is "special":
1212  *   - spilled register state (STACK_SPILL);
1213  *   - dynptr state (STACK_DYNPTR);
1214  *   - iter state (STACK_ITER).
1215  *   - irq flag state (STACK_IRQ_FLAG)
1216  */
1217 static bool is_stack_slot_special(const struct bpf_stack_state *stack)
1218 {
1219 	enum bpf_stack_slot_type type = stack->slot_type[BPF_REG_SIZE - 1];
1220 
1221 	switch (type) {
1222 	case STACK_SPILL:
1223 	case STACK_DYNPTR:
1224 	case STACK_ITER:
1225 	case STACK_IRQ_FLAG:
1226 		return true;
1227 	case STACK_INVALID:
1228 	case STACK_POISON:
1229 	case STACK_MISC:
1230 	case STACK_ZERO:
1231 		return false;
1232 	default:
1233 		WARN_ONCE(1, "unknown stack slot type %d\n", type);
1234 		return true;
1235 	}
1236 }
1237 
1238 /* The reg state of a pointer or a bounded scalar was saved when
1239  * it was spilled to the stack.
1240  */
1241 
1242 /*
1243  * Mark stack slot as STACK_MISC, unless it is already:
1244  * - STACK_INVALID, in which case they are equivalent.
1245  * - STACK_ZERO, in which case we preserve more precise STACK_ZERO.
1246  * - STACK_POISON, which truly forbids access to the slot.
1247  * Regardless of allow_ptr_leaks setting (i.e., privileged or unprivileged
1248  * mode), we won't promote STACK_INVALID to STACK_MISC. In privileged case it is
1249  * unnecessary as both are considered equivalent when loading data and pruning,
1250  * in case of unprivileged mode it will be incorrect to allow reads of invalid
1251  * slots.
1252  */
1253 static void mark_stack_slot_misc(struct bpf_verifier_env *env, u8 *stype)
1254 {
1255 	if (*stype == STACK_ZERO)
1256 		return;
1257 	if (*stype == STACK_INVALID || *stype == STACK_POISON)
1258 		return;
1259 	*stype = STACK_MISC;
1260 }
1261 
1262 static void scrub_spilled_slot(u8 *stype)
1263 {
1264 	if (*stype != STACK_INVALID && *stype != STACK_POISON)
1265 		*stype = STACK_MISC;
1266 }
1267 
1268 /* copy array src of length n * size bytes to dst. dst is reallocated if it's too
1269  * small to hold src. This is different from krealloc since we don't want to preserve
1270  * the contents of dst.
1271  *
1272  * Leaves dst untouched if src is NULL or length is zero. Returns NULL if memory could
1273  * not be allocated.
1274  */
1275 static void *copy_array(void *dst, const void *src, size_t n, size_t size, gfp_t flags)
1276 {
1277 	size_t alloc_bytes;
1278 	void *orig = dst;
1279 	size_t bytes;
1280 
1281 	if (ZERO_OR_NULL_PTR(src))
1282 		goto out;
1283 
1284 	if (unlikely(check_mul_overflow(n, size, &bytes)))
1285 		return NULL;
1286 
1287 	alloc_bytes = max(ksize(orig), kmalloc_size_roundup(bytes));
1288 	dst = krealloc(orig, alloc_bytes, flags);
1289 	if (!dst) {
1290 		kfree(orig);
1291 		return NULL;
1292 	}
1293 
1294 	memcpy(dst, src, bytes);
1295 out:
1296 	return dst ? dst : ZERO_SIZE_PTR;
1297 }
1298 
1299 /* resize an array from old_n items to new_n items. the array is reallocated if it's too
1300  * small to hold new_n items. new items are zeroed out if the array grows.
1301  *
1302  * Contrary to krealloc_array, does not free arr if new_n is zero.
1303  */
1304 static void *realloc_array(void *arr, size_t old_n, size_t new_n, size_t size)
1305 {
1306 	size_t alloc_size;
1307 	void *new_arr;
1308 
1309 	if (!new_n || old_n == new_n)
1310 		goto out;
1311 
1312 	alloc_size = kmalloc_size_roundup(size_mul(new_n, size));
1313 	new_arr = krealloc(arr, alloc_size, GFP_KERNEL_ACCOUNT);
1314 	if (!new_arr) {
1315 		kfree(arr);
1316 		return NULL;
1317 	}
1318 	arr = new_arr;
1319 
1320 	if (new_n > old_n)
1321 		memset(arr + old_n * size, 0, (new_n - old_n) * size);
1322 
1323 out:
1324 	return arr ? arr : ZERO_SIZE_PTR;
1325 }
1326 
1327 static int copy_reference_state(struct bpf_verifier_state *dst, const struct bpf_verifier_state *src)
1328 {
1329 	dst->refs = copy_array(dst->refs, src->refs, src->acquired_refs,
1330 			       sizeof(struct bpf_reference_state), GFP_KERNEL_ACCOUNT);
1331 	if (!dst->refs)
1332 		return -ENOMEM;
1333 
1334 	dst->acquired_refs = src->acquired_refs;
1335 	dst->active_locks = src->active_locks;
1336 	dst->active_preempt_locks = src->active_preempt_locks;
1337 	dst->active_rcu_locks = src->active_rcu_locks;
1338 	dst->active_irq_id = src->active_irq_id;
1339 	dst->active_lock_id = src->active_lock_id;
1340 	dst->active_lock_ptr = src->active_lock_ptr;
1341 	return 0;
1342 }
1343 
1344 static int copy_stack_state(struct bpf_func_state *dst, const struct bpf_func_state *src)
1345 {
1346 	size_t n = src->allocated_stack / BPF_REG_SIZE;
1347 
1348 	dst->stack = copy_array(dst->stack, src->stack, n, sizeof(struct bpf_stack_state),
1349 				GFP_KERNEL_ACCOUNT);
1350 	if (!dst->stack)
1351 		return -ENOMEM;
1352 
1353 	dst->allocated_stack = src->allocated_stack;
1354 
1355 	/* copy stack args state */
1356 	n = src->out_stack_arg_cnt;
1357 	if (n) {
1358 		dst->stack_arg_regs = copy_array(dst->stack_arg_regs, src->stack_arg_regs, n,
1359 						 sizeof(struct bpf_reg_state),
1360 						 GFP_KERNEL_ACCOUNT);
1361 		if (!dst->stack_arg_regs)
1362 			return -ENOMEM;
1363 	}
1364 
1365 	dst->out_stack_arg_cnt = src->out_stack_arg_cnt;
1366 	return 0;
1367 }
1368 
1369 static int resize_reference_state(struct bpf_verifier_state *state, size_t n)
1370 {
1371 	state->refs = realloc_array(state->refs, state->acquired_refs, n,
1372 				    sizeof(struct bpf_reference_state));
1373 	if (!state->refs)
1374 		return -ENOMEM;
1375 
1376 	state->acquired_refs = n;
1377 	return 0;
1378 }
1379 
1380 /* Possibly update state->allocated_stack to be at least size bytes. Also
1381  * possibly update the function's high-water mark in its bpf_subprog_info.
1382  */
1383 static int grow_stack_state(struct bpf_verifier_env *env, struct bpf_func_state *state, int size)
1384 {
1385 	size_t old_n = state->allocated_stack / BPF_REG_SIZE, n;
1386 
1387 	/* The stack size is always a multiple of BPF_REG_SIZE. */
1388 	size = round_up(size, BPF_REG_SIZE);
1389 	n = size / BPF_REG_SIZE;
1390 
1391 	if (old_n >= n)
1392 		return 0;
1393 
1394 	state->stack = realloc_array(state->stack, old_n, n, sizeof(struct bpf_stack_state));
1395 	if (!state->stack)
1396 		return -ENOMEM;
1397 
1398 	state->allocated_stack = size;
1399 
1400 	/* update known max for given subprogram */
1401 	if (env->subprog_info[state->subprogno].stack_depth < size)
1402 		env->subprog_info[state->subprogno].stack_depth = size;
1403 
1404 	return 0;
1405 }
1406 
1407 static int grow_stack_arg_slots(struct bpf_verifier_env *env,
1408 				struct bpf_func_state *state, int cnt)
1409 {
1410 	size_t old_n = state->out_stack_arg_cnt;
1411 
1412 	if (old_n >= cnt)
1413 		return 0;
1414 
1415 	state->stack_arg_regs = realloc_array(state->stack_arg_regs, old_n, cnt,
1416 					      sizeof(struct bpf_reg_state));
1417 	if (!state->stack_arg_regs)
1418 		return -ENOMEM;
1419 
1420 	state->out_stack_arg_cnt = cnt;
1421 	return 0;
1422 }
1423 
1424 /* Acquire a pointer id from the env and update the state->refs to include
1425  * this new pointer reference.
1426  * On success, returns a valid pointer id to associate with the register
1427  * On failure, returns a negative errno.
1428  */
1429 static struct bpf_reference_state *acquire_reference_state(struct bpf_verifier_env *env, int insn_idx)
1430 {
1431 	struct bpf_verifier_state *state = env->cur_state;
1432 	int new_ofs = state->acquired_refs;
1433 	int err;
1434 
1435 	err = resize_reference_state(state, state->acquired_refs + 1);
1436 	if (err)
1437 		return NULL;
1438 	state->refs[new_ofs].insn_idx = insn_idx;
1439 
1440 	return &state->refs[new_ofs];
1441 }
1442 
1443 static int acquire_reference(struct bpf_verifier_env *env, int insn_idx, int parent_id)
1444 {
1445 	struct bpf_reference_state *s;
1446 
1447 	s = acquire_reference_state(env, insn_idx);
1448 	if (!s)
1449 		return -ENOMEM;
1450 	s->type = REF_TYPE_PTR;
1451 	s->id = ++env->id_gen;
1452 	s->parent_id = parent_id;
1453 	bpf_diag_record_ref_acquire(env, insn_idx, s->id);
1454 	return s->id;
1455 }
1456 
1457 static int acquire_lock_state(struct bpf_verifier_env *env, int insn_idx, enum ref_state_type type,
1458 			      int id, void *ptr)
1459 {
1460 	struct bpf_verifier_state *state = env->cur_state;
1461 	struct bpf_reference_state *s;
1462 
1463 	s = acquire_reference_state(env, insn_idx);
1464 	if (!s)
1465 		return -ENOMEM;
1466 	s->type = type;
1467 	s->id = id;
1468 	s->ptr = ptr;
1469 
1470 	state->active_locks++;
1471 	state->active_lock_id = id;
1472 	state->active_lock_ptr = ptr;
1473 	bpf_diag_record_context(env, insn_idx, BPF_DIAG_CONTEXT_LOCK, true,
1474 				state->active_locks);
1475 	return 0;
1476 }
1477 
1478 static int acquire_irq_state(struct bpf_verifier_env *env, int insn_idx)
1479 {
1480 	struct bpf_verifier_state *state = env->cur_state;
1481 	struct bpf_reference_state *s;
1482 
1483 	s = acquire_reference_state(env, insn_idx);
1484 	if (!s)
1485 		return -ENOMEM;
1486 	s->type = REF_TYPE_IRQ;
1487 	s->id = ++env->id_gen;
1488 
1489 	state->active_irq_id = s->id;
1490 	bpf_diag_record_context(env, insn_idx, BPF_DIAG_CONTEXT_IRQ, true,
1491 				bpf_diag_irq_depth(state));
1492 	return s->id;
1493 }
1494 
1495 static void release_reference_state(struct bpf_verifier_state *state, int idx)
1496 {
1497 	int last_idx;
1498 	size_t rem;
1499 
1500 	/* IRQ state requires the relative ordering of elements remaining the
1501 	 * same, since it relies on the refs array to behave as a stack, so that
1502 	 * it can detect out-of-order IRQ restore. Hence use memmove to shift
1503 	 * the array instead of swapping the final element into the deleted idx.
1504 	 */
1505 	last_idx = state->acquired_refs - 1;
1506 	rem = state->acquired_refs - idx - 1;
1507 	if (last_idx && idx != last_idx)
1508 		memmove(&state->refs[idx], &state->refs[idx + 1], sizeof(*state->refs) * rem);
1509 	memset(&state->refs[last_idx], 0, sizeof(*state->refs));
1510 	state->acquired_refs--;
1511 	return;
1512 }
1513 
1514 static bool find_reference_state(struct bpf_verifier_state *state, int id)
1515 {
1516 	int i;
1517 
1518 	for (i = 0; i < state->acquired_refs; i++) {
1519 		if (state->refs[i].type != REF_TYPE_PTR)
1520 			continue;
1521 		if (state->refs[i].id == id)
1522 			return true;
1523 	}
1524 
1525 	return false;
1526 }
1527 
1528 static bool reg_is_referenced(struct bpf_verifier_env *env, const struct bpf_reg_state *reg)
1529 {
1530 	return find_reference_state(env->cur_state, reg->id);
1531 }
1532 
1533 static int release_lock_state(struct bpf_verifier_env *env, int type, int id, void *ptr)
1534 {
1535 	struct bpf_verifier_state *state = env->cur_state;
1536 	void *prev_ptr = NULL;
1537 	u32 prev_id = 0;
1538 	int i;
1539 
1540 	for (i = 0; i < state->acquired_refs; i++) {
1541 		if (state->refs[i].type == type && state->refs[i].id == id &&
1542 		    state->refs[i].ptr == ptr) {
1543 			release_reference_state(state, i);
1544 			state->active_locks--;
1545 			/* Reassign active lock (id, ptr). */
1546 			state->active_lock_id = prev_id;
1547 			state->active_lock_ptr = prev_ptr;
1548 			bpf_diag_record_context(env, env->insn_idx, BPF_DIAG_CONTEXT_LOCK,
1549 						false, state->active_locks);
1550 			return 0;
1551 		}
1552 		if (state->refs[i].type & REF_TYPE_LOCK_MASK) {
1553 			prev_id = state->refs[i].id;
1554 			prev_ptr = state->refs[i].ptr;
1555 		}
1556 	}
1557 	return -EINVAL;
1558 }
1559 
1560 static int release_irq_state(struct bpf_verifier_env *env, int id)
1561 {
1562 	struct bpf_verifier_state *state = env->cur_state;
1563 	u32 prev_id = 0;
1564 	int i;
1565 
1566 	if (id != state->active_irq_id)
1567 		return -EACCES;
1568 
1569 	for (i = 0; i < state->acquired_refs; i++) {
1570 		if (state->refs[i].type != REF_TYPE_IRQ)
1571 			continue;
1572 		if (state->refs[i].id == id) {
1573 			release_reference_state(state, i);
1574 			state->active_irq_id = prev_id;
1575 			bpf_diag_record_context(env, env->insn_idx, BPF_DIAG_CONTEXT_IRQ,
1576 						false, bpf_diag_irq_depth(state));
1577 			return 0;
1578 		} else {
1579 			prev_id = state->refs[i].id;
1580 		}
1581 	}
1582 	return -EINVAL;
1583 }
1584 
1585 static struct bpf_reference_state *find_lock_state(struct bpf_verifier_state *state, enum ref_state_type type,
1586 						   int id, void *ptr)
1587 {
1588 	int i;
1589 
1590 	for (i = 0; i < state->acquired_refs; i++) {
1591 		struct bpf_reference_state *s = &state->refs[i];
1592 
1593 		if (!(s->type & type))
1594 			continue;
1595 
1596 		if (s->id == id && s->ptr == ptr)
1597 			return s;
1598 	}
1599 	return NULL;
1600 }
1601 
1602 static void free_func_state(struct bpf_func_state *state)
1603 {
1604 	if (!state)
1605 		return;
1606 	kfree(state->stack_arg_regs);
1607 	kfree(state->stack);
1608 	kfree(state);
1609 }
1610 
1611 void bpf_clear_jmp_history(struct bpf_verifier_state *state)
1612 {
1613 	kfree(state->jmp_history);
1614 	state->jmp_history = NULL;
1615 	state->jmp_history_cnt = 0;
1616 }
1617 
1618 void bpf_free_verifier_state(struct bpf_verifier_state *state,
1619 			    bool free_self)
1620 {
1621 	int i;
1622 
1623 	for (i = 0; i <= state->curframe; i++) {
1624 		free_func_state(state->frame[i]);
1625 		state->frame[i] = NULL;
1626 	}
1627 	kfree(state->refs);
1628 	bpf_clear_jmp_history(state);
1629 	if (free_self)
1630 		kfree(state);
1631 }
1632 
1633 /* copy verifier state from src to dst growing dst stack space
1634  * when necessary to accommodate larger src stack
1635  */
1636 static int copy_func_state(struct bpf_func_state *dst,
1637 			   const struct bpf_func_state *src)
1638 {
1639 	memcpy(dst, src, offsetof(struct bpf_func_state, stack));
1640 	/* Instruction accounting is path-local, not part of verifier state. */
1641 	dst->insns_subtotal = 0;
1642 	return copy_stack_state(dst, src);
1643 }
1644 
1645 int bpf_copy_verifier_state(struct bpf_verifier_state *dst_state,
1646 			   const struct bpf_verifier_state *src)
1647 {
1648 	struct bpf_func_state *dst;
1649 	int i, err;
1650 
1651 	dst_state->jmp_history = copy_array(dst_state->jmp_history, src->jmp_history,
1652 					  src->jmp_history_cnt, sizeof(*dst_state->jmp_history),
1653 					  GFP_KERNEL_ACCOUNT);
1654 	if (!dst_state->jmp_history)
1655 		return -ENOMEM;
1656 	dst_state->jmp_history_cnt = src->jmp_history_cnt;
1657 
1658 	/* if dst has more stack frames then src frame, free them, this is also
1659 	 * necessary in case of exceptional exits using bpf_throw.
1660 	 */
1661 	for (i = src->curframe + 1; i <= dst_state->curframe; i++) {
1662 		free_func_state(dst_state->frame[i]);
1663 		dst_state->frame[i] = NULL;
1664 	}
1665 	err = copy_reference_state(dst_state, src);
1666 	if (err)
1667 		return err;
1668 	dst_state->speculative = src->speculative;
1669 	dst_state->in_sleepable = src->in_sleepable;
1670 	dst_state->curframe = src->curframe;
1671 	dst_state->branches = src->branches;
1672 	dst_state->parent = src->parent;
1673 	dst_state->first_insn_idx = src->first_insn_idx;
1674 	dst_state->last_insn_idx = src->last_insn_idx;
1675 	dst_state->dfs_depth = src->dfs_depth;
1676 	dst_state->callback_unroll_depth = src->callback_unroll_depth;
1677 	dst_state->may_goto_depth = src->may_goto_depth;
1678 	dst_state->equal_state = src->equal_state;
1679 	for (i = 0; i <= src->curframe; i++) {
1680 		dst = dst_state->frame[i];
1681 		if (!dst) {
1682 			dst = kzalloc_obj(*dst, GFP_KERNEL_ACCOUNT);
1683 			if (!dst)
1684 				return -ENOMEM;
1685 			dst_state->frame[i] = dst;
1686 		}
1687 		err = copy_func_state(dst, src->frame[i]);
1688 		if (err)
1689 			return err;
1690 	}
1691 	return 0;
1692 }
1693 
1694 static u32 state_htab_size(struct bpf_verifier_env *env)
1695 {
1696 	return env->prog->len;
1697 }
1698 
1699 struct list_head *bpf_explored_state(struct bpf_verifier_env *env, int idx)
1700 {
1701 	struct bpf_verifier_state *cur = env->cur_state;
1702 	struct bpf_func_state *state = cur->frame[cur->curframe];
1703 
1704 	return &env->explored_states[(idx ^ state->callsite) % state_htab_size(env)];
1705 }
1706 
1707 static bool same_callsites(struct bpf_verifier_state *a, struct bpf_verifier_state *b)
1708 {
1709 	int fr;
1710 
1711 	if (a->curframe != b->curframe)
1712 		return false;
1713 
1714 	for (fr = a->curframe; fr >= 0; fr--)
1715 		if (a->frame[fr]->callsite != b->frame[fr]->callsite)
1716 			return false;
1717 
1718 	return true;
1719 }
1720 
1721 void bpf_free_backedges(struct bpf_scc_visit *visit)
1722 {
1723 	struct bpf_scc_backedge *backedge, *next;
1724 
1725 	for (backedge = visit->backedges; backedge; backedge = next) {
1726 		bpf_free_verifier_state(&backedge->state, false);
1727 		next = backedge->next;
1728 		kfree(backedge);
1729 	}
1730 	visit->backedges = NULL;
1731 }
1732 
1733 static int pop_stack(struct bpf_verifier_env *env, int *prev_insn_idx,
1734 		     int *insn_idx, bool pop_log)
1735 {
1736 	struct bpf_verifier_state *cur = env->cur_state;
1737 	struct bpf_verifier_stack_elem *elem, *head = env->head;
1738 	int err;
1739 
1740 	if (env->head == NULL)
1741 		return -ENOENT;
1742 
1743 	if (cur) {
1744 		err = bpf_copy_verifier_state(cur, &head->st);
1745 		if (err)
1746 			return err;
1747 		bpf_diag_event_log_restore(env, head->diag_log_pos);
1748 	}
1749 	if (pop_log)
1750 		bpf_vlog_reset(&env->log, head->log_pos);
1751 	if (insn_idx)
1752 		*insn_idx = head->insn_idx;
1753 	if (prev_insn_idx)
1754 		*prev_insn_idx = head->prev_insn_idx;
1755 	elem = head->next;
1756 	bpf_free_verifier_state(&head->st, false);
1757 	kfree(head);
1758 	env->head = elem;
1759 	env->stack_size--;
1760 	return 0;
1761 }
1762 
1763 static bool error_recoverable_with_nospec(int err)
1764 {
1765 	/* Should only return true for non-fatal errors that are allowed to
1766 	 * occur during speculative verification. For these we can insert a
1767 	 * nospec and the program might still be accepted. Do not include
1768 	 * something like ENOMEM because it is likely to re-occur for the next
1769 	 * architectural path once it has been recovered-from in all speculative
1770 	 * paths.
1771 	 */
1772 	return err == -EPERM || err == -EACCES || err == -EINVAL;
1773 }
1774 
1775 static struct bpf_verifier_state *push_stack(struct bpf_verifier_env *env,
1776 					     int insn_idx, int prev_insn_idx,
1777 					     bool speculative)
1778 {
1779 	struct bpf_verifier_state *cur = env->cur_state;
1780 	struct bpf_verifier_stack_elem *elem;
1781 	int err;
1782 
1783 	elem = kzalloc_obj(struct bpf_verifier_stack_elem, GFP_KERNEL_ACCOUNT);
1784 	if (!elem)
1785 		return ERR_PTR(-ENOMEM);
1786 
1787 	elem->insn_idx = insn_idx;
1788 	elem->prev_insn_idx = prev_insn_idx;
1789 	elem->next = env->head;
1790 	elem->log_pos = env->log.end_pos;
1791 	elem->diag_log_pos = bpf_diag_event_log_save(env);
1792 	env->head = elem;
1793 	env->stack_size++;
1794 	err = bpf_copy_verifier_state(&elem->st, cur);
1795 	if (err)
1796 		return ERR_PTR(-ENOMEM);
1797 	elem->st.speculative |= speculative;
1798 	if (env->stack_size > BPF_COMPLEXITY_LIMIT_JMP_SEQ) {
1799 		verbose(env, "The sequence of %d jumps is too complex.\n",
1800 			env->stack_size);
1801 		return ERR_PTR(-E2BIG);
1802 	}
1803 	if (elem->st.parent) {
1804 		++elem->st.parent->branches;
1805 		/* WARN_ON(branches > 2) technically makes sense here,
1806 		 * but
1807 		 * 1. speculative states will bump 'branches' for non-branch
1808 		 * instructions
1809 		 * 2. is_state_visited() heuristics may decide not to create
1810 		 * a new state for a sequence of branches and all such current
1811 		 * and cloned states will be pointing to a single parent state
1812 		 * which might have large 'branches' count.
1813 		 */
1814 	}
1815 	return &elem->st;
1816 }
1817 
1818 static const char *reg_arg_name(struct bpf_verifier_env *env, argno_t argno)
1819 {
1820 	char *buf = env->tmp_arg_name;
1821 	int len = sizeof(env->tmp_arg_name);
1822 	int arg, regno = reg_from_argno(argno);
1823 
1824 	if (regno >= 0) {
1825 		snprintf(buf, len, "R%d", regno);
1826 	} else {
1827 		arg = arg_from_argno(argno);
1828 		snprintf(buf, len, "*(R11-%u)", (arg - MAX_BPF_FUNC_REG_ARGS) * BPF_REG_SIZE);
1829 	}
1830 
1831 	return buf;
1832 }
1833 
1834 static const int caller_saved[CALLER_SAVED_REGS] = {
1835 	BPF_REG_0, BPF_REG_1, BPF_REG_2, BPF_REG_3, BPF_REG_4, BPF_REG_5
1836 };
1837 
1838 static void bpf_diag_record_caller_saved(struct bpf_verifier_env *env,
1839 					 struct bpf_reg_state *regs)
1840 {
1841 	int i;
1842 
1843 	for (i = 1; i < CALLER_SAVED_REGS; i++) {
1844 		bpf_diag_record_scrub(env, &regs[caller_saved[i]],
1845 				      BPF_DIAG_MOD_CALLER_SAVED);
1846 	}
1847 }
1848 
1849 /* This helper doesn't clear reg->id */
1850 static void ___mark_reg_known(struct bpf_reg_state *reg, u64 imm)
1851 {
1852 	reg->var_off = tnum_const(imm);
1853 	reg->r64 = cnum64_from_urange(imm, imm);
1854 	reg->r32 = cnum32_from_urange((u32)imm, (u32)imm);
1855 }
1856 
1857 /* Mark the unknown part of a register (variable offset or scalar value) as
1858  * known to have the value @imm.
1859  */
1860 static void __mark_reg_known(struct bpf_reg_state *reg, u64 imm)
1861 {
1862 	/* Clear off and union(map_ptr, range) */
1863 	memset(((u8 *)reg) + sizeof(reg->type), 0,
1864 	       offsetof(struct bpf_reg_state, var_off) - sizeof(reg->type));
1865 	reg->id = 0;
1866 	reg->parent_id = 0;
1867 	reg->map_uid = 0;
1868 	___mark_reg_known(reg, imm);
1869 }
1870 
1871 static void __mark_reg32_known(struct bpf_reg_state *reg, u64 imm)
1872 {
1873 	reg->var_off = tnum_const_subreg(reg->var_off, imm);
1874 	reg->r32 = cnum32_from_urange((u32)imm, (u32)imm);
1875 }
1876 
1877 /* Mark the 'variable offset' part of a register as zero.  This should be
1878  * used only on registers holding a pointer type.
1879  */
1880 static void __mark_reg_known_zero(struct bpf_reg_state *reg)
1881 {
1882 	__mark_reg_known(reg, 0);
1883 }
1884 
1885 static void __mark_reg_const_zero(const struct bpf_verifier_env *env, struct bpf_reg_state *reg)
1886 {
1887 	__mark_reg_known(reg, 0);
1888 	reg->type = SCALAR_VALUE;
1889 	/* all scalars are assumed imprecise initially (unless unprivileged,
1890 	 * in which case everything is forced to be precise)
1891 	 */
1892 	reg->precise = !env->bpf_capable;
1893 }
1894 
1895 static void mark_reg_known_zero(struct bpf_verifier_env *env,
1896 				struct bpf_reg_state *regs, u32 regno)
1897 {
1898 	__mark_reg_known_zero(regs + regno);
1899 }
1900 
1901 static void __mark_dynptr_reg(struct bpf_reg_state *reg, enum bpf_dynptr_type type,
1902 			      bool first_slot, int id, int parent_id)
1903 {
1904 	/* reg->type has no meaning for STACK_DYNPTR, but when we set reg for
1905 	 * callback arguments, it does need to be CONST_PTR_TO_DYNPTR, so simply
1906 	 * set it unconditionally as it is ignored for STACK_DYNPTR anyway.
1907 	 */
1908 	__mark_reg_known_zero(reg);
1909 	reg->type = CONST_PTR_TO_DYNPTR;
1910 	/* Give each dynptr a unique id to uniquely associate slices to it. */
1911 	reg->id = id;
1912 	reg->parent_id = parent_id;
1913 	reg->dynptr.type = type;
1914 	reg->dynptr.first_slot = first_slot;
1915 }
1916 
1917 /*
1918  * Refine the return type of the bpf_map_lookup_elem() for special map types:
1919  * map-in-map, xskmap, sockmap and sockhash.
1920  */
1921 static void refine_map_lookup_value(struct bpf_reg_state *reg)
1922 {
1923 	enum bpf_type_flag maybe_null = reg->type & PTR_MAYBE_NULL;
1924 	const struct bpf_map *map = reg->map_ptr;
1925 
1926 	if (map->inner_map_meta) {
1927 		reg->type = CONST_PTR_TO_MAP | maybe_null;
1928 		reg->map_ptr = map->inner_map_meta;
1929 		/*
1930 		 * transfer reg's id which is unique for every map_lookup_elem
1931 		 * as UID of the inner map.
1932 		 */
1933 		reg->map_uid = reg->id;
1934 	} else if (map->map_type == BPF_MAP_TYPE_XSKMAP) {
1935 		reg->type = PTR_TO_XDP_SOCK | maybe_null;
1936 		reg->map_uid = 0;
1937 	} else if (map->map_type == BPF_MAP_TYPE_SOCKMAP ||
1938 		   map->map_type == BPF_MAP_TYPE_SOCKHASH) {
1939 		reg->type = PTR_TO_SOCKET | maybe_null;
1940 		reg->map_uid = 0;
1941 	}
1942 }
1943 
1944 static void mark_ptr_not_null_reg(struct bpf_reg_state *reg)
1945 {
1946 	reg->type &= ~PTR_MAYBE_NULL;
1947 }
1948 
1949 static void mark_reg_graph_node(struct bpf_reg_state *regs, u32 regno,
1950 				struct btf_field_graph_root *ds_head)
1951 {
1952 	__mark_reg_known(&regs[regno], ds_head->node_offset);
1953 	regs[regno].type = PTR_TO_BTF_ID | MEM_ALLOC;
1954 	regs[regno].btf = ds_head->btf;
1955 	regs[regno].btf_id = ds_head->value_btf_id;
1956 }
1957 
1958 static bool reg_is_pkt_pointer(const struct bpf_reg_state *reg)
1959 {
1960 	return type_is_pkt_pointer(reg->type);
1961 }
1962 
1963 static bool reg_is_pkt_pointer_any(const struct bpf_reg_state *reg)
1964 {
1965 	return reg_is_pkt_pointer(reg) ||
1966 	       reg->type == PTR_TO_PACKET_END;
1967 }
1968 
1969 static bool reg_is_dynptr_slice_pkt(const struct bpf_reg_state *reg)
1970 {
1971 	return base_type(reg->type) == PTR_TO_MEM &&
1972 	       (reg->type &
1973 		(DYNPTR_TYPE_SKB | DYNPTR_TYPE_XDP | DYNPTR_TYPE_SKB_META));
1974 }
1975 
1976 /* Unmodified PTR_TO_PACKET[_META,_END] register from ctx access. */
1977 static bool reg_is_init_pkt_pointer(const struct bpf_reg_state *reg,
1978 				    enum bpf_reg_type which)
1979 {
1980 	/* The register can already have a range from prior markings.
1981 	 * This is fine as long as it hasn't been advanced from its
1982 	 * origin.
1983 	 */
1984 	return reg->type == which &&
1985 	       reg->id == 0 &&
1986 	       tnum_equals_const(reg->var_off, 0);
1987 }
1988 
1989 static void __mark_reg32_unbounded(struct bpf_reg_state *reg)
1990 {
1991 	reg->r32 = CNUM32_UNBOUNDED;
1992 }
1993 
1994 static void __mark_reg64_unbounded(struct bpf_reg_state *reg)
1995 {
1996 	reg->r64 = CNUM64_UNBOUNDED;
1997 }
1998 
1999 /* Reset the min/max bounds of a register */
2000 static void __mark_reg_unbounded(struct bpf_reg_state *reg)
2001 {
2002 	__mark_reg64_unbounded(reg);
2003 	__mark_reg32_unbounded(reg);
2004 }
2005 
2006 static void reset_reg64_and_tnum(struct bpf_reg_state *reg)
2007 {
2008 	__mark_reg64_unbounded(reg);
2009 	reg->var_off = tnum_unknown;
2010 }
2011 
2012 static void reset_reg32_and_tnum(struct bpf_reg_state *reg)
2013 {
2014 	__mark_reg32_unbounded(reg);
2015 	reg->var_off = tnum_unknown;
2016 }
2017 
2018 static struct cnum32 cnum32_from_tnum(struct tnum tnum)
2019 {
2020 	tnum = tnum_subreg(tnum);
2021 	if ((tnum.mask & S32_MIN) || (tnum.value & S32_MIN))
2022 		/* min signed is max(sign bit) | min(other bits) */
2023 		/* max signed is min(sign bit) | max(other bits) */
2024 		return cnum32_from_srange(tnum.value | (tnum.mask & S32_MIN),
2025 					  tnum.value | (tnum.mask & S32_MAX));
2026 	else
2027 		return cnum32_from_urange(tnum.value, (tnum.value | tnum.mask));
2028 }
2029 
2030 static struct cnum64 cnum64_from_tnum(struct tnum tnum)
2031 {
2032 	if ((tnum.mask & S64_MIN) || (tnum.value & S64_MIN))
2033 		/* min signed is max(sign bit) | min(other bits) */
2034 		/* max signed is min(sign bit) | max(other bits) */
2035 		return cnum64_from_srange(tnum.value | (tnum.mask & S64_MIN),
2036 					  tnum.value | (tnum.mask & S64_MAX));
2037 	else
2038 		return cnum64_from_urange(tnum.value, (tnum.value | tnum.mask));
2039 }
2040 
2041 static void __update_reg32_bounds(struct bpf_reg_state *reg)
2042 {
2043 	cnum32_intersect_with(&reg->r32, cnum32_from_tnum(reg->var_off));
2044 }
2045 
2046 static void __update_reg64_bounds(struct bpf_reg_state *reg)
2047 {
2048 	u64 tnum_next, tmax;
2049 	bool umin_in_tnum;
2050 
2051 	cnum64_intersect_with(&reg->r64, cnum64_from_tnum(reg->var_off));
2052 
2053 	/* Check if u64 and tnum overlap in a single value */
2054 	tnum_next = tnum_step(reg->var_off, reg_umin(reg));
2055 	umin_in_tnum = (reg_umin(reg) & ~reg->var_off.mask) == reg->var_off.value;
2056 	tmax = reg->var_off.value | reg->var_off.mask;
2057 	if (umin_in_tnum && tnum_next > reg_umax(reg)) {
2058 		/* The u64 range and the tnum only overlap in umin.
2059 		 * u64:  ---[xxxxxx]-----
2060 		 * tnum: --xx----------x-
2061 		 */
2062 		___mark_reg_known(reg, reg_umin(reg));
2063 	} else if (!umin_in_tnum && tnum_next == tmax) {
2064 		/* The u64 range and the tnum only overlap in the maximum value
2065 		 * represented by the tnum, called tmax.
2066 		 * u64:  ---[xxxxxx]-----
2067 		 * tnum: xx-----x--------
2068 		 */
2069 		___mark_reg_known(reg, tmax);
2070 	} else if (!umin_in_tnum && tnum_next <= reg_umax(reg) &&
2071 		   tnum_step(reg->var_off, tnum_next) > reg_umax(reg)) {
2072 		/* The u64 range and the tnum only overlap in between umin
2073 		 * (excluded) and umax.
2074 		 * u64:  ---[xxxxxx]-----
2075 		 * tnum: xx----x-------x-
2076 		 */
2077 		___mark_reg_known(reg, tnum_next);
2078 	}
2079 }
2080 
2081 static void __update_reg_bounds(struct bpf_reg_state *reg)
2082 {
2083 	__update_reg32_bounds(reg);
2084 	__update_reg64_bounds(reg);
2085 }
2086 
2087 static void deduce_bounds_32_from_64(struct bpf_reg_state *reg)
2088 {
2089 	cnum32_intersect_with(&reg->r32, cnum32_from_cnum64(reg->r64));
2090 }
2091 
2092 static void deduce_bounds_64_from_32(struct bpf_reg_state *reg)
2093 {
2094 	reg->r64 = cnum64_cnum32_intersect(reg->r64, reg->r32);
2095 }
2096 
2097 static void __reg_deduce_bounds(struct bpf_reg_state *reg)
2098 {
2099 	deduce_bounds_32_from_64(reg);
2100 	deduce_bounds_64_from_32(reg);
2101 }
2102 
2103 /* Attempts to improve var_off based on unsigned min/max information */
2104 static void __reg_bound_offset(struct bpf_reg_state *reg)
2105 {
2106 	struct tnum var64_off = tnum_intersect(reg->var_off,
2107 					       tnum_range(reg_umin(reg),
2108 							  reg_umax(reg)));
2109 	struct tnum var32_off = tnum_intersect(tnum_subreg(var64_off),
2110 					       tnum_range(reg_u32_min(reg),
2111 							  reg_u32_max(reg)));
2112 
2113 	reg->var_off = tnum_or(tnum_clear_subreg(var64_off), var32_off);
2114 }
2115 
2116 static bool range_bounds_violation(struct bpf_reg_state *reg);
2117 
2118 static void reg_bounds_sync(struct bpf_reg_state *reg)
2119 {
2120 	/* If the input reg_state is invalid, we can exit early */
2121 	if (range_bounds_violation(reg))
2122 		return;
2123 	/* We might have learned new bounds from the var_off. */
2124 	__update_reg_bounds(reg);
2125 	/* We might have learned something about the sign bit. */
2126 	__reg_deduce_bounds(reg);
2127 	__reg_deduce_bounds(reg);
2128 	/* We might have learned some bits from the bounds. */
2129 	__reg_bound_offset(reg);
2130 	/* Intersecting with the old var_off might have improved our bounds
2131 	 * slightly, e.g. if umax was 0x7f...f and var_off was (0; 0xf...fc),
2132 	 * then new var_off is (0; 0x7f...fc) which improves our umax.
2133 	 */
2134 	__update_reg_bounds(reg);
2135 }
2136 
2137 static bool const_tnum_range_mismatch(struct bpf_reg_state *reg)
2138 {
2139 	if (!tnum_is_const(reg->var_off))
2140 		return false;
2141 
2142 	return !cnum64_is_const(reg->r64) || reg->r64.base != reg->var_off.value;
2143 }
2144 
2145 static bool const_tnum_range_mismatch_32(struct bpf_reg_state *reg)
2146 {
2147 	if (!tnum_subreg_is_const(reg->var_off))
2148 		return false;
2149 
2150 	return !cnum32_is_const(reg->r32) || reg->r32.base != tnum_subreg(reg->var_off).value;
2151 }
2152 
2153 static bool range_bounds_violation(struct bpf_reg_state *reg)
2154 {
2155 	return cnum32_is_empty(reg->r32) || cnum64_is_empty(reg->r64);
2156 }
2157 
2158 static int reg_bounds_sanity_check(struct bpf_verifier_env *env,
2159 				   struct bpf_reg_state *reg, const char *ctx)
2160 {
2161 	const char *msg;
2162 
2163 	if (range_bounds_violation(reg)) {
2164 		msg = "range bounds violation";
2165 		goto out;
2166 	}
2167 
2168 	if (const_tnum_range_mismatch(reg)) {
2169 		msg = "const tnum out of sync with range bounds";
2170 		goto out;
2171 	}
2172 
2173 	if (const_tnum_range_mismatch_32(reg)) {
2174 		msg = "const subreg tnum out of sync with range bounds";
2175 		goto out;
2176 	}
2177 
2178 	return 0;
2179 out:
2180 	verifier_bug(env, "REG INVARIANTS VIOLATION (%s): %s r64={.base=%#llx, .size=%#llx} "
2181 		     "r32={.base=%#x, .size=%#x} var_off=(%#llx, %#llx)",
2182 		     ctx, msg,
2183 		     reg->r64.base, reg->r64.size,
2184 		     reg->r32.base, reg->r32.size,
2185 		     reg->var_off.value, reg->var_off.mask);
2186 	if (env->test_reg_invariants)
2187 		return -EFAULT;
2188 	__mark_reg_unbounded(reg);
2189 	return 0;
2190 }
2191 
2192 /* Mark a register as having a completely unknown (scalar) value. */
2193 void bpf_mark_reg_unknown_imprecise(struct bpf_reg_state *reg)
2194 {
2195 	memset(reg, 0, sizeof(*reg));
2196 	reg->type = SCALAR_VALUE;
2197 	reg->var_off = tnum_unknown;
2198 	__mark_reg_unbounded(reg);
2199 }
2200 
2201 /* Mark a register as having a completely unknown (scalar) value,
2202  * initialize .precise as true when not bpf capable.
2203  */
2204 static void __mark_reg_unknown(const struct bpf_verifier_env *env,
2205 			       struct bpf_reg_state *reg)
2206 {
2207 	bpf_mark_reg_unknown_imprecise(reg);
2208 	reg->precise = !env->bpf_capable;
2209 }
2210 
2211 static void mark_reg_unknown(struct bpf_verifier_env *env,
2212 			     struct bpf_reg_state *regs, u32 regno)
2213 {
2214 	__mark_reg_unknown(env, regs + regno);
2215 }
2216 
2217 static int __mark_reg_s32_range(struct bpf_verifier_env *env,
2218 				struct bpf_reg_state *regs,
2219 				u32 regno,
2220 				s32 s32_min,
2221 				s32 s32_max)
2222 {
2223 	struct bpf_reg_state *reg = regs + regno;
2224 
2225 	reg_set_srange32(reg,
2226 			 max_t(s32, reg_s32_min(reg), s32_min),
2227 			 min_t(s32, reg_s32_max(reg), s32_max));
2228 	reg_set_srange64(reg,
2229 			 max_t(s64, reg_smin(reg), s32_min),
2230 			 min_t(s64, reg_smax(reg), s32_max));
2231 
2232 	reg_bounds_sync(reg);
2233 
2234 	return reg_bounds_sanity_check(env, reg, "s32_range");
2235 }
2236 
2237 void bpf_mark_reg_not_init(const struct bpf_verifier_env *env,
2238 			   struct bpf_reg_state *reg)
2239 {
2240 	__mark_reg_unknown(env, reg);
2241 	reg->type = NOT_INIT;
2242 }
2243 
2244 static int mark_btf_ld_reg(struct bpf_verifier_env *env,
2245 			   struct bpf_reg_state *regs, u32 regno,
2246 			   enum bpf_reg_type reg_type,
2247 			   struct btf *btf, u32 btf_id,
2248 			   enum bpf_type_flag flag)
2249 {
2250 	switch (reg_type) {
2251 	case SCALAR_VALUE:
2252 		mark_reg_unknown(env, regs, regno);
2253 		return 0;
2254 	case PTR_TO_BTF_ID:
2255 		mark_reg_known_zero(env, regs, regno);
2256 		regs[regno].type = PTR_TO_BTF_ID | flag;
2257 		regs[regno].btf = btf;
2258 		regs[regno].btf_id = btf_id;
2259 		if (type_may_be_null(flag))
2260 			regs[regno].id = ++env->id_gen;
2261 		return 0;
2262 	case PTR_TO_MEM:
2263 		mark_reg_known_zero(env, regs, regno);
2264 		regs[regno].type = PTR_TO_MEM | flag;
2265 		regs[regno].mem_size = 0;
2266 		return 0;
2267 	default:
2268 		verifier_bug(env, "unexpected reg_type %d in %s\n", reg_type, __func__);
2269 		return -EFAULT;
2270 	}
2271 }
2272 
2273 static void init_reg_state(struct bpf_verifier_env *env,
2274 			   struct bpf_func_state *state)
2275 {
2276 	struct bpf_reg_state *regs = state->regs;
2277 	int i;
2278 
2279 	for (i = 0; i < MAX_BPF_REG; i++) {
2280 		bpf_mark_reg_not_init(env, &regs[i]);
2281 	}
2282 
2283 	/* frame pointer */
2284 	regs[BPF_REG_FP].type = PTR_TO_STACK;
2285 	mark_reg_known_zero(env, regs, BPF_REG_FP);
2286 	regs[BPF_REG_FP].frameno = state->frameno;
2287 }
2288 
2289 static struct bpf_retval_range retval_range(s32 minval, s32 maxval)
2290 {
2291 	/*
2292 	 * return_32bit is set to false by default and set explicitly
2293 	 * by the caller when necessary.
2294 	 */
2295 	return (struct bpf_retval_range){ minval, maxval, false };
2296 }
2297 
2298 static void init_func_state(struct bpf_verifier_env *env,
2299 			    struct bpf_func_state *state,
2300 			    int callsite, int frameno, int subprogno)
2301 {
2302 	state->callsite = callsite;
2303 	state->frameno = frameno;
2304 	bpf_diag_init_frame(env, state);
2305 	state->subprogno = subprogno;
2306 	state->callback_ret_range = retval_range(0, 0);
2307 	init_reg_state(env, state);
2308 	mark_verifier_state_scratched(env);
2309 }
2310 
2311 /* Similar to push_stack(), but for async callbacks */
2312 static struct bpf_verifier_state *push_async_cb(struct bpf_verifier_env *env,
2313 						int insn_idx, int prev_insn_idx,
2314 						int subprog, bool is_sleepable)
2315 {
2316 	struct bpf_verifier_stack_elem *elem;
2317 	struct bpf_func_state *frame;
2318 
2319 	elem = kzalloc_obj(struct bpf_verifier_stack_elem, GFP_KERNEL_ACCOUNT);
2320 	if (!elem)
2321 		return ERR_PTR(-ENOMEM);
2322 
2323 	elem->insn_idx = insn_idx;
2324 	elem->prev_insn_idx = prev_insn_idx;
2325 	elem->next = env->head;
2326 	elem->log_pos = env->log.end_pos;
2327 	elem->diag_log_pos = bpf_diag_event_log_save(env);
2328 	env->head = elem;
2329 	env->stack_size++;
2330 	if (env->stack_size > BPF_COMPLEXITY_LIMIT_JMP_SEQ) {
2331 		verbose(env,
2332 			"The sequence of %d jumps is too complex for async cb.\n",
2333 			env->stack_size);
2334 		return ERR_PTR(-E2BIG);
2335 	}
2336 	/* Unlike push_stack() do not bpf_copy_verifier_state().
2337 	 * The caller state doesn't matter.
2338 	 * This is async callback. It starts in a fresh stack.
2339 	 * Initialize it similar to do_check_common().
2340 	 */
2341 	elem->st.branches = 1;
2342 	elem->st.in_sleepable = is_sleepable;
2343 	frame = kzalloc_obj(*frame, GFP_KERNEL_ACCOUNT);
2344 	if (!frame)
2345 		return ERR_PTR(-ENOMEM);
2346 	init_func_state(env, frame,
2347 			BPF_MAIN_FUNC /* callsite */,
2348 			0 /* frameno within this callchain */,
2349 			subprog /* subprog number within this prog */);
2350 	elem->st.frame[0] = frame;
2351 	return &elem->st;
2352 }
2353 
2354 static int cmp_subprogs(const void *a, const void *b)
2355 {
2356 	return ((struct bpf_subprog_info *)a)->start -
2357 	       ((struct bpf_subprog_info *)b)->start;
2358 }
2359 
2360 /* Find subprogram that contains instruction at 'off' */
2361 struct bpf_subprog_info *bpf_find_containing_subprog(struct bpf_verifier_env *env, int off)
2362 {
2363 	struct bpf_subprog_info *vals = env->subprog_info;
2364 	int l, r, m;
2365 
2366 	if (off >= env->prog->len || off < 0 || env->subprog_cnt == 0)
2367 		return NULL;
2368 
2369 	l = 0;
2370 	r = env->subprog_cnt - 1;
2371 	while (l < r) {
2372 		m = l + (r - l + 1) / 2;
2373 		if (vals[m].start <= off)
2374 			l = m;
2375 		else
2376 			r = m - 1;
2377 	}
2378 	return &vals[l];
2379 }
2380 
2381 /* Find subprogram that starts exactly at 'off' */
2382 int bpf_find_subprog(struct bpf_verifier_env *env, int off)
2383 {
2384 	struct bpf_subprog_info *p;
2385 
2386 	p = bpf_find_containing_subprog(env, off);
2387 	if (!p || p->start != off)
2388 		return -ENOENT;
2389 	return p - env->subprog_info;
2390 }
2391 
2392 static int add_subprog(struct bpf_verifier_env *env, int off)
2393 {
2394 	int insn_cnt = env->prog->len;
2395 	int ret;
2396 
2397 	if (off >= insn_cnt || off < 0) {
2398 		verbose(env, "call to invalid destination\n");
2399 		return -EINVAL;
2400 	}
2401 	ret = bpf_find_subprog(env, off);
2402 	if (ret >= 0)
2403 		return ret;
2404 	if (env->subprog_cnt >= BPF_MAX_SUBPROGS) {
2405 		verbose(env, "too many subprograms\n");
2406 		return -E2BIG;
2407 	}
2408 	/* determine subprog starts. The end is one before the next starts */
2409 	env->subprog_info[env->subprog_cnt++].start = off;
2410 	sort(env->subprog_info, env->subprog_cnt,
2411 	     sizeof(env->subprog_info[0]), cmp_subprogs, NULL);
2412 	return env->subprog_cnt - 1;
2413 }
2414 
2415 static int bpf_find_exception_callback_insn_off(struct bpf_verifier_env *env)
2416 {
2417 	struct bpf_prog_aux *aux = env->prog->aux;
2418 	struct btf *btf = aux->btf;
2419 	const struct btf_type *t;
2420 	u32 main_btf_id, id;
2421 	const char *name;
2422 	int ret, i;
2423 
2424 	/* Non-zero func_info_cnt implies valid btf */
2425 	if (!aux->func_info_cnt)
2426 		return 0;
2427 	main_btf_id = aux->func_info[0].type_id;
2428 
2429 	t = btf_type_by_id(btf, main_btf_id);
2430 	if (!t) {
2431 		verbose(env, "invalid btf id for main subprog in func_info\n");
2432 		return -EINVAL;
2433 	}
2434 
2435 	name = btf_find_decl_tag_value(btf, t, -1, "exception_callback:");
2436 	if (IS_ERR(name)) {
2437 		ret = PTR_ERR(name);
2438 		/* If there is no tag present, there is no exception callback */
2439 		if (ret == -ENOENT)
2440 			ret = 0;
2441 		else if (ret == -EEXIST)
2442 			verbose(env, "multiple exception callback tags for main subprog\n");
2443 		return ret;
2444 	}
2445 
2446 	ret = btf_find_by_name_kind(btf, name, BTF_KIND_FUNC);
2447 	if (ret < 0) {
2448 		verbose(env, "exception callback '%s' could not be found in BTF\n", name);
2449 		return ret;
2450 	}
2451 	id = ret;
2452 	t = btf_type_by_id(btf, id);
2453 	if (btf_func_linkage(t) != BTF_FUNC_GLOBAL) {
2454 		verbose(env, "exception callback '%s' must have global linkage\n", name);
2455 		return -EINVAL;
2456 	}
2457 	ret = 0;
2458 	for (i = 0; i < aux->func_info_cnt; i++) {
2459 		if (aux->func_info[i].type_id != id)
2460 			continue;
2461 		ret = aux->func_info[i].insn_off;
2462 		/* Further func_info and subprog checks will also happen
2463 		 * later, so assume this is the right insn_off for now.
2464 		 */
2465 		if (!ret) {
2466 			verbose(env, "invalid exception callback insn_off in func_info: 0\n");
2467 			ret = -EINVAL;
2468 		}
2469 	}
2470 	if (!ret) {
2471 		verbose(env, "exception callback type id not found in func_info\n");
2472 		ret = -EINVAL;
2473 	}
2474 	return ret;
2475 }
2476 
2477 #define MAX_KFUNC_BTFS	256
2478 
2479 struct bpf_kfunc_btf {
2480 	struct btf *btf;
2481 	struct module *module;
2482 	u16 offset;
2483 };
2484 
2485 struct bpf_kfunc_btf_tab {
2486 	struct bpf_kfunc_btf descs[MAX_KFUNC_BTFS];
2487 	u32 nr_descs;
2488 };
2489 
2490 static int kfunc_desc_cmp_by_id_off(const void *a, const void *b)
2491 {
2492 	const struct bpf_kfunc_desc *d0 = a;
2493 	const struct bpf_kfunc_desc *d1 = b;
2494 
2495 	/* func_id is not greater than BTF_MAX_TYPE */
2496 	return d0->func_id - d1->func_id ?: d0->offset - d1->offset;
2497 }
2498 
2499 static int kfunc_btf_cmp_by_off(const void *a, const void *b)
2500 {
2501 	const struct bpf_kfunc_btf *d0 = a;
2502 	const struct bpf_kfunc_btf *d1 = b;
2503 
2504 	return d0->offset - d1->offset;
2505 }
2506 
2507 static struct bpf_kfunc_desc *
2508 find_kfunc_desc(const struct bpf_prog *prog, u32 func_id, u16 offset)
2509 {
2510 	struct bpf_kfunc_desc desc = {
2511 		.func_id = func_id,
2512 		.offset = offset,
2513 	};
2514 	struct bpf_kfunc_desc_tab *tab;
2515 
2516 	tab = prog->aux->kfunc_tab;
2517 	return bsearch(&desc, tab->descs, tab->nr_descs,
2518 		       sizeof(tab->descs[0]), kfunc_desc_cmp_by_id_off);
2519 }
2520 
2521 int bpf_get_kfunc_addr(const struct bpf_prog *prog, u32 func_id,
2522 		       u16 btf_fd_idx, u8 **func_addr)
2523 {
2524 	const struct bpf_kfunc_desc *desc;
2525 
2526 	desc = find_kfunc_desc(prog, func_id, btf_fd_idx);
2527 	if (!desc)
2528 		return -EFAULT;
2529 
2530 	*func_addr = (u8 *)desc->addr;
2531 	return 0;
2532 }
2533 
2534 #define BPF_FD_SLOT_BTF	1UL
2535 
2536 static void fd_slot_set_map(struct bpf_fd_array *slot, struct bpf_map *map)
2537 {
2538 	slot->val = (unsigned long)map;
2539 }
2540 
2541 static void fd_slot_set_btf(struct bpf_fd_array *slot, struct btf *btf)
2542 {
2543 	slot->val = (unsigned long)btf | BPF_FD_SLOT_BTF;
2544 }
2545 
2546 static struct bpf_map *fd_slot_map(struct bpf_fd_array slot)
2547 {
2548 	if (slot.val & BPF_FD_SLOT_BTF)
2549 		return NULL;
2550 	return (struct bpf_map *)slot.val;
2551 }
2552 
2553 static struct btf *fd_slot_btf(struct bpf_fd_array slot)
2554 {
2555 	if (!(slot.val & BPF_FD_SLOT_BTF))
2556 		return NULL;
2557 	return (struct btf *)(slot.val & ~BPF_FD_SLOT_BTF);
2558 }
2559 
2560 static struct btf *
2561 fd_array_get_btf_continuous(struct bpf_verifier_env *env, u32 idx)
2562 {
2563 	struct btf *btf;
2564 
2565 	if (idx >= env->fd_array_cnt) {
2566 		verbose(env, "kfunc fd_idx %u out of bounds, fd_array_cnt %u\n",
2567 			idx, env->fd_array_cnt);
2568 		return ERR_PTR(-EINVAL);
2569 	}
2570 	btf = fd_slot_btf(env->fd_array[idx]);
2571 	if (!btf) {
2572 		verbose(env, "kfunc fd_idx %u is not a module BTF\n", idx);
2573 		return ERR_PTR(-EINVAL);
2574 	}
2575 	btf_get(btf);
2576 	return btf;
2577 }
2578 
2579 static struct btf *
2580 fd_array_get_btf_sparse(struct bpf_verifier_env *env, u32 idx)
2581 {
2582 	struct btf *btf;
2583 	int btf_fd;
2584 
2585 	if (copy_from_bpfptr_offset(&btf_fd, env->fd_array_raw,
2586 				    (size_t)idx * sizeof(btf_fd), sizeof(btf_fd)))
2587 		return ERR_PTR(-EFAULT);
2588 	btf = btf_get_by_fd(btf_fd);
2589 	if (IS_ERR(btf)) {
2590 		verbose(env, "invalid module BTF fd specified\n");
2591 		return btf;
2592 	}
2593 	return btf;
2594 }
2595 
2596 static struct btf *fd_array_get_btf(struct bpf_verifier_env *env, u32 idx)
2597 {
2598 	if (env->signature) {
2599 		verbose(env, "signed program cannot bind any BTF\n");
2600 		return ERR_PTR(-EACCES);
2601 	}
2602 	if (env->fd_array)
2603 		return fd_array_get_btf_continuous(env, idx);
2604 	if (!bpfptr_is_null(env->fd_array_raw))
2605 		return fd_array_get_btf_sparse(env, idx);
2606 
2607 	verbose(env, "kfunc offset > 0 without fd_array is invalid\n");
2608 	return ERR_PTR(-EPROTO);
2609 }
2610 
2611 static struct btf *__find_kfunc_desc_btf(struct bpf_verifier_env *env,
2612 					 s16 offset)
2613 {
2614 	struct bpf_kfunc_btf kf_btf = { .offset = offset };
2615 	struct bpf_kfunc_btf_tab *tab;
2616 	struct bpf_kfunc_btf *b;
2617 	struct module *mod;
2618 	struct btf *btf;
2619 
2620 	tab = env->prog->aux->kfunc_btf_tab;
2621 	b = bsearch(&kf_btf, tab->descs, tab->nr_descs,
2622 		    sizeof(tab->descs[0]), kfunc_btf_cmp_by_off);
2623 	if (!b) {
2624 		if (tab->nr_descs == MAX_KFUNC_BTFS) {
2625 			verbose(env, "too many different module BTFs\n");
2626 			return ERR_PTR(-E2BIG);
2627 		}
2628 
2629 		btf = fd_array_get_btf(env, offset);
2630 		if (IS_ERR(btf))
2631 			return btf;
2632 		if (!btf_is_module(btf)) {
2633 			verbose(env, "BTF fd for kfunc is not a module BTF\n");
2634 			btf_put(btf);
2635 			return ERR_PTR(-EINVAL);
2636 		}
2637 
2638 		mod = btf_try_get_module(btf);
2639 		if (!mod) {
2640 			btf_put(btf);
2641 			return ERR_PTR(-ENXIO);
2642 		}
2643 
2644 		b = &tab->descs[tab->nr_descs++];
2645 		b->btf = btf;
2646 		b->module = mod;
2647 		b->offset = offset;
2648 
2649 		/* sort() reorders entries by value, so b may no longer point
2650 		 * to the right entry after this
2651 		 */
2652 		sort(tab->descs, tab->nr_descs, sizeof(tab->descs[0]),
2653 		     kfunc_btf_cmp_by_off, NULL);
2654 	} else {
2655 		btf = b->btf;
2656 	}
2657 
2658 	return btf;
2659 }
2660 
2661 void bpf_free_kfunc_btf_tab(struct bpf_kfunc_btf_tab *tab)
2662 {
2663 	if (!tab)
2664 		return;
2665 
2666 	while (tab->nr_descs--) {
2667 		module_put(tab->descs[tab->nr_descs].module);
2668 		btf_put(tab->descs[tab->nr_descs].btf);
2669 	}
2670 	kfree(tab);
2671 }
2672 
2673 static struct btf *find_kfunc_desc_btf(struct bpf_verifier_env *env, s16 offset)
2674 {
2675 	if (offset) {
2676 		if (offset < 0) {
2677 			/* In the future, this can be allowed to increase limit
2678 			 * of fd index into fd_array, interpreted as u16.
2679 			 */
2680 			verbose(env, "negative offset disallowed for kernel module function call\n");
2681 			return ERR_PTR(-EINVAL);
2682 		}
2683 
2684 		return __find_kfunc_desc_btf(env, offset);
2685 	}
2686 	return btf_vmlinux ?: ERR_PTR(-ENOENT);
2687 }
2688 
2689 static struct btf *find_kfunc_desc_btf_cached(struct bpf_verifier_env *env, s16 offset)
2690 {
2691 	struct bpf_kfunc_btf kf_btf = { .offset = offset };
2692 	struct bpf_kfunc_btf_tab *tab;
2693 	struct bpf_kfunc_btf *b;
2694 
2695 	if (!offset)
2696 		return btf_vmlinux ?: ERR_PTR(-ENOENT);
2697 	if (offset < 0)
2698 		return ERR_PTR(-EINVAL);
2699 
2700 	tab = env->prog->aux->kfunc_btf_tab;
2701 	if (!tab)
2702 		return ERR_PTR(-ENOENT);
2703 
2704 	b = bsearch(&kf_btf, tab->descs, tab->nr_descs,
2705 		    sizeof(tab->descs[0]), kfunc_btf_cmp_by_off);
2706 	return b ? b->btf : ERR_PTR(-ENOENT);
2707 }
2708 
2709 #define KF_IMPL_SUFFIX "_impl"
2710 
2711 static const struct btf_type *find_kfunc_impl_proto(struct bpf_verifier_log *log,
2712 						    struct btf *btf,
2713 						    const char *func_name)
2714 {
2715 	const struct btf_type *func;
2716 	char buf[KSYM_NAME_LEN];
2717 	s32 impl_id;
2718 	int len;
2719 
2720 	len = snprintf(buf, sizeof(buf), "%s%s", func_name, KF_IMPL_SUFFIX);
2721 	if (len < 0 || len >= sizeof(buf)) {
2722 		bpf_log(log, "function name %s%s is too long\n",
2723 			func_name, KF_IMPL_SUFFIX);
2724 		return NULL;
2725 	}
2726 
2727 	impl_id = btf_find_by_name_kind(btf, buf, BTF_KIND_FUNC);
2728 	if (impl_id <= 0) {
2729 		bpf_log(log, "cannot find function %s in BTF\n", buf);
2730 		return NULL;
2731 	}
2732 
2733 	func = btf_type_by_id(btf, impl_id);
2734 
2735 	return btf_type_by_id(btf, func->type);
2736 }
2737 
2738 static int fetch_kfunc_meta(struct bpf_verifier_env *env,
2739 			    s32 func_id,
2740 			    s16 offset,
2741 			    struct bpf_kfunc_meta *kfunc)
2742 {
2743 	const struct btf_type *func, *func_proto;
2744 	const char *func_name;
2745 	u32 *kfunc_flags;
2746 	struct btf *btf;
2747 
2748 	if (func_id <= 0) {
2749 		verbose(env, "invalid kernel function btf_id %d\n", func_id);
2750 		return -EINVAL;
2751 	}
2752 
2753 	btf = find_kfunc_desc_btf(env, offset);
2754 	if (IS_ERR(btf)) {
2755 		verbose(env, "failed to find BTF for kernel function\n");
2756 		return PTR_ERR(btf);
2757 	}
2758 
2759 	/*
2760 	 * Note that kfunc_flags may be NULL at this point, which
2761 	 * means that we couldn't find func_id in any relevant
2762 	 * kfunc_id_set. This most likely indicates an invalid kfunc
2763 	 * call.  However we don't fail with an error here,
2764 	 * and let the caller decide what to do with NULL kfunc->flags.
2765 	 */
2766 	kfunc_flags = btf_kfunc_flags(btf, func_id, env->prog);
2767 
2768 	func = btf_type_by_id(btf, func_id);
2769 	if (!func || !btf_type_is_func(func)) {
2770 		verbose(env, "kernel btf_id %d is not a function\n", func_id);
2771 		return -EINVAL;
2772 	}
2773 
2774 	func_name = btf_name_by_offset(btf, func->name_off);
2775 
2776 	/*
2777 	 * An actual prototype of a kfunc with KF_IMPLICIT_ARGS flag
2778 	 * can be found through the counterpart _impl kfunc.
2779 	 */
2780 	if (kfunc_flags && (*kfunc_flags & KF_IMPLICIT_ARGS))
2781 		func_proto = find_kfunc_impl_proto(&env->log, btf, func_name);
2782 	else
2783 		func_proto = btf_type_by_id(btf, func->type);
2784 
2785 	if (!func_proto || !btf_type_is_func_proto(func_proto)) {
2786 		verbose(env, "kernel function btf_id %d does not have a valid func_proto\n",
2787 			func_id);
2788 		return -EINVAL;
2789 	}
2790 
2791 	memset(kfunc, 0, sizeof(*kfunc));
2792 	kfunc->btf = btf;
2793 	kfunc->id = func_id;
2794 	kfunc->name = func_name;
2795 	kfunc->proto = func_proto;
2796 	kfunc->flags = kfunc_flags;
2797 
2798 	return 0;
2799 }
2800 
2801 static int gen_kfunc_arg_proto(struct bpf_verifier_env *env, struct bpf_call_arg_meta *meta,
2802 			       struct bpf_func_proto *proto);
2803 
2804 int bpf_add_kfunc_call(struct bpf_verifier_env *env, u32 func_id, u16 offset)
2805 {
2806 	struct bpf_call_arg_meta meta;
2807 	struct bpf_kfunc_btf_tab *btf_tab;
2808 	struct btf_func_model func_model;
2809 	struct bpf_kfunc_desc_tab *tab;
2810 	struct bpf_prog_aux *prog_aux;
2811 	struct bpf_kfunc_meta kfunc;
2812 	struct bpf_kfunc_desc *desc;
2813 	unsigned long addr;
2814 	int err;
2815 
2816 	prog_aux = env->prog->aux;
2817 	tab = prog_aux->kfunc_tab;
2818 	btf_tab = prog_aux->kfunc_btf_tab;
2819 	if (!tab) {
2820 		if (!btf_vmlinux) {
2821 			verbose(env, "calling kernel function is not supported without CONFIG_DEBUG_INFO_BTF\n");
2822 			return -ENOTSUPP;
2823 		}
2824 
2825 		if (!env->prog->jit_requested) {
2826 			verbose(env, "JIT is required for calling kernel function\n");
2827 			return -ENOTSUPP;
2828 		}
2829 
2830 		if (!bpf_jit_supports_kfunc_call()) {
2831 			verbose(env, "JIT does not support calling kernel function\n");
2832 			return -ENOTSUPP;
2833 		}
2834 
2835 		if (!env->prog->gpl_compatible) {
2836 			verbose(env, "cannot call kernel function from non-GPL compatible program\n");
2837 			return -EINVAL;
2838 		}
2839 
2840 		tab = kzalloc_obj(*tab, GFP_KERNEL_ACCOUNT);
2841 		if (!tab)
2842 			return -ENOMEM;
2843 		prog_aux->kfunc_tab = tab;
2844 	}
2845 
2846 	env->prog->jit_required = 1;
2847 
2848 	/* func_id == 0 is always invalid, but instead of returning an error, be
2849 	 * conservative and wait until the code elimination pass before returning
2850 	 * error, so that invalid calls that get pruned out can be in BPF programs
2851 	 * loaded from userspace.  It is also required that offset be untouched
2852 	 * for such calls.
2853 	 */
2854 	if (!func_id && !offset)
2855 		return 0;
2856 
2857 	if (!btf_tab && offset) {
2858 		btf_tab = kzalloc_obj(*btf_tab, GFP_KERNEL_ACCOUNT);
2859 		if (!btf_tab)
2860 			return -ENOMEM;
2861 		prog_aux->kfunc_btf_tab = btf_tab;
2862 	}
2863 
2864 	if (find_kfunc_desc(env->prog, func_id, offset))
2865 		return 0;
2866 
2867 	if (tab->nr_descs == MAX_KFUNC_DESCS) {
2868 		verbose(env, "too many different kernel function calls\n");
2869 		return -E2BIG;
2870 	}
2871 
2872 	err = fetch_kfunc_meta(env, func_id, offset, &kfunc);
2873 	if (err)
2874 		return err;
2875 
2876 	addr = kallsyms_lookup_name(kfunc.name);
2877 	if (!addr) {
2878 		verbose(env, "cannot find address for kernel function %s\n", kfunc.name);
2879 		return -EINVAL;
2880 	}
2881 
2882 	if (bpf_dev_bound_kfunc_id(func_id)) {
2883 		err = bpf_dev_bound_kfunc_check(&env->log, prog_aux);
2884 		if (err)
2885 			return err;
2886 	}
2887 
2888 	err = btf_distill_func_proto(&env->log, kfunc.btf, kfunc.proto, kfunc.name, &func_model);
2889 	if (err)
2890 		return err;
2891 
2892 	memset(&meta, 0, sizeof(meta));
2893 	meta.btf = kfunc.btf;
2894 	meta.func_id = kfunc.id;
2895 	meta.func_proto = kfunc.proto;
2896 	meta.func_name = kfunc.name;
2897 	meta.kfunc_flags = kfunc.flags ? *kfunc.flags : 0;
2898 
2899 	tab = krealloc(tab, struct_size(tab, descs, tab->nr_descs + 1), GFP_KERNEL_ACCOUNT);
2900 	if (!tab)
2901 		return -ENOMEM;
2902 	prog_aux->kfunc_tab = tab;
2903 
2904 	desc = &tab->descs[tab->nr_descs];
2905 	memset(desc, 0, sizeof(*desc));
2906 
2907 	err = gen_kfunc_arg_proto(env, &meta, &desc->proto);
2908 	if (err)
2909 		return err;
2910 
2911 	desc->func_id = func_id;
2912 	desc->offset = offset;
2913 	desc->addr = addr;
2914 	desc->func_model = func_model;
2915 	tab->nr_descs++;
2916 	sort(tab->descs, tab->nr_descs, sizeof(tab->descs[0]),
2917 	     kfunc_desc_cmp_by_id_off, NULL);
2918 	return 0;
2919 }
2920 
2921 static int add_subprogs(struct bpf_verifier_env *env)
2922 {
2923 	struct bpf_subprog_info *subprog = env->subprog_info;
2924 	int i, ret, insn_cnt = env->prog->len, ex_cb_insn;
2925 	struct bpf_insn *insn = env->prog->insnsi;
2926 	const char *operation, *suggestion;
2927 
2928 	/* Add entry function. */
2929 	ret = add_subprog(env, 0);
2930 	if (ret)
2931 		return ret;
2932 
2933 	for (i = 0; i < insn_cnt; i++, insn++) {
2934 		if (!bpf_pseudo_func(insn) && !bpf_pseudo_call(insn))
2935 			continue;
2936 
2937 		if (!env->bpf_capable) {
2938 			if (bpf_pseudo_func(insn)) {
2939 				operation = "BPF function reference";
2940 				suggestion = "Load this program with the required capability, or avoid BPF function references in unprivileged programs.";
2941 			} else {
2942 				operation = "BPF-to-BPF function call";
2943 				suggestion = "Load this program with the required capability, or avoid BPF-to-BPF function calls in unprivileged programs.";
2944 			}
2945 			verbose(env, "loading/calling other bpf or kernel functions are allowed for CAP_BPF and CAP_SYS_ADMIN\n");
2946 			bpf_diag_policy(
2947 				env, i, operation,
2948 				"loading or calling other BPF functions requires CAP_BPF or CAP_SYS_ADMIN",
2949 				suggestion);
2950 			return -EPERM;
2951 		}
2952 
2953 		ret = add_subprog(env, i + insn->imm + 1);
2954 		if (ret < 0)
2955 			return ret;
2956 	}
2957 
2958 	ret = bpf_find_exception_callback_insn_off(env);
2959 	if (ret < 0)
2960 		return ret;
2961 	ex_cb_insn = ret;
2962 
2963 	/* If ex_cb_insn > 0, this means that the main program has a subprog
2964 	 * marked using BTF decl tag to serve as the exception callback.
2965 	 */
2966 	if (ex_cb_insn) {
2967 		ret = add_subprog(env, ex_cb_insn);
2968 		if (ret < 0)
2969 			return ret;
2970 		for (i = 1; i < env->subprog_cnt; i++) {
2971 			if (env->subprog_info[i].start != ex_cb_insn)
2972 				continue;
2973 			env->exception_callback_subprog = i;
2974 			bpf_mark_subprog_exc_cb(env, i);
2975 			break;
2976 		}
2977 	}
2978 
2979 	/* Add a fake 'exit' subprog which could simplify subprog iteration
2980 	 * logic. 'subprog_cnt' should not be increased.
2981 	 */
2982 	subprog[env->subprog_cnt].start = insn_cnt;
2983 
2984 	if (env->log.level & BPF_LOG_LEVEL2)
2985 		for (i = 0; i < env->subprog_cnt; i++)
2986 			verbose(env, "func#%d @%d\n", i, subprog[i].start);
2987 
2988 	return 0;
2989 }
2990 
2991 static int add_kfuncs(struct bpf_verifier_env *env)
2992 {
2993 	struct bpf_insn *insn = env->prog->insnsi;
2994 	int i, ret, insn_cnt = env->prog->len;
2995 
2996 	for (i = 0; i < insn_cnt; i++, insn++) {
2997 		if (!bpf_pseudo_kfunc_call(insn))
2998 			continue;
2999 
3000 		if (!env->bpf_capable) {
3001 			verbose(env, "loading/calling other bpf or kernel functions are allowed for CAP_BPF and CAP_SYS_ADMIN\n");
3002 			bpf_diag_policy(
3003 				env, i, "kernel function call",
3004 				"calling kernel functions requires CAP_BPF or CAP_SYS_ADMIN",
3005 				"Load this program with the required capability, or avoid kernel function calls in unprivileged programs.");
3006 			return -EPERM;
3007 		}
3008 
3009 		ret = bpf_add_kfunc_call(env, insn->imm, insn->off);
3010 		if (ret < 0)
3011 			return ret;
3012 	}
3013 
3014 	return 0;
3015 }
3016 
3017 static int check_subprogs(struct bpf_verifier_env *env)
3018 {
3019 	int i, subprog_start, subprog_end, off, cur_subprog = 0;
3020 	struct bpf_subprog_info *subprog = env->subprog_info;
3021 	struct bpf_insn *insn = env->prog->insnsi;
3022 	int insn_cnt = env->prog->len;
3023 
3024 	/* now check that all jumps are within the same subprog */
3025 	subprog_start = subprog[cur_subprog].start;
3026 	subprog_end = subprog[cur_subprog + 1].start;
3027 	for (i = 0; i < insn_cnt; i++) {
3028 		u8 code = insn[i].code;
3029 
3030 		if (code == (BPF_JMP | BPF_CALL) &&
3031 		    insn[i].src_reg == 0 &&
3032 		    insn[i].imm == BPF_FUNC_tail_call) {
3033 			subprog[cur_subprog].has_tail_call = true;
3034 			subprog[cur_subprog].tail_call_reachable = true;
3035 		}
3036 		if (BPF_CLASS(code) == BPF_LD &&
3037 		    (BPF_MODE(code) == BPF_ABS || BPF_MODE(code) == BPF_IND))
3038 			subprog[cur_subprog].has_ld_abs = true;
3039 		if (BPF_CLASS(code) != BPF_JMP && BPF_CLASS(code) != BPF_JMP32)
3040 			goto next;
3041 		if (BPF_OP(code) == BPF_CALL)
3042 			goto next;
3043 		if (BPF_OP(code) == BPF_EXIT) {
3044 			subprog[cur_subprog].exit_idx = i;
3045 			goto next;
3046 		}
3047 		if (insn_is_gotox(&insn[i]))
3048 			goto next;
3049 		off = i + bpf_jmp_offset(&insn[i]) + 1;
3050 		if (off < subprog_start || off >= subprog_end) {
3051 			verbose(env, "jump out of range from insn %d to %d\n", i, off);
3052 			bpf_diag_program_structure(
3053 				env, i, "jump out of range",
3054 				"Keep branch targets within the same subprogram, or use an explicit subprogram call.",
3055 				"Instruction %d jumps to instruction %d, but subprogram %d only contains instructions %d through %d. "
3056 				"A branch target must stay inside the same subprogram.",
3057 				i, off, cur_subprog, subprog_start, subprog_end - 1);
3058 			return -EINVAL;
3059 		}
3060 next:
3061 		if (i == subprog_end - 1) {
3062 			/* to avoid fall-through from one subprog into another
3063 			 * the last insn of the subprog should be either exit
3064 			 * or unconditional jump back or bpf_throw call
3065 			 */
3066 			if (code != (BPF_JMP | BPF_EXIT) &&
3067 			    code != (BPF_JMP32 | BPF_JA) &&
3068 			    code != (BPF_JMP | BPF_JA) &&
3069 			    !insn_is_gotox(&insn[i])) {
3070 				verbose(env, "last insn is not an exit or jmp\n");
3071 				bpf_diag_program_structure(
3072 					env, i, "subprogram can fall through",
3073 					"End each subprogram with an exit or an explicit jump that keeps control flow inside the subprogram.",
3074 					"Subprogram %d reaches its last instruction %d without an exit or jump, so control could continue into the next subprogram.",
3075 					cur_subprog, i);
3076 				return -EINVAL;
3077 			}
3078 			subprog_start = subprog_end;
3079 			cur_subprog++;
3080 			if (cur_subprog < env->subprog_cnt)
3081 				subprog_end = subprog[cur_subprog + 1].start;
3082 		}
3083 	}
3084 	return 0;
3085 }
3086 
3087 /*
3088  * Sort subprogs in topological order so that leaf subprogs come first and
3089  * their callers come later. This is a DFS post-order traversal of the call
3090  * graph. Scan only reachable instructions (those in the computed postorder) of
3091  * the current subprog to discover callees (direct subprogs and sync
3092  * callbacks).
3093  */
3094 static int sort_subprogs_topo(struct bpf_verifier_env *env)
3095 {
3096 	struct bpf_subprog_info *si = env->subprog_info;
3097 	int *insn_postorder = env->cfg.insn_postorder;
3098 	struct bpf_insn *insn = env->prog->insnsi;
3099 	int cnt = env->subprog_cnt;
3100 	int *dfs_stack = NULL;
3101 	int top = 0, order = 0;
3102 	int i, ret = 0;
3103 	u8 *color = NULL;
3104 
3105 	color = kvzalloc_objs(*color, cnt, GFP_KERNEL_ACCOUNT);
3106 	dfs_stack = kvmalloc_objs(*dfs_stack, cnt, GFP_KERNEL_ACCOUNT);
3107 	if (!color || !dfs_stack) {
3108 		ret = -ENOMEM;
3109 		goto out;
3110 	}
3111 
3112 	/*
3113 	 * DFS post-order traversal.
3114 	 * Color values: 0 = unvisited, 1 = on stack, 2 = done.
3115 	 */
3116 	for (i = 0; i < cnt; i++) {
3117 		if (color[i])
3118 			continue;
3119 		color[i] = 1;
3120 		dfs_stack[top++] = i;
3121 
3122 		while (top > 0) {
3123 			int cur = dfs_stack[top - 1];
3124 			int po_start = si[cur].postorder_start;
3125 			int po_end = si[cur + 1].postorder_start;
3126 			bool pushed = false;
3127 			int j;
3128 
3129 			for (j = po_start; j < po_end; j++) {
3130 				int idx = insn_postorder[j];
3131 				int callee;
3132 
3133 				if (!bpf_pseudo_call(&insn[idx]) && !bpf_pseudo_func(&insn[idx]))
3134 					continue;
3135 				callee = bpf_find_subprog(env, idx + insn[idx].imm + 1);
3136 				if (callee < 0) {
3137 					ret = -EFAULT;
3138 					goto out;
3139 				}
3140 				if (color[callee] == 2)
3141 					continue;
3142 				if (color[callee] == 1) {
3143 					if (bpf_pseudo_func(&insn[idx]))
3144 						continue;
3145 					verbose(env, "recursive call from %s() to %s()\n",
3146 						bpf_subprog_name(env, cur),
3147 						bpf_subprog_name(env, callee));
3148 					bpf_diag_program_structure(
3149 						env, idx, "recursive subprogram call",
3150 						"Rewrite the recursion as an explicit bounded loop, or split the logic so subprogram calls do not form a cycle.",
3151 						"This bpf2bpf call would make the subprogram call graph recursive. "
3152 						"The verifier requires a finite, acyclic call graph so it can bound stack depth and analysis.");
3153 					ret = -EINVAL;
3154 					goto out;
3155 				}
3156 				color[callee] = 1;
3157 				dfs_stack[top++] = callee;
3158 				pushed = true;
3159 				break;
3160 			}
3161 
3162 			if (!pushed) {
3163 				color[cur] = 2;
3164 				env->subprog_topo_order[order++] = cur;
3165 				top--;
3166 			}
3167 		}
3168 	}
3169 
3170 	if (env->log.level & BPF_LOG_LEVEL2)
3171 		for (i = 0; i < cnt; i++)
3172 			verbose(env, "topo_order[%d] = %s\n",
3173 				i, bpf_subprog_name(env, env->subprog_topo_order[i]));
3174 out:
3175 	kvfree(dfs_stack);
3176 	kvfree(color);
3177 	return ret;
3178 }
3179 
3180 static void mark_stack_slots_scratched(struct bpf_verifier_env *env,
3181 				       int spi, int nr_slots)
3182 {
3183 	int i;
3184 
3185 	for (i = 0; i < nr_slots; i++)
3186 		mark_stack_slot_scratched(env, spi - i);
3187 }
3188 
3189 static int __check_reg_arg(struct bpf_verifier_env *env, struct bpf_reg_state *regs, u32 regno,
3190 			   enum bpf_reg_arg_type t)
3191 {
3192 	struct bpf_reg_state *reg;
3193 
3194 	mark_reg_scratched(env, regno);
3195 
3196 	reg = &regs[regno];
3197 	if (t == SRC_OP) {
3198 		/* check whether register used as source operand can be read */
3199 		if (reg->type == NOT_INIT) {
3200 			verbose(env, "R%d !read_ok\n", regno);
3201 			bpf_diag_unreadable_reg(env, env->insn_idx, regno);
3202 			return -EACCES;
3203 		}
3204 		/* We don't need to worry about FP liveness because it's read-only */
3205 		if (regno == BPF_REG_FP)
3206 			return 0;
3207 
3208 		return 0;
3209 	} else {
3210 		/* check whether register used as dest operand can be written to */
3211 		if (regno == BPF_REG_FP) {
3212 			verbose(env, "frame pointer is read only\n");
3213 			return -EACCES;
3214 		}
3215 		if (t == DST_OP)
3216 			mark_reg_unknown(env, regs, regno);
3217 	}
3218 	return 0;
3219 }
3220 
3221 static int check_reg_arg(struct bpf_verifier_env *env, u32 regno,
3222 			 enum bpf_reg_arg_type t)
3223 {
3224 	struct bpf_verifier_state *vstate = env->cur_state;
3225 	struct bpf_func_state *state = vstate->frame[vstate->curframe];
3226 
3227 	return __check_reg_arg(env, state->regs, regno, t);
3228 }
3229 
3230 static void mark_indirect_target(struct bpf_verifier_env *env, int idx)
3231 {
3232 	env->insn_aux_data[idx].indirect_target = true;
3233 }
3234 
3235 #define LR_FRAMENO_BITS	4
3236 #define LR_SPI_BITS	6
3237 #define LR_ENTRY_BITS	(LR_SPI_BITS + LR_FRAMENO_BITS + 1)
3238 #define LR_SIZE_BITS	4
3239 #define LR_FRAMENO_MASK	((1ull << LR_FRAMENO_BITS) - 1)
3240 #define LR_SPI_MASK	((1ull << LR_SPI_BITS)     - 1)
3241 #define LR_SIZE_MASK	((1ull << LR_SIZE_BITS)    - 1)
3242 #define LR_SPI_OFF	LR_FRAMENO_BITS
3243 #define LR_IS_REG_OFF	(LR_SPI_BITS + LR_FRAMENO_BITS)
3244 #define LINKED_REGS_MAX	5
3245 
3246 static_assert(MAX_CALL_FRAMES <= (1 << LR_FRAMENO_BITS));
3247 static_assert(LINKED_REGS_MAX < (1 << LR_SIZE_BITS));
3248 static_assert(LINKED_REGS_MAX * LR_ENTRY_BITS + LR_SIZE_BITS <= 64);
3249 
3250 struct linked_reg {
3251 	u8 frameno;
3252 	union {
3253 		u8 spi;
3254 		u8 regno;
3255 	};
3256 	bool is_reg;
3257 };
3258 
3259 struct linked_regs {
3260 	int cnt;
3261 	struct linked_reg entries[LINKED_REGS_MAX];
3262 };
3263 
3264 static struct linked_reg *linked_regs_push(struct linked_regs *s)
3265 {
3266 	if (s->cnt < LINKED_REGS_MAX)
3267 		return &s->entries[s->cnt++];
3268 
3269 	return NULL;
3270 }
3271 
3272 /*
3273  * Use u64 as a vector of 5 11-bit values, use first 4-bits to track
3274  * number of elements currently in stack.
3275  * Pack one history entry for linked registers as 11 bits in the following format:
3276  * - 4-bits frameno
3277  * - 6-bits spi_or_reg
3278  * - 1-bit  is_reg
3279  */
3280 static u64 linked_regs_pack(struct linked_regs *s)
3281 {
3282 	u64 val = 0;
3283 	int i;
3284 
3285 	for (i = 0; i < s->cnt; ++i) {
3286 		struct linked_reg *e = &s->entries[i];
3287 		u64 tmp = 0;
3288 
3289 		tmp |= e->frameno;
3290 		tmp |= e->spi << LR_SPI_OFF;
3291 		tmp |= (e->is_reg ? 1 : 0) << LR_IS_REG_OFF;
3292 
3293 		val <<= LR_ENTRY_BITS;
3294 		val |= tmp;
3295 	}
3296 	val <<= LR_SIZE_BITS;
3297 	val |= s->cnt;
3298 	return val;
3299 }
3300 
3301 static void linked_regs_unpack(u64 val, struct linked_regs *s)
3302 {
3303 	int i;
3304 
3305 	s->cnt = val & LR_SIZE_MASK;
3306 	val >>= LR_SIZE_BITS;
3307 
3308 	for (i = 0; i < s->cnt; ++i) {
3309 		struct linked_reg *e = &s->entries[i];
3310 
3311 		e->frameno =  val & LR_FRAMENO_MASK;
3312 		e->spi     = (val >> LR_SPI_OFF) & LR_SPI_MASK;
3313 		e->is_reg  = (val >> LR_IS_REG_OFF) & 0x1;
3314 		val >>= LR_ENTRY_BITS;
3315 	}
3316 }
3317 
3318 const char *bpf_disasm_kfunc_name(void *data, const struct bpf_insn *insn)
3319 {
3320 	const struct btf_type *func;
3321 	struct btf *desc_btf;
3322 
3323 	if (insn->src_reg != BPF_PSEUDO_KFUNC_CALL)
3324 		return NULL;
3325 
3326 	desc_btf = find_kfunc_desc_btf_cached(data, insn->off);
3327 	if (IS_ERR(desc_btf))
3328 		return "<error>";
3329 
3330 	func = btf_type_by_id(desc_btf, insn->imm);
3331 	if (!func || !btf_type_is_func(func))
3332 		return "<error>";
3333 	return btf_name_by_offset(desc_btf, func->name_off);
3334 }
3335 
3336 void bpf_verbose_insn(struct bpf_verifier_env *env, struct bpf_insn *insn)
3337 {
3338 	const struct bpf_insn_cbs cbs = {
3339 		.cb_call	= bpf_disasm_kfunc_name,
3340 		.cb_print	= verbose,
3341 		.private_data	= env,
3342 	};
3343 
3344 	print_bpf_insn(&cbs, insn, env->allow_ptr_leaks);
3345 }
3346 
3347 /* If any register R in hist->linked_regs is marked as precise in bt,
3348  * do bt_set_frame_{reg,slot}(bt, R) for all registers in hist->linked_regs.
3349  */
3350 void bpf_bt_sync_linked_regs(struct backtrack_state *bt, struct bpf_jmp_history_entry *hist)
3351 {
3352 	struct linked_regs linked_regs;
3353 	bool some_precise = false;
3354 	int i;
3355 
3356 	if (!hist || hist->linked_regs == 0)
3357 		return;
3358 
3359 	linked_regs_unpack(hist->linked_regs, &linked_regs);
3360 	for (i = 0; i < linked_regs.cnt; ++i) {
3361 		struct linked_reg *e = &linked_regs.entries[i];
3362 
3363 		if ((e->is_reg && bt_is_frame_reg_set(bt, e->frameno, e->regno)) ||
3364 		    (!e->is_reg && bt_is_frame_slot_set(bt, e->frameno, e->spi))) {
3365 			some_precise = true;
3366 			break;
3367 		}
3368 	}
3369 
3370 	if (!some_precise)
3371 		return;
3372 
3373 	for (i = 0; i < linked_regs.cnt; ++i) {
3374 		struct linked_reg *e = &linked_regs.entries[i];
3375 
3376 		if (e->is_reg)
3377 			bpf_bt_set_frame_reg(bt, e->frameno, e->regno);
3378 		else
3379 			bpf_bt_set_frame_slot(bt, e->frameno, e->spi);
3380 	}
3381 }
3382 
3383 int mark_chain_precision(struct bpf_verifier_env *env, int regno)
3384 {
3385 	return bpf_mark_chain_precision(env, env->cur_state, regno, NULL);
3386 }
3387 
3388 /* mark_chain_precision_batch() assumes that env->bt is set in the caller to
3389  * desired reg and stack masks across all relevant frames
3390  */
3391 static int mark_chain_precision_batch(struct bpf_verifier_env *env,
3392 				      struct bpf_verifier_state *starting_state)
3393 {
3394 	return bpf_mark_chain_precision(env, starting_state, -1, NULL);
3395 }
3396 
3397 /* check if register is a constant scalar value */
3398 static bool is_reg_const(struct bpf_reg_state *reg, bool subreg32)
3399 {
3400 	return reg->type == SCALAR_VALUE &&
3401 	       tnum_is_const(subreg32 ? tnum_subreg(reg->var_off) : reg->var_off);
3402 }
3403 
3404 /* assuming is_reg_const() is true, return constant value of a register */
3405 static u64 reg_const_value(struct bpf_reg_state *reg, bool subreg32)
3406 {
3407 	return subreg32 ? tnum_subreg(reg->var_off).value : reg->var_off.value;
3408 }
3409 
3410 static bool is_pointer_regtype(enum bpf_reg_type type)
3411 {
3412 	return type != SCALAR_VALUE && type != NOT_INIT;
3413 }
3414 
3415 static bool __is_pointer_value(bool allow_ptr_leaks,
3416 			       const struct bpf_reg_state *reg)
3417 {
3418 	if (allow_ptr_leaks)
3419 		return false;
3420 
3421 	return is_pointer_regtype(reg->type);
3422 }
3423 
3424 static void clear_scalar_id(struct bpf_reg_state *reg)
3425 {
3426 	reg->id = 0;
3427 	reg->delta = 0;
3428 }
3429 
3430 static void assign_scalar_id_before_mov(struct bpf_verifier_env *env,
3431 					struct bpf_reg_state *src_reg)
3432 {
3433 	if (src_reg->type != SCALAR_VALUE)
3434 		return;
3435 	/*
3436 	 * The verifier is processing rX = rY insn and
3437 	 * rY->id has special linked register already.
3438 	 * Cleared it, since multiple rX += const are not supported.
3439 	 */
3440 	if (src_reg->id & BPF_ADD_CONST)
3441 		clear_scalar_id(src_reg);
3442 	/*
3443 	 * Ensure that src_reg has a valid ID that will be copied to
3444 	 * dst_reg and then will be used by sync_linked_regs() to
3445 	 * propagate min/max range.
3446 	 */
3447 	if (!src_reg->id && !tnum_is_const(src_reg->var_off))
3448 		src_reg->id = ++env->id_gen;
3449 }
3450 
3451 static void save_register_state(struct bpf_verifier_env *env,
3452 				struct bpf_func_state *state,
3453 				int spi, struct bpf_reg_state *reg,
3454 				int size)
3455 {
3456 	int i;
3457 
3458 	bpf_diag_mod_begin(env, &state->stack[spi].spilled_ptr, reg, BPF_DIAG_MOD_SPILL);
3459 	state->stack[spi].spilled_ptr = *reg;
3460 
3461 	for (i = BPF_REG_SIZE; i > BPF_REG_SIZE - size; i--)
3462 		state->stack[spi].slot_type[i - 1] = STACK_SPILL;
3463 
3464 	/* size < 8 bytes spill */
3465 	for (; i; i--)
3466 		mark_stack_slot_misc(env, &state->stack[spi].slot_type[i - 1]);
3467 
3468 	bpf_diag_mod_end(env);
3469 }
3470 
3471 static bool is_bpf_st_mem(struct bpf_insn *insn)
3472 {
3473 	return BPF_CLASS(insn->code) == BPF_ST && BPF_MODE(insn->code) == BPF_MEM;
3474 }
3475 
3476 static int get_reg_width(struct bpf_reg_state *reg)
3477 {
3478 	return fls64(reg_umax(reg));
3479 }
3480 
3481 /* See comment for mark_fastcall_pattern_for_call() */
3482 static void check_fastcall_stack_contract(struct bpf_verifier_env *env,
3483 					  struct bpf_func_state *state, int insn_idx, int off)
3484 {
3485 	struct bpf_subprog_info *subprog = &env->subprog_info[state->subprogno];
3486 	struct bpf_insn_aux_data *aux = env->insn_aux_data;
3487 	int i;
3488 
3489 	if (subprog->fastcall_stack_off <= off || aux[insn_idx].fastcall_pattern)
3490 		return;
3491 	/* access to the region [max_stack_depth .. fastcall_stack_off)
3492 	 * from something that is not a part of the fastcall pattern,
3493 	 * disable fastcall rewrites for current subprogram by setting
3494 	 * fastcall_stack_off to a value smaller than any possible offset.
3495 	 */
3496 	subprog->fastcall_stack_off = S16_MIN;
3497 	/* reset fastcall aux flags within subprogram,
3498 	 * happens at most once per subprogram
3499 	 */
3500 	for (i = subprog->start; i < (subprog + 1)->start; ++i) {
3501 		aux[i].fastcall_spills_num = 0;
3502 		aux[i].fastcall_pattern = 0;
3503 	}
3504 }
3505 
3506 static void scrub_special_slot(struct bpf_func_state *state, int spi)
3507 {
3508 	int i;
3509 
3510 	/* regular write of data into stack destroys any spilled ptr */
3511 	state->stack[spi].spilled_ptr.type = NOT_INIT;
3512 	/* Mark slots as STACK_MISC if they belonged to spilled ptr/dynptr/iter. */
3513 	if (is_stack_slot_special(&state->stack[spi]))
3514 		for (i = 0; i < BPF_REG_SIZE; i++)
3515 			scrub_spilled_slot(&state->stack[spi].slot_type[i]);
3516 }
3517 
3518 /* check_stack_{read,write}_fixed_off functions track spill/fill of registers,
3519  * stack boundary and alignment are checked in check_mem_access()
3520  */
3521 static int check_stack_write_fixed_off(struct bpf_verifier_env *env,
3522 				       /* stack frame we're writing to */
3523 				       struct bpf_func_state *state,
3524 				       int off, int size, int value_regno,
3525 				       int insn_idx)
3526 {
3527 	struct bpf_func_state *cur; /* state of the current function */
3528 	int i, slot = -off - 1, spi = slot / BPF_REG_SIZE, err;
3529 	struct bpf_insn *insn = &env->prog->insnsi[insn_idx];
3530 	struct bpf_reg_state *reg = NULL;
3531 	int insn_flags = INSN_F_STACK_ACCESS;
3532 	int hist_spi = spi, hist_frame = state->frameno;
3533 
3534 	/* caller checked that off % size == 0 and -MAX_BPF_STACK <= off < 0,
3535 	 * so it's aligned access and [off, off + size) are within stack limits
3536 	 */
3537 	if (!env->allow_ptr_leaks &&
3538 	    bpf_is_spilled_reg(&state->stack[spi]) &&
3539 	    !bpf_is_spilled_scalar_reg(&state->stack[spi]) &&
3540 	    size != BPF_REG_SIZE) {
3541 		const char *reason;
3542 
3543 		verbose(env, "attempt to corrupt spilled pointer on stack\n");
3544 		reason = bpf_diag_fmt(env,
3545 				      "This store writes %d bytes at stack offset %d into a stack slot that currently holds a spilled pointer. "
3546 			"Partial writes to spilled pointers are rejected because they can corrupt pointer metadata and leak kernel pointers.",
3547 			size, off);
3548 		bpf_diag_memory(
3549 			env, insn_idx, "stack spill corruption", reason,
3550 			"Write the full 8-byte spilled pointer slot, or use a separate stack slot for scalar data before overwriting only part of it.");
3551 		return -EACCES;
3552 	}
3553 
3554 	cur = env->cur_state->frame[env->cur_state->curframe];
3555 	if (value_regno >= 0)
3556 		reg = &cur->regs[value_regno];
3557 	if (!env->bypass_spec_v4) {
3558 		bool sanitize = reg && is_pointer_regtype(reg->type);
3559 
3560 		for (i = 0; i < size; i++) {
3561 			u8 type = state->stack[spi].slot_type[(slot - i) %
3562 							      BPF_REG_SIZE];
3563 
3564 			if (type != STACK_MISC && type != STACK_ZERO) {
3565 				sanitize = true;
3566 				break;
3567 			}
3568 		}
3569 
3570 		if (sanitize)
3571 			env->insn_aux_data[insn_idx].nospec_result = true;
3572 	}
3573 
3574 	err = destroy_if_dynptr_stack_slot(env, state, spi);
3575 	if (err)
3576 		return err;
3577 
3578 	check_fastcall_stack_contract(env, state, insn_idx, off);
3579 	mark_stack_slot_scratched(env, spi);
3580 	if (reg && !(off % BPF_REG_SIZE) && reg->type == SCALAR_VALUE && env->bpf_capable) {
3581 		bool reg_value_fits;
3582 
3583 		reg_value_fits = get_reg_width(reg) <= BITS_PER_BYTE * size;
3584 		/* Make sure that reg had an ID to build a relation on spill. */
3585 		if (reg_value_fits)
3586 			assign_scalar_id_before_mov(env, reg);
3587 		save_register_state(env, state, spi, reg, size);
3588 		/* Break the relation on a narrowing spill. */
3589 		if (!reg_value_fits)
3590 			clear_scalar_id(&state->stack[spi].spilled_ptr);
3591 	} else if (!reg && !(off % BPF_REG_SIZE) && is_bpf_st_mem(insn) &&
3592 		   env->bpf_capable) {
3593 		struct bpf_reg_state *tmp_reg = &env->fake_reg[0];
3594 
3595 		memset(tmp_reg, 0, sizeof(*tmp_reg));
3596 		__mark_reg_known(tmp_reg, insn->imm);
3597 		tmp_reg->type = SCALAR_VALUE;
3598 		save_register_state(env, state, spi, tmp_reg, size);
3599 	} else if (reg && is_pointer_regtype(reg->type)) {
3600 		/* register containing pointer is being spilled into stack */
3601 		if (size != BPF_REG_SIZE) {
3602 			verbose_linfo(env, insn_idx, "; ");
3603 			verbose(env, "invalid size of register spill\n");
3604 			return -EACCES;
3605 		}
3606 		if (state != cur && reg->type == PTR_TO_STACK) {
3607 			verbose(env, "cannot spill pointers to stack into stack frame of the caller\n");
3608 			return -EINVAL;
3609 		}
3610 		save_register_state(env, state, spi, reg, size);
3611 	} else {
3612 		u8 type = STACK_MISC;
3613 
3614 		if (bpf_is_spilled_reg(&state->stack[spi]))
3615 			bpf_diag_record_scrub(env, &state->stack[spi].spilled_ptr,
3616 					      BPF_DIAG_MOD_WRITE);
3617 		scrub_special_slot(state, spi);
3618 
3619 		/* when we zero initialize stack slots mark them as such */
3620 		if ((reg && bpf_register_is_null(reg)) ||
3621 		    (!reg && is_bpf_st_mem(insn) && insn->imm == 0)) {
3622 			/* STACK_ZERO case happened because register spill
3623 			 * wasn't properly aligned at the stack slot boundary,
3624 			 * so it's not a register spill anymore; force
3625 			 * originating register to be precise to make
3626 			 * STACK_ZERO correct for subsequent states
3627 			 */
3628 			err = mark_chain_precision(env, value_regno);
3629 			if (err)
3630 				return err;
3631 			type = STACK_ZERO;
3632 		}
3633 
3634 		/* Mark slots affected by this stack write. */
3635 		for (i = 0; i < size; i++)
3636 			state->stack[spi].slot_type[(slot - i) % BPF_REG_SIZE] = type;
3637 		insn_flags = 0; /* not a register spill */
3638 	}
3639 
3640 	if (insn_flags)
3641 		return bpf_push_jmp_history(env, env->cur_state, insn_flags,
3642 					    hist_spi, hist_frame, 0);
3643 	return 0;
3644 }
3645 
3646 /* Write the stack: 'stack[ptr_reg + off] = value_regno'. 'ptr_reg' is
3647  * known to contain a variable offset.
3648  * This function checks whether the write is permitted and conservatively
3649  * tracks the effects of the write, considering that each stack slot in the
3650  * dynamic range is potentially written to.
3651  *
3652  * 'value_regno' can be -1, meaning that an unknown value is being written to
3653  * the stack.
3654  *
3655  * Spilled pointers in range are not marked as written because we don't know
3656  * what's going to be actually written. This means that read propagation for
3657  * future reads cannot be terminated by this write.
3658  *
3659  * For privileged programs, uninitialized stack slots are considered
3660  * initialized by this write (even though we don't know exactly what offsets
3661  * are going to be written to). The idea is that we don't want the verifier to
3662  * reject future reads that access slots written to through variable offsets.
3663  */
3664 static int check_stack_write_var_off(struct bpf_verifier_env *env,
3665 				     /* func where register points to */
3666 				     struct bpf_func_state *state,
3667 				     struct bpf_reg_state *ptr_reg, int off, int size,
3668 				     int value_regno, int insn_idx)
3669 {
3670 	struct bpf_func_state *cur; /* state of the current function */
3671 	int min_off, max_off;
3672 	int i, err;
3673 	struct bpf_reg_state *value_reg = NULL;
3674 	struct bpf_insn *insn = &env->prog->insnsi[insn_idx];
3675 	bool writing_zero = false;
3676 	/* set if the fact that we're writing a zero is used to let any
3677 	 * stack slots remain STACK_ZERO
3678 	 */
3679 	bool zero_used = false;
3680 
3681 	cur = env->cur_state->frame[env->cur_state->curframe];
3682 	min_off = reg_smin(ptr_reg) + off;
3683 	max_off = reg_smax(ptr_reg) + off + size;
3684 	if (value_regno >= 0)
3685 		value_reg = &cur->regs[value_regno];
3686 	if ((value_reg && bpf_register_is_null(value_reg)) ||
3687 	    (!value_reg && is_bpf_st_mem(insn) && insn->imm == 0))
3688 		writing_zero = true;
3689 
3690 	for (i = min_off; i < max_off; i++) {
3691 		int spi;
3692 
3693 		spi = bpf_get_spi(i);
3694 		err = destroy_if_dynptr_stack_slot(env, state, spi);
3695 		if (err)
3696 			return err;
3697 	}
3698 
3699 	check_fastcall_stack_contract(env, state, insn_idx, min_off);
3700 	/* Variable offset writes destroy any spilled pointers in range. */
3701 	for (i = min_off; i < max_off; i++) {
3702 		u8 new_type, *stype;
3703 		int slot, spi;
3704 
3705 		slot = -i - 1;
3706 		spi = slot / BPF_REG_SIZE;
3707 		stype = &state->stack[spi].slot_type[slot % BPF_REG_SIZE];
3708 		mark_stack_slot_scratched(env, spi);
3709 
3710 		if (!env->allow_ptr_leaks && *stype != STACK_MISC && *stype != STACK_ZERO) {
3711 			/* Reject the write if range we may write to has not
3712 			 * been initialized beforehand. If we didn't reject
3713 			 * here, the ptr status would be erased below (even
3714 			 * though not all slots are actually overwritten),
3715 			 * possibly opening the door to leaks.
3716 			 *
3717 			 * We do however catch STACK_INVALID case below, and
3718 			 * only allow reading possibly uninitialized memory
3719 			 * later for CAP_PERFMON, as the write may not happen to
3720 			 * that slot.
3721 			 */
3722 			verbose(env, "spilled ptr in range of var-offset stack write; insn %d, ptr off: %d",
3723 				insn_idx, i);
3724 			return -EINVAL;
3725 		}
3726 
3727 		/* If writing_zero and the spi slot contains a spill of value 0,
3728 		 * maintain the spill type.
3729 		 */
3730 		if (writing_zero && *stype == STACK_SPILL &&
3731 		    bpf_is_spilled_scalar_reg(&state->stack[spi])) {
3732 			struct bpf_reg_state *spill_reg = &state->stack[spi].spilled_ptr;
3733 
3734 			if (tnum_is_const(spill_reg->var_off) && spill_reg->var_off.value == 0) {
3735 				zero_used = true;
3736 				continue;
3737 			}
3738 		}
3739 
3740 		/*
3741 		 * Scrub slots if variable-offset stack write goes over spilled pointers.
3742 		 * Otherwise bpf_is_spilled_reg() may == true && spilled_ptr.type == NOT_INIT
3743 		 * and valid program is rejected by check_stack_read_fixed_off()
3744 		 * with obscure "invalid size of register fill" message.
3745 		 */
3746 		scrub_special_slot(state, spi);
3747 
3748 		/* Update the slot type. */
3749 		new_type = STACK_MISC;
3750 		if (writing_zero && *stype == STACK_ZERO) {
3751 			new_type = STACK_ZERO;
3752 			zero_used = true;
3753 		}
3754 		/* If the slot is STACK_INVALID, we check whether it's OK to
3755 		 * pretend that it will be initialized by this write. The slot
3756 		 * might not actually be written to, and so if we mark it as
3757 		 * initialized future reads might leak uninitialized memory.
3758 		 * For privileged programs, we will accept such reads to slots
3759 		 * that may or may not be written because, if we're reject
3760 		 * them, the error would be too confusing.
3761 		 * Conservatively, treat STACK_POISON in a similar way.
3762 		 */
3763 		if ((*stype == STACK_INVALID || *stype == STACK_POISON) &&
3764 		    !env->allow_uninit_stack) {
3765 			verbose(env, "uninit stack in range of var-offset write prohibited for !root; insn %d, off: %d",
3766 					insn_idx, i);
3767 			return -EINVAL;
3768 		}
3769 		*stype = new_type;
3770 	}
3771 	if (zero_used) {
3772 		/* backtracking doesn't work for STACK_ZERO yet. */
3773 		err = mark_chain_precision(env, value_regno);
3774 		if (err)
3775 			return err;
3776 	}
3777 	bpf_diag_record_scrub_stack(env, state, min_off, max_off,
3778 				    BPF_DIAG_MOD_VAR_WRITE);
3779 	return 0;
3780 }
3781 
3782 /* When register 'dst_regno' is assigned some values from stack[min_off,
3783  * max_off), we set the register's type according to the types of the
3784  * respective stack slots. If all the stack values are known to be zeros, then
3785  * so is the destination reg. Otherwise, the register is considered to be
3786  * SCALAR. This function does not deal with register filling; the caller must
3787  * ensure that all spilled registers in the stack range have been marked as
3788  * read.
3789  *
3790  * STACK_SPILL bytes backed by spilled scalar const zeroes are also considered
3791  * zero bytes. In that case, mark the contributing stack slots precise so
3792  * pruning cannot reuse a zero-spill state for a later non-zero spill state.
3793  *
3794  * Returns an error if precision backtracking fails.
3795  */
3796 static int mark_reg_stack_read(struct bpf_verifier_env *env,
3797 			       /* func where src register points to */
3798 			       struct bpf_func_state *ptr_state,
3799 			       int min_off, int max_off, int dst_regno)
3800 {
3801 	struct bpf_verifier_state *vstate = env->cur_state;
3802 	struct bpf_func_state *state = vstate->frame[vstate->curframe];
3803 	u64 zero_spill_mask = 0;
3804 	int i, slot, spi;
3805 	u8 *stype;
3806 	int zeros = 0;
3807 
3808 	for (i = min_off; i < max_off; i++) {
3809 		slot = -i - 1;
3810 		spi = slot / BPF_REG_SIZE;
3811 		mark_stack_slot_scratched(env, spi);
3812 		stype = ptr_state->stack[spi].slot_type;
3813 		if (stype[slot % BPF_REG_SIZE] == STACK_ZERO) {
3814 			zeros++;
3815 			continue;
3816 		}
3817 		if (stype[slot % BPF_REG_SIZE] == STACK_SPILL &&
3818 		    bpf_register_is_null(&ptr_state->stack[spi].spilled_ptr)) {
3819 			zero_spill_mask |= 1ull << spi;
3820 			zeros++;
3821 			continue;
3822 		}
3823 		break;
3824 	}
3825 	if (zeros == max_off - min_off) {
3826 		/* Any access_size read into register is zero extended,
3827 		 * so the whole register == const_zero.
3828 		 */
3829 		__mark_reg_const_zero(env, &state->regs[dst_regno]);
3830 		if (zero_spill_mask) {
3831 			bpf_bt_set_frame_slot_mask(&env->bt, ptr_state->frameno, zero_spill_mask);
3832 			return mark_chain_precision_batch(env, env->cur_state);
3833 		}
3834 	} else {
3835 		/* have read misc data from the stack */
3836 		mark_reg_unknown(env, state->regs, dst_regno);
3837 	}
3838 
3839 	return 0;
3840 }
3841 
3842 static void bpf_diag_stack_read_uninit(struct bpf_verifier_env *env, int off, int i,
3843 				       int size)
3844 {
3845 	const char *reason;
3846 
3847 	reason = bpf_diag_fmt(env,
3848 			      "This rejected read uses %d bytes at stack offset %d, but byte %d in that range is uninitialized on this path. "
3849 		"Programs loaded with CAP_PERFMON can be allowed to read uninitialized stack bytes, but this program is being rejected without that allowance.",
3850 		size, off, i);
3851 	bpf_diag_memory(
3852 		env, env->insn_idx, "uninitialized stack read", reason,
3853 		"Initialize every byte in the stack range before reading it, adjust the offset and size so the read covers only initialized bytes, "
3854 		"or load with CAP_PERFMON if uninitialized stack reads are intended.");
3855 }
3856 
3857 /* Read the stack at 'off' and put the results into the register indicated by
3858  * 'dst_regno'. It handles reg filling if the addressed stack slot is a
3859  * spilled reg.
3860  *
3861  * 'dst_regno' can be -1, meaning that the read value is not going to a
3862  * register.
3863  *
3864  * The access is assumed to be within the current stack bounds.
3865  */
3866 static int check_stack_read_fixed_off(struct bpf_verifier_env *env,
3867 				      /* func where src register points to */
3868 				      struct bpf_func_state *reg_state,
3869 				      int off, int size, int dst_regno)
3870 {
3871 	struct bpf_verifier_state *vstate = env->cur_state;
3872 	struct bpf_func_state *state = vstate->frame[vstate->curframe];
3873 	int i, slot = -off - 1, spi = slot / BPF_REG_SIZE;
3874 	struct bpf_reg_state *reg;
3875 	u8 *stype, type;
3876 	int err;
3877 	int insn_flags = INSN_F_STACK_ACCESS;
3878 	int hist_spi = spi, hist_frame = reg_state->frameno;
3879 
3880 	stype = reg_state->stack[spi].slot_type;
3881 	reg = &reg_state->stack[spi].spilled_ptr;
3882 
3883 	mark_stack_slot_scratched(env, spi);
3884 	check_fastcall_stack_contract(env, state, env->insn_idx, off);
3885 
3886 	/*
3887 	 * Refine the in-progress load record's origin to the source stack slot.
3888 	 */
3889 	if (dst_regno >= 0)
3890 		bpf_diag_mod_begin(env, &state->regs[dst_regno], reg, BPF_DIAG_MOD_WRITE);
3891 
3892 	if (bpf_is_spilled_reg(&reg_state->stack[spi])) {
3893 		u8 spill_size = 1;
3894 
3895 		for (i = BPF_REG_SIZE - 1; i > 0 && stype[i - 1] == STACK_SPILL; i--)
3896 			spill_size++;
3897 
3898 		if (size != BPF_REG_SIZE || spill_size != BPF_REG_SIZE) {
3899 			if (reg->type != SCALAR_VALUE) {
3900 				verbose_linfo(env, env->insn_idx, "; ");
3901 				verbose(env, "invalid size of register fill\n");
3902 				return -EACCES;
3903 			}
3904 
3905 			if (dst_regno < 0)
3906 				return 0;
3907 
3908 			if (size <= spill_size &&
3909 			    bpf_stack_narrow_access_ok(off, size, spill_size)) {
3910 				if (env->bpf_capable && size == 4 && spill_size == 4 &&
3911 				    get_reg_width(reg) <= 32)
3912 					/* Ensure stack slot has an ID to build a relation
3913 					 * with the destination register on fill.
3914 					 */
3915 					assign_scalar_id_before_mov(env, reg);
3916 				state->regs[dst_regno] = *reg;
3917 
3918 				/* Break the relation on a narrowing fill.
3919 				 * coerce_reg_to_size will adjust the boundaries.
3920 				 */
3921 				if (get_reg_width(reg) > size * BITS_PER_BYTE)
3922 					clear_scalar_id(&state->regs[dst_regno]);
3923 			} else {
3924 				int spill_cnt = 0, zero_cnt = 0;
3925 
3926 				for (i = 0; i < size; i++) {
3927 					type = stype[(slot - i) % BPF_REG_SIZE];
3928 					if (type == STACK_SPILL) {
3929 						spill_cnt++;
3930 						continue;
3931 					}
3932 					if (type == STACK_MISC)
3933 						continue;
3934 					if (type == STACK_ZERO) {
3935 						zero_cnt++;
3936 						continue;
3937 					}
3938 					if (type == STACK_INVALID && env->allow_uninit_stack)
3939 						continue;
3940 					if (type == STACK_POISON) {
3941 						verbose(env, "reading from stack off %d+%d size %d, slot poisoned by dead code elimination\n",
3942 							off, i, size);
3943 					} else {
3944 						verbose(env, "invalid read from stack off %d+%d size %d\n",
3945 							off, i, size);
3946 						bpf_diag_stack_read_uninit(env, off, i, size);
3947 					}
3948 					return -EACCES;
3949 				}
3950 
3951 				if (spill_cnt == size &&
3952 				    tnum_is_const(reg->var_off) && reg->var_off.value == 0) {
3953 					__mark_reg_const_zero(env, &state->regs[dst_regno]);
3954 					/* this IS register fill, so keep insn_flags */
3955 				} else if (zero_cnt == size) {
3956 					/* similarly to mark_reg_stack_read(), preserve zeroes */
3957 					__mark_reg_const_zero(env, &state->regs[dst_regno]);
3958 					insn_flags = 0; /* not restoring original register state */
3959 				} else {
3960 					err = mark_reg_stack_read(env, reg_state, off, off + size,
3961 								  dst_regno);
3962 					if (err)
3963 						return err;
3964 					insn_flags = 0; /* not restoring original register state */
3965 				}
3966 			}
3967 		} else if (dst_regno >= 0) {
3968 			/* restore register state from stack */
3969 			if (env->bpf_capable)
3970 				/* Ensure stack slot has an ID to build a relation
3971 				 * with the destination register on fill.
3972 				 */
3973 				assign_scalar_id_before_mov(env, reg);
3974 			state->regs[dst_regno] = *reg;
3975 			/* mark reg as written since spilled pointer state likely
3976 			 * has its liveness marks cleared by is_state_visited()
3977 			 * which resets stack/reg liveness for state transitions
3978 			 */
3979 		} else if (__is_pointer_value(env->allow_ptr_leaks, reg)) {
3980 			/* If dst_regno==-1, the caller is asking us whether
3981 			 * it is acceptable to use this value as a SCALAR_VALUE
3982 			 * (e.g. for XADD).
3983 			 * We must not allow unprivileged callers to do that
3984 			 * with spilled pointers.
3985 			 */
3986 			verbose(env, "leaking pointer from stack off %d\n",
3987 				off);
3988 			return -EACCES;
3989 		}
3990 	} else {
3991 		for (i = 0; i < size; i++) {
3992 			type = stype[(slot - i) % BPF_REG_SIZE];
3993 			if (type == STACK_MISC)
3994 				continue;
3995 			if (type == STACK_ZERO)
3996 				continue;
3997 			if (type == STACK_INVALID && env->allow_uninit_stack)
3998 				continue;
3999 			if (type == STACK_POISON) {
4000 				verbose(env, "reading from stack off %d+%d size %d, slot poisoned by dead code elimination\n",
4001 					off, i, size);
4002 			} else {
4003 				verbose(env, "invalid read from stack off %d+%d size %d\n",
4004 					off, i, size);
4005 				bpf_diag_stack_read_uninit(env, off, i, size);
4006 			}
4007 			return -EACCES;
4008 		}
4009 		if (dst_regno >= 0) {
4010 			err = mark_reg_stack_read(env, reg_state, off, off + size, dst_regno);
4011 			if (err)
4012 				return err;
4013 		}
4014 		insn_flags = 0; /* we are not restoring spilled register */
4015 	}
4016 	if (insn_flags)
4017 		return bpf_push_jmp_history(env, env->cur_state, insn_flags,
4018 					    hist_spi, hist_frame, 0);
4019 	return 0;
4020 }
4021 
4022 enum bpf_access_src {
4023 	ACCESS_DIRECT = 1,  /* the access is performed by an instruction */
4024 	ACCESS_HELPER = 2,  /* the access is performed by a helper */
4025 };
4026 
4027 static int check_stack_range_initialized(struct bpf_verifier_env *env, struct bpf_reg_state *reg,
4028 					 argno_t argno, int off, int access_size,
4029 					 bool zero_size_allowed,
4030 					 enum bpf_access_type type,
4031 					 struct bpf_call_arg_meta *meta);
4032 
4033 static struct bpf_reg_state *reg_state(struct bpf_verifier_env *env, int regno)
4034 {
4035 	return cur_regs(env) + regno;
4036 }
4037 
4038 /* Read the stack at 'reg + off' and put the result into the register
4039  * 'dst_regno'.
4040  * 'off' includes the pointer register's fixed offset(i.e. 'reg->off'),
4041  * but not its variable offset.
4042  * 'size' is assumed to be <= reg size and the access is assumed to be aligned.
4043  *
4044  * As opposed to check_stack_read_fixed_off, this function doesn't deal with
4045  * filling registers (i.e. reads of spilled register cannot be detected when
4046  * the offset is not fixed). We conservatively mark 'dst_regno' as containing
4047  * SCALAR_VALUE. That's why we assert that the 'reg' has a variable
4048  * offset; for a fixed offset check_stack_read_fixed_off should be used
4049  * instead.
4050  */
4051 static int check_stack_read_var_off(struct bpf_verifier_env *env, struct bpf_reg_state *reg,
4052 				    argno_t ptr_argno, int off, int size, int dst_regno)
4053 {
4054 	struct bpf_func_state *ptr_state = bpf_func(env, reg);
4055 	int err;
4056 	int min_off, max_off;
4057 
4058 	/* Note that we pass a NULL meta, so raw access will not be permitted.
4059 	 */
4060 	err = check_stack_range_initialized(env, reg, ptr_argno, off, size,
4061 					    false, BPF_READ, NULL);
4062 	if (err)
4063 		return err;
4064 
4065 	min_off = reg_smin(reg) + off;
4066 	max_off = reg_smax(reg) + off;
4067 	err = mark_reg_stack_read(env, ptr_state, min_off, max_off + size,
4068 				  dst_regno);
4069 	if (err)
4070 		return err;
4071 	check_fastcall_stack_contract(env, ptr_state, env->insn_idx, min_off);
4072 	return 0;
4073 }
4074 
4075 /* check_stack_read dispatches to check_stack_read_fixed_off or
4076  * check_stack_read_var_off.
4077  *
4078  * The caller must ensure that the offset falls within the allocated stack
4079  * bounds.
4080  *
4081  * 'dst_regno' is a register which will receive the value from the stack. It
4082  * can be -1, meaning that the read value is not going to a register.
4083  */
4084 static int check_stack_read(struct bpf_verifier_env *env,
4085 			    struct bpf_reg_state *reg, argno_t ptr_argno, int off, int size,
4086 			    int dst_regno)
4087 {
4088 	struct bpf_func_state *state = bpf_func(env, reg);
4089 	int err;
4090 	/* Some accesses are only permitted with a static offset. */
4091 	bool var_off = !tnum_is_const(reg->var_off);
4092 
4093 	/* The offset is required to be static when reads don't go to a
4094 	 * register, in order to not leak pointers (see
4095 	 * check_stack_read_fixed_off).
4096 	 */
4097 	if (dst_regno < 0 && var_off) {
4098 		const char *reason;
4099 		char tn_buf[48];
4100 
4101 		tnum_strn(tn_buf, sizeof(tn_buf), reg->var_off);
4102 		verbose(env, "variable offset stack pointer cannot be passed into helper function; var_off=%s off=%d size=%d\n",
4103 			tn_buf, off, size);
4104 		reason = bpf_diag_fmt(env,
4105 				      "The helper would access the stack through variable offset %s plus fixed offset %d and size %d. "
4106 			"Helper stack memory arguments require a constant stack offset and a precise initialized range.",
4107 			tn_buf, off, size);
4108 		bpf_diag_memory(
4109 			env, env->insn_idx, "variable stack access", reason,
4110 			"Use a fixed stack offset for helper memory arguments, or copy the needed bytes into a fixed stack slot first.");
4111 		return -EACCES;
4112 	}
4113 	/* Variable offset is prohibited for unprivileged mode for simplicity
4114 	 * since it requires corresponding support in Spectre masking for stack
4115 	 * ALU. See also retrieve_ptr_limit(). The check in
4116 	 * check_stack_access_for_ptr_arithmetic() called by
4117 	 * adjust_ptr_min_max_vals() prevents users from creating stack pointers
4118 	 * with variable offsets, therefore no check is required here. Further,
4119 	 * just checking it here would be insufficient as speculative stack
4120 	 * writes could still lead to unsafe speculative behaviour.
4121 	 */
4122 	if (!var_off) {
4123 		off += reg->var_off.value;
4124 		err = check_stack_read_fixed_off(env, state, off, size,
4125 						 dst_regno);
4126 	} else {
4127 		/* Variable offset stack reads need more conservative handling
4128 		 * than fixed offset ones. Note that dst_regno >= 0 on this
4129 		 * branch.
4130 		 */
4131 		err = check_stack_read_var_off(env, reg, ptr_argno, off, size,
4132 					       dst_regno);
4133 	}
4134 	return err;
4135 }
4136 
4137 /* check_stack_write dispatches to check_stack_write_fixed_off or
4138  * check_stack_write_var_off.
4139  *
4140  * 'reg' is the register used as a pointer into the stack.
4141  * 'value_regno' is the register whose value we're writing to the stack. It can
4142  * be -1, meaning that we're not writing from a register.
4143  *
4144  * The caller must ensure that the offset falls within the maximum stack size.
4145  */
4146 static int check_stack_write(struct bpf_verifier_env *env,
4147 			     struct bpf_reg_state *reg, int off, int size,
4148 			     int value_regno, int insn_idx)
4149 {
4150 	struct bpf_func_state *state = bpf_func(env, reg);
4151 	int err;
4152 
4153 	if (tnum_is_const(reg->var_off)) {
4154 		off += reg->var_off.value;
4155 		err = check_stack_write_fixed_off(env, state, off, size,
4156 						  value_regno, insn_idx);
4157 	} else {
4158 		/* Variable offset stack reads need more conservative handling
4159 		 * than fixed offset ones.
4160 		 */
4161 		err = check_stack_write_var_off(env, state,
4162 						reg, off, size,
4163 						value_regno, insn_idx);
4164 	}
4165 	return err;
4166 }
4167 
4168 /*
4169  * Write a value to the outgoing stack arg area.
4170  * off is a negative offset from r11 (e.g. -8 for arg6, -16 for arg7).
4171  */
4172 static int check_stack_arg_write(struct bpf_verifier_env *env, struct bpf_func_state *state,
4173 				 int off, struct bpf_reg_state *value_reg)
4174 {
4175 	int max_stack_arg_regs = MAX_BPF_FUNC_ARGS - MAX_BPF_FUNC_REG_ARGS;
4176 	struct bpf_subprog_info *subprog = &env->subprog_info[state->subprogno];
4177 	int spi = -off / BPF_REG_SIZE - 1;
4178 	struct bpf_reg_state *arg;
4179 	int err;
4180 
4181 	if (spi >= max_stack_arg_regs) {
4182 		verbose(env, "stack arg write offset %d exceeds max %d stack args\n",
4183 			off, max_stack_arg_regs);
4184 		return -EINVAL;
4185 	}
4186 
4187 	err = grow_stack_arg_slots(env, state, spi + 1);
4188 	if (err)
4189 		return err;
4190 
4191 	/* Track the max outgoing stack arg slot count. */
4192 	if (spi + 1 > subprog->max_out_stack_arg_cnt)
4193 		subprog->max_out_stack_arg_cnt = spi + 1;
4194 
4195 	arg = &state->stack_arg_regs[spi];
4196 	bpf_diag_mod_begin(env, arg, value_reg, BPF_DIAG_MOD_WRITE);
4197 
4198 	if (value_reg) {
4199 		state->stack_arg_regs[spi] = *value_reg;
4200 	} else {
4201 		/* BPF_ST: store immediate, treat as scalar */
4202 		arg->type = SCALAR_VALUE;
4203 		__mark_reg_known(arg, env->prog->insnsi[env->insn_idx].imm);
4204 	}
4205 	bpf_diag_mod_end(env);
4206 	state->no_stack_arg_load = true;
4207 	return bpf_push_jmp_history(env, env->cur_state,
4208 				    INSN_F_STACK_ARG_ACCESS, spi, 0, 0);
4209 }
4210 
4211 /*
4212  * Read a value from the incoming stack arg area.
4213  * off is a positive offset from r11 (e.g. +8 for arg6, +16 for arg7).
4214  */
4215 static int check_stack_arg_read(struct bpf_verifier_env *env, struct bpf_func_state *state,
4216 				int off, int dst_regno)
4217 {
4218 	struct bpf_subprog_info *subprog = &env->subprog_info[state->subprogno];
4219 	struct bpf_verifier_state *vstate = env->cur_state;
4220 	int spi = off / BPF_REG_SIZE - 1;
4221 	struct bpf_func_state *caller, *cur;
4222 	struct bpf_reg_state *arg;
4223 
4224 	if (state->no_stack_arg_load) {
4225 		verbose(env, "r11 load must be before any r11 store or call insn\n");
4226 		return -EINVAL;
4227 	}
4228 
4229 	if (spi + 1 > bpf_in_stack_arg_cnt(subprog)) {
4230 		verbose(env, "invalid read from stack arg off %d depth %d\n",
4231 			off, bpf_in_stack_arg_cnt(subprog) * BPF_REG_SIZE);
4232 		return -EACCES;
4233 	}
4234 
4235 	caller = vstate->frame[vstate->curframe - 1];
4236 	arg = &caller->stack_arg_regs[spi];
4237 	cur = vstate->frame[vstate->curframe];
4238 	bpf_diag_mod_begin(env, &cur->regs[dst_regno], arg, BPF_DIAG_MOD_WRITE);
4239 	cur->regs[dst_regno] = *arg;
4240 	bpf_diag_mod_end(env);
4241 	return bpf_push_jmp_history(env, env->cur_state,
4242 				    INSN_F_STACK_ARG_ACCESS, spi, 0, 0);
4243 }
4244 
4245 static int mark_stack_arg_precision(struct bpf_verifier_env *env, int arg_idx)
4246 {
4247 	struct bpf_func_state *caller = cur_func(env);
4248 	int spi = arg_idx - MAX_BPF_FUNC_REG_ARGS;
4249 
4250 	bt_set_frame_stack_arg_slot(&env->bt, caller->frameno, spi);
4251 	return mark_chain_precision_batch(env, env->cur_state);
4252 }
4253 
4254 static int mark_arg_precision(struct bpf_verifier_env *env, argno_t argno)
4255 {
4256 	int regno = reg_from_argno(argno);
4257 
4258 	if (regno >= 0)
4259 		return mark_chain_precision(env, regno);
4260 	return mark_stack_arg_precision(env, arg_idx_from_argno(argno));
4261 }
4262 
4263 static int check_outgoing_stack_args(struct bpf_verifier_env *env, struct bpf_func_state *caller,
4264 				     int nargs, const char *callee_name, const struct btf *btf,
4265 				     const struct btf_param *args)
4266 {
4267 	int i, spi;
4268 
4269 	for (i = MAX_BPF_FUNC_REG_ARGS; i < nargs; i++) {
4270 		spi = i - MAX_BPF_FUNC_REG_ARGS;
4271 		if (spi >= caller->out_stack_arg_cnt ||
4272 		    caller->stack_arg_regs[spi].type == NOT_INIT) {
4273 			const char *arg_name = NULL;
4274 
4275 			if (args && args[i].name_off)
4276 				arg_name = btf_name_by_offset(btf, args[i].name_off);
4277 			verbose(env, "callee expects %d args, stack arg%d is not initialized\n",
4278 				nargs, spi + 1);
4279 			bpf_diag_stack_arg_uninit(env, env->insn_idx, nargs, spi,
4280 						  callee_name, arg_name);
4281 			return -EFAULT;
4282 		}
4283 	}
4284 
4285 	return 0;
4286 }
4287 
4288 static struct bpf_reg_state *get_func_arg_reg(struct bpf_func_state *caller,
4289 					      struct bpf_reg_state *regs, int arg)
4290 {
4291 	if (arg < MAX_BPF_FUNC_REG_ARGS)
4292 		return &regs[arg + 1];
4293 
4294 	return &caller->stack_arg_regs[arg - MAX_BPF_FUNC_REG_ARGS];
4295 }
4296 
4297 static int check_map_access_type(struct bpf_verifier_env *env, struct bpf_reg_state *reg,
4298 				 int off, int size, enum bpf_access_type type)
4299 {
4300 	struct bpf_map *map = reg->map_ptr;
4301 	u32 cap = bpf_map_flags_to_cap(map);
4302 
4303 	if (type == BPF_WRITE && !(cap & BPF_MAP_CAN_WRITE)) {
4304 		verbose(env, "write into map forbidden, value_size=%d off=%lld size=%d\n",
4305 			map->value_size, reg_smin(reg) + off, size);
4306 		return -EACCES;
4307 	}
4308 
4309 	if (type == BPF_READ && !(cap & BPF_MAP_CAN_READ)) {
4310 		verbose(env, "read from map forbidden, value_size=%d off=%lld size=%d\n",
4311 			map->value_size, reg_smin(reg) + off, size);
4312 		return -EACCES;
4313 	}
4314 
4315 	return 0;
4316 }
4317 
4318 /* check read/write into memory region (e.g., map value, ringbuf sample, etc) */
4319 static int __check_mem_access(struct bpf_verifier_env *env, struct bpf_reg_state *reg, argno_t argno,
4320 			      int off, int size, u32 mem_size,
4321 			      bool zero_size_allowed)
4322 {
4323 	bool size_ok = size > 0 || (size == 0 && zero_size_allowed);
4324 
4325 	if (off >= 0 && size_ok && (u64)off + size <= mem_size)
4326 		return 0;
4327 
4328 	switch (reg->type) {
4329 	case PTR_TO_MAP_KEY:
4330 		verbose(env, "invalid access to map key, key_size=%d off=%d size=%d\n",
4331 			mem_size, off, size);
4332 		break;
4333 	case PTR_TO_MAP_VALUE:
4334 		verbose(env, "invalid access to map value, value_size=%d off=%d size=%d\n",
4335 			mem_size, off, size);
4336 		break;
4337 	case PTR_TO_PACKET:
4338 	case PTR_TO_PACKET_META:
4339 	case PTR_TO_PACKET_END:
4340 		verbose(env, "invalid access to packet, off=%d size=%d, %s(id=%d,off=%d,r=%d)\n",
4341 			off, size, reg_arg_name(env, argno), reg->id, off, mem_size);
4342 		break;
4343 	case PTR_TO_CTX:
4344 		verbose(env, "invalid access to context, ctx_size=%d off=%d size=%d\n",
4345 			mem_size, off, size);
4346 		break;
4347 	case PTR_TO_MEM:
4348 	default:
4349 		verbose(env, "invalid access to memory, mem_size=%u off=%d size=%d\n",
4350 			mem_size, off, size);
4351 	}
4352 
4353 	return -EACCES;
4354 }
4355 
4356 /* check read/write into a memory region with possible variable offset */
4357 static int check_mem_region_access(struct bpf_verifier_env *env, struct bpf_reg_state *reg, argno_t argno,
4358 				   int off, int size, u32 mem_size,
4359 				   bool zero_size_allowed)
4360 {
4361 	const char *proof = "";
4362 	const char *start;
4363 	s64 max_start, max_end;
4364 	int err;
4365 
4366 	/* We may have adjusted the register pointing to memory region, so we
4367 	 * need to try adding each of min_value and max_value to off
4368 	 * to make sure our theoretical access will be safe.
4369 	 *
4370 	 * The minimum value is only important with signed
4371 	 * comparisons where we can't assume the floor of a
4372 	 * value is 0.  If we are using signed variables for our
4373 	 * index'es we need to make sure that whatever we use
4374 	 * will have a set floor within our range.
4375 	 */
4376 	if (reg_smin(reg) < 0 &&
4377 	    (reg_smin(reg) == S64_MIN ||
4378 	     (off + reg_smin(reg) != (s64)(s32)(off + reg_smin(reg))) ||
4379 	      reg_smin(reg) + off < 0)) {
4380 		verbose(env, "%s min value is negative, either use unsigned index or do a if (index >=0) check.\n",
4381 			reg_arg_name(env, argno));
4382 		err = -EACCES;
4383 		if (bpf_diag_enabled(env)) {
4384 			start = bpf_diag_fmt_s64_sum(env, reg_smin(reg), off);
4385 			proof = bpf_diag_fmt(
4386 				env, "the minimal bound for a memory access is a negative value: %s",
4387 				start);
4388 		}
4389 		goto report_error;
4390 	}
4391 
4392 	err = __check_mem_access(env, reg, argno, reg_smin(reg) + off, size,
4393 				 mem_size, zero_size_allowed);
4394 	if (err) {
4395 		verbose(env, "%s min value is outside of the allowed memory range\n",
4396 			reg_arg_name(env, argno));
4397 		if (bpf_diag_enabled(env)) {
4398 			start = bpf_diag_fmt_s64_sum(env, reg_smin(reg), off);
4399 			proof = bpf_diag_fmt(
4400 				env, "the minimal bound for a memory access is %s and is outside of the object of size %u",
4401 				start, mem_size);
4402 		}
4403 		goto report_error;
4404 	}
4405 
4406 	/* If we haven't set a max value then we need to bail since we can't be
4407 	 * sure we won't do bad things.
4408 	 * If reg_umax(reg) + off could overflow, treat that as unbounded too.
4409 	 */
4410 	if (reg_umax(reg) >= BPF_MAX_VAR_OFF) {
4411 		verbose(env, "%s unbounded memory access, make sure to bounds check any such access\n",
4412 			reg_arg_name(env, argno));
4413 		err = -EACCES;
4414 		if (bpf_diag_enabled(env))
4415 			proof = bpf_diag_fmt(
4416 				env, "the maximal bound for a memory access is %llu and exceeds maximum allowed offset of %u",
4417 				reg_umax(reg), BPF_MAX_VAR_OFF);
4418 		goto report_error;
4419 	}
4420 
4421 	err = __check_mem_access(env, reg, argno, reg_umax(reg) + off, size,
4422 				 mem_size, zero_size_allowed);
4423 	if (err) {
4424 		verbose(env, "%s max value is outside of the allowed memory range\n",
4425 			reg_arg_name(env, argno));
4426 		if (bpf_diag_enabled(env)) {
4427 			max_start = (s64)reg_umax(reg) + off;
4428 			max_end = max_start + size;
4429 			proof = bpf_diag_fmt(
4430 				env, "the maximal bound for a memory access is %lld: start %lld + access_size %d, beyond object_size %u",
4431 				max_end, max_start, size, mem_size);
4432 		}
4433 		goto report_error;
4434 	}
4435 
4436 	return 0;
4437 
4438 report_error:
4439 	bpf_diag_mem_bounds(env, env->insn_idx, reg_from_argno(argno),
4440 			    reg_arg_name(env, argno), reg_type_str(env, reg->type), proof,
4441 			    off, size, mem_size, reg);
4442 	return err;
4443 }
4444 
4445 static int __check_ptr_off_reg(struct bpf_verifier_env *env,
4446 			       const struct bpf_reg_state *reg, argno_t argno,
4447 			       bool fixed_off_ok)
4448 {
4449 	/* Access to this pointer-typed register or passing it to a helper
4450 	 * is only allowed in its original, unmodified form.
4451 	 */
4452 
4453 	if (!tnum_is_const(reg->var_off)) {
4454 		char tn_buf[48];
4455 
4456 		tnum_strn(tn_buf, sizeof(tn_buf), reg->var_off);
4457 		verbose(env, "variable %s access var_off=%s disallowed\n",
4458 			reg_type_str(env, reg->type), tn_buf);
4459 		return -EACCES;
4460 	}
4461 
4462 	if (reg_smin(reg) < 0) {
4463 		verbose(env, "negative offset %s ptr %s off=%lld disallowed\n",
4464 			reg_type_str(env, reg->type), reg_arg_name(env, argno), reg->var_off.value);
4465 		return -EACCES;
4466 	}
4467 
4468 	if (!fixed_off_ok && reg->var_off.value != 0) {
4469 		verbose(env, "dereference of modified %s ptr %s off=%lld disallowed\n",
4470 			reg_type_str(env, reg->type), reg_arg_name(env, argno), reg->var_off.value);
4471 		bpf_diag_invalid_deref(env, env->insn_idx, reg_from_argno(argno),
4472 				       reg_arg_name(env, argno), reg,
4473 					      BPF_DIAG_DEREF_MODIFIED_PTR, reg->var_off.value);
4474 		return -EACCES;
4475 	}
4476 
4477 	return 0;
4478 }
4479 
4480 static int check_ptr_off_reg(struct bpf_verifier_env *env,
4481 		             const struct bpf_reg_state *reg, int regno)
4482 {
4483 	return __check_ptr_off_reg(env, reg, argno_from_reg(regno), false);
4484 }
4485 
4486 static int map_kptr_match_type(struct bpf_verifier_env *env,
4487 			       struct btf_field *kptr_field,
4488 			       struct bpf_reg_state *reg, u32 regno)
4489 {
4490 	const char *targ_name = btf_type_name(kptr_field->kptr.btf, kptr_field->kptr.btf_id);
4491 	int perm_flags;
4492 	const char *reg_name = "";
4493 
4494 	if (base_type(reg->type) != PTR_TO_BTF_ID)
4495 		goto bad_type;
4496 
4497 	if (btf_is_kernel(reg->btf)) {
4498 		perm_flags = PTR_MAYBE_NULL | PTR_TRUSTED | MEM_RCU;
4499 
4500 		/* Only unreferenced case accepts untrusted pointers */
4501 		if (kptr_field->type == BPF_KPTR_UNREF)
4502 			perm_flags |= PTR_UNTRUSTED;
4503 	} else {
4504 		perm_flags = PTR_MAYBE_NULL | MEM_ALLOC;
4505 		if (kptr_field->type == BPF_KPTR_PERCPU)
4506 			perm_flags |= MEM_PERCPU;
4507 	}
4508 
4509 	if (type_flag(reg->type) & ~perm_flags)
4510 		goto bad_type;
4511 
4512 	/*
4513 	 * A BPF_KPTR_PERCPU field is read back as MEM_PERCPU, so the value
4514 	 * stored in it must carry the same flag.
4515 	 */
4516 	if ((kptr_field->type == BPF_KPTR_PERCPU) != !!(reg->type & MEM_PERCPU))
4517 		goto bad_type;
4518 
4519 	/* We need to verify reg->type and reg->btf, before accessing reg->btf */
4520 	reg_name = btf_type_name(reg->btf, reg->btf_id);
4521 
4522 	/* For ref_ptr case, release function check should ensure we get one
4523 	 * referenced PTR_TO_BTF_ID, and that its fixed offset is 0. For the
4524 	 * normal store of unreferenced kptr, we must ensure var_off is zero.
4525 	 * Since ref_ptr cannot be accessed directly by BPF insns, check for
4526 	 * reg->id is not needed here.
4527 	 */
4528 	if (__check_ptr_off_reg(env, reg, argno_from_reg(regno), true))
4529 		return -EACCES;
4530 
4531 	/* A full type match is needed, as BTF can be vmlinux, module or prog BTF, and
4532 	 * we also need to take into account the reg->var_off.
4533 	 *
4534 	 * We want to support cases like:
4535 	 *
4536 	 * struct foo {
4537 	 *         struct bar br;
4538 	 *         struct baz bz;
4539 	 * };
4540 	 *
4541 	 * struct foo *v;
4542 	 * v = func();	      // PTR_TO_BTF_ID
4543 	 * val->foo = v;      // reg->var_off is zero, btf and btf_id match type
4544 	 * val->bar = &v->br; // reg->var_off is still zero, but we need to retry with
4545 	 *                    // first member type of struct after comparison fails
4546 	 * val->baz = &v->bz; // reg->var_off is non-zero, so struct needs to be walked
4547 	 *                    // to match type
4548 	 *
4549 	 * In the kptr_ref case, check_func_arg_reg_off already ensures reg->var_off
4550 	 * is zero. We must also ensure that btf_struct_ids_match does not walk
4551 	 * the struct to match type against first member of struct, i.e. reject
4552 	 * second case from above. Hence, when type is BPF_KPTR_REF, we set
4553 	 * strict mode to true for type match.
4554 	 */
4555 	if (!btf_struct_ids_match(&env->log, reg->btf, reg->btf_id, reg->var_off.value,
4556 				  kptr_field->kptr.btf, kptr_field->kptr.btf_id,
4557 				  kptr_field->type != BPF_KPTR_UNREF,
4558 				  !type_is_alloc(reg->type)))
4559 		goto bad_type;
4560 	return 0;
4561 bad_type:
4562 	verbose(env, "invalid kptr access, R%d type=%s%s ", regno,
4563 		reg_type_str(env, reg->type), reg_name);
4564 	verbose(env, "expected=%s%s", reg_type_str(env, PTR_TO_BTF_ID), targ_name);
4565 	if (kptr_field->type == BPF_KPTR_UNREF)
4566 		verbose(env, " or %s%s\n", reg_type_str(env, PTR_TO_BTF_ID | PTR_UNTRUSTED),
4567 			targ_name);
4568 	else
4569 		verbose(env, "\n");
4570 	return -EINVAL;
4571 }
4572 
4573 static bool in_sleepable(struct bpf_verifier_env *env)
4574 {
4575 	return env->cur_state->in_sleepable;
4576 }
4577 
4578 /* The non-sleepable programs and sleepable programs with explicit bpf_rcu_read_lock()
4579  * can dereference RCU protected pointers and result is PTR_TRUSTED.
4580  */
4581 static bool in_rcu_cs(struct bpf_verifier_env *env)
4582 {
4583 	return env->cur_state->active_rcu_locks ||
4584 	       env->cur_state->active_preempt_locks ||
4585 	       env->cur_state->active_locks ||
4586 	       env->cur_state->active_irq_id ||
4587 	       !in_sleepable(env);
4588 }
4589 
4590 /* Once GCC supports btf_type_tag the following mechanism will be replaced with tag check */
4591 BTF_SET_START(rcu_protected_types)
4592 #ifdef CONFIG_NET
4593 BTF_ID(struct, prog_test_ref_kfunc)
4594 #endif
4595 #ifdef CONFIG_CGROUPS
4596 BTF_ID(struct, cgroup)
4597 #endif
4598 #ifdef CONFIG_BPF_JIT
4599 BTF_ID(struct, bpf_cpumask)
4600 #endif
4601 BTF_ID(struct, task_struct)
4602 #ifdef CONFIG_CRYPTO
4603 BTF_ID(struct, bpf_crypto_ctx)
4604 #endif
4605 #ifdef CONFIG_INET
4606 BTF_ID(struct, bpf_ksock)
4607 #endif
4608 BTF_SET_END(rcu_protected_types)
4609 
4610 static bool rcu_protected_object(const struct btf *btf, u32 btf_id)
4611 {
4612 	if (!btf_is_kernel(btf))
4613 		return true;
4614 	return btf_id_set_contains(&rcu_protected_types, btf_id);
4615 }
4616 
4617 static struct btf_record *kptr_pointee_btf_record(struct btf_field *kptr_field)
4618 {
4619 	struct btf_struct_meta *meta;
4620 
4621 	if (btf_is_kernel(kptr_field->kptr.btf))
4622 		return NULL;
4623 
4624 	meta = btf_find_struct_meta(kptr_field->kptr.btf,
4625 				    kptr_field->kptr.btf_id);
4626 
4627 	return meta ? meta->record : NULL;
4628 }
4629 
4630 static bool rcu_safe_kptr(const struct btf_field *field)
4631 {
4632 	const struct btf_field_kptr *kptr = &field->kptr;
4633 
4634 	return field->type == BPF_KPTR_PERCPU ||
4635 	       (field->type == BPF_KPTR_REF && rcu_protected_object(kptr->btf, kptr->btf_id));
4636 }
4637 
4638 static u32 btf_ld_kptr_type(struct bpf_verifier_env *env, struct btf_field *kptr_field)
4639 {
4640 	struct btf_record *rec;
4641 	u32 ret;
4642 
4643 	ret = PTR_MAYBE_NULL;
4644 	if (rcu_safe_kptr(kptr_field) && in_rcu_cs(env)) {
4645 		ret |= MEM_RCU;
4646 		if (kptr_field->type == BPF_KPTR_PERCPU)
4647 			ret |= MEM_PERCPU;
4648 		else if (!btf_is_kernel(kptr_field->kptr.btf))
4649 			ret |= MEM_ALLOC;
4650 
4651 		rec = kptr_pointee_btf_record(kptr_field);
4652 		if (rec && btf_record_has_field(rec, BPF_GRAPH_NODE))
4653 			ret |= NON_OWN_REF;
4654 	} else {
4655 		ret |= PTR_UNTRUSTED;
4656 	}
4657 
4658 	return ret;
4659 }
4660 
4661 static int mark_uptr_ld_reg(struct bpf_verifier_env *env, u32 regno,
4662 			    struct btf_field *field)
4663 {
4664 	struct bpf_reg_state *reg;
4665 	const struct btf_type *t;
4666 
4667 	t = btf_type_by_id(field->kptr.btf, field->kptr.btf_id);
4668 	mark_reg_known_zero(env, cur_regs(env), regno);
4669 	reg = reg_state(env, regno);
4670 	reg->type = PTR_TO_MEM | PTR_MAYBE_NULL;
4671 	reg->mem_size = t->size;
4672 	reg->id = ++env->id_gen;
4673 
4674 	return 0;
4675 }
4676 
4677 static int check_map_kptr_access(struct bpf_verifier_env *env,
4678 				 int value_regno, int insn_idx,
4679 				 struct btf_field *kptr_field)
4680 {
4681 	struct bpf_insn *insn = &env->prog->insnsi[insn_idx];
4682 	int class = BPF_CLASS(insn->code);
4683 	struct bpf_reg_state *val_reg;
4684 	int ret;
4685 
4686 	/* Things we already checked for in check_map_access and caller:
4687 	 *  - Reject cases where variable offset may touch kptr
4688 	 *  - size of access (must be BPF_DW)
4689 	 *  - tnum_is_const(reg->var_off)
4690 	 *  - kptr_field->offset == off + reg->var_off.value
4691 	 */
4692 	/* Only BPF_[LDX,STX,ST] | BPF_MEM | BPF_DW is supported */
4693 	if (BPF_MODE(insn->code) != BPF_MEM) {
4694 		verbose(env, "kptr in map can only be accessed using BPF_MEM instruction mode\n");
4695 		return -EACCES;
4696 	}
4697 
4698 	/* We only allow loading referenced kptr, since it will be marked as
4699 	 * untrusted, similar to unreferenced kptr.
4700 	 */
4701 	if (class != BPF_LDX &&
4702 	    (kptr_field->type == BPF_KPTR_REF || kptr_field->type == BPF_KPTR_PERCPU)) {
4703 		verbose(env, "store to referenced kptr disallowed\n");
4704 		return -EACCES;
4705 	}
4706 	if (class != BPF_LDX && kptr_field->type == BPF_UPTR) {
4707 		verbose(env, "store to uptr disallowed\n");
4708 		return -EACCES;
4709 	}
4710 
4711 	if (class == BPF_LDX) {
4712 		if (kptr_field->type == BPF_UPTR)
4713 			return mark_uptr_ld_reg(env, value_regno, kptr_field);
4714 
4715 		/* We can simply mark the value_regno receiving the pointer
4716 		 * value from map as PTR_TO_BTF_ID, with the correct type.
4717 		 */
4718 		ret = mark_btf_ld_reg(env, cur_regs(env), value_regno, PTR_TO_BTF_ID,
4719 				      kptr_field->kptr.btf, kptr_field->kptr.btf_id,
4720 				      btf_ld_kptr_type(env, kptr_field));
4721 		if (ret < 0)
4722 			return ret;
4723 	} else if (class == BPF_STX) {
4724 		val_reg = reg_state(env, value_regno);
4725 		if (bpf_register_is_null(val_reg)) {
4726 			/*
4727 			 * This store is valid only because the scalar is known to be
4728 			 * zero. Mark it precise so another scalar cannot be pruned
4729 			 * against this state.
4730 			 */
4731 			return mark_chain_precision(env, value_regno);
4732 		}
4733 		if (map_kptr_match_type(env, kptr_field, val_reg, value_regno))
4734 			return -EACCES;
4735 	} else if (class == BPF_ST) {
4736 		if (insn->imm) {
4737 			verbose(env, "BPF_ST imm must be 0 when storing to kptr at off=%u\n",
4738 				kptr_field->offset);
4739 			return -EACCES;
4740 		}
4741 	} else {
4742 		verbose(env, "kptr in map can only be accessed using BPF_LDX/BPF_STX/BPF_ST\n");
4743 		return -EACCES;
4744 	}
4745 	return 0;
4746 }
4747 
4748 /*
4749  * Return the size of the memory region accessible from a pointer to map value.
4750  * For INSN_ARRAY maps whole bpf_insn_array->ips array is accessible.
4751  */
4752 static u32 map_mem_size(const struct bpf_map *map)
4753 {
4754 	if (map->map_type == BPF_MAP_TYPE_INSN_ARRAY)
4755 		return map->max_entries * sizeof(long);
4756 
4757 	return map->value_size;
4758 }
4759 
4760 /* check read/write into a map element with possible variable offset */
4761 static int check_map_access(struct bpf_verifier_env *env, struct bpf_reg_state *reg, argno_t argno,
4762 			    int off, int size, bool zero_size_allowed,
4763 			    enum bpf_access_src src)
4764 {
4765 	struct bpf_map *map = reg->map_ptr;
4766 	u32 mem_size = map_mem_size(map);
4767 	struct btf_record *rec;
4768 	int err, i;
4769 
4770 	err = check_mem_region_access(env, reg, argno, off, size, mem_size, zero_size_allowed);
4771 	if (err)
4772 		return err;
4773 
4774 	if (IS_ERR_OR_NULL(map->record))
4775 		return 0;
4776 	rec = map->record;
4777 	for (i = 0; i < rec->cnt; i++) {
4778 		struct btf_field *field = &rec->fields[i];
4779 		u32 p = field->offset;
4780 
4781 		/* If any part of a field  can be touched by load/store, reject
4782 		 * this program. To check that [x1, x2) overlaps with [y1, y2),
4783 		 * it is sufficient to check x1 < y2 && y1 < x2.
4784 		 */
4785 		if (reg_smin(reg) + off < p + field->size &&
4786 		    p < reg_umax(reg) + off + size) {
4787 			switch (field->type) {
4788 			case BPF_KPTR_UNREF:
4789 			case BPF_KPTR_REF:
4790 			case BPF_KPTR_PERCPU:
4791 			case BPF_UPTR:
4792 				if (src != ACCESS_DIRECT) {
4793 					verbose(env, "%s cannot be accessed indirectly by helper\n",
4794 						btf_field_type_name(field->type));
4795 					return -EACCES;
4796 				}
4797 				if (!tnum_is_const(reg->var_off)) {
4798 					verbose(env, "%s access cannot have variable offset\n",
4799 						btf_field_type_name(field->type));
4800 					return -EACCES;
4801 				}
4802 				if (p != off + reg->var_off.value) {
4803 					verbose(env, "%s access misaligned expected=%u off=%llu\n",
4804 						btf_field_type_name(field->type),
4805 						p, off + reg->var_off.value);
4806 					return -EACCES;
4807 				}
4808 				if (size != bpf_size_to_bytes(BPF_DW)) {
4809 					verbose(env, "%s access size must be BPF_DW\n",
4810 						btf_field_type_name(field->type));
4811 					return -EACCES;
4812 				}
4813 				break;
4814 			default:
4815 				verbose(env, "%s cannot be accessed directly by load/store\n",
4816 					btf_field_type_name(field->type));
4817 				return -EACCES;
4818 			}
4819 		}
4820 	}
4821 	return 0;
4822 }
4823 
4824 static bool may_access_direct_pkt_data(struct bpf_verifier_env *env,
4825 			       const struct bpf_func_proto *fn,
4826 			       enum bpf_access_type t)
4827 {
4828 	enum bpf_prog_type prog_type = resolve_prog_type(env->prog);
4829 
4830 	switch (prog_type) {
4831 	/* Program types only with direct read access go here! */
4832 	case BPF_PROG_TYPE_LWT_IN:
4833 	case BPF_PROG_TYPE_LWT_OUT:
4834 	case BPF_PROG_TYPE_LWT_SEG6LOCAL:
4835 	case BPF_PROG_TYPE_SK_REUSEPORT:
4836 	case BPF_PROG_TYPE_FLOW_DISSECTOR:
4837 	case BPF_PROG_TYPE_CGROUP_SKB:
4838 		if (t == BPF_WRITE)
4839 			return false;
4840 		fallthrough;
4841 
4842 	/* Program types with direct read + write access go here! */
4843 	case BPF_PROG_TYPE_SCHED_CLS:
4844 	case BPF_PROG_TYPE_SCHED_ACT:
4845 	case BPF_PROG_TYPE_XDP:
4846 	case BPF_PROG_TYPE_LWT_XMIT:
4847 	case BPF_PROG_TYPE_SK_SKB:
4848 	case BPF_PROG_TYPE_SK_MSG:
4849 		if (fn)
4850 			return fn->pkt_access;
4851 
4852 		env->seen_direct_write = true;
4853 		return true;
4854 
4855 	case BPF_PROG_TYPE_CGROUP_SOCKOPT:
4856 		if (t == BPF_WRITE)
4857 			env->seen_direct_write = true;
4858 
4859 		return true;
4860 
4861 	default:
4862 		return false;
4863 	}
4864 }
4865 
4866 static int check_packet_access(struct bpf_verifier_env *env, struct bpf_reg_state *reg, argno_t argno, int off,
4867 			       int size, bool zero_size_allowed)
4868 {
4869 	int err;
4870 
4871 	if (reg->range < 0) {
4872 		verbose(env, "%s offset is outside of the packet\n", reg_arg_name(env, argno));
4873 		return -EINVAL;
4874 	}
4875 
4876 	err = check_mem_region_access(env, reg, argno, off, size, reg->range, zero_size_allowed);
4877 	if (err)
4878 		return err;
4879 
4880 	/* __check_mem_access has made sure "off + size - 1" is within u16.
4881 	 * reg_umax(reg) can't be bigger than MAX_PACKET_OFF which is 0xffff,
4882 	 * otherwise find_good_pkt_pointers would have refused to set range info
4883 	 * that __check_mem_access would have rejected this pkt access.
4884 	 * Therefore, "off + reg_umax(reg) + size - 1" won't overflow u32.
4885 	 */
4886 	env->prog->aux->max_pkt_offset =
4887 		max_t(u32, env->prog->aux->max_pkt_offset,
4888 		      off + reg_umax(reg) + size - 1);
4889 
4890 	return 0;
4891 }
4892 
4893 static bool is_var_ctx_off_allowed(struct bpf_prog *prog)
4894 {
4895 	return resolve_prog_type(prog) == BPF_PROG_TYPE_SYSCALL;
4896 }
4897 
4898 /* check access to 'struct bpf_context' fields.  Supports fixed offsets only */
4899 static int __check_ctx_access(struct bpf_verifier_env *env, int insn_idx, int off, int size,
4900 			      enum bpf_access_type t, struct bpf_insn_access_aux *info)
4901 {
4902 	if (env->ops->is_valid_access &&
4903 	    env->ops->is_valid_access(off, size, t, env->prog, info)) {
4904 		/* A non zero info.ctx_field_size indicates that this field is a
4905 		 * candidate for later verifier transformation to load the whole
4906 		 * field and then apply a mask when accessed with a narrower
4907 		 * access than actual ctx access size. A zero info.ctx_field_size
4908 		 * will only allow for whole field access and rejects any other
4909 		 * type of narrower access.
4910 		 */
4911 		if (base_type(info->reg_type) == PTR_TO_BTF_ID) {
4912 			if (info->ref_id &&
4913 			    !find_reference_state(env->cur_state, info->ref_id)) {
4914 				verbose(env, "invalid bpf_context access off=%d. Reference may already be released\n",
4915 					off);
4916 				return -EACCES;
4917 			}
4918 		} else {
4919 			env->insn_aux_data[insn_idx].ctx_field_size = info->ctx_field_size;
4920 		}
4921 		/* remember the offset of last byte accessed in ctx */
4922 		if (env->prog->aux->max_ctx_offset < off + size)
4923 			env->prog->aux->max_ctx_offset = off + size;
4924 		return 0;
4925 	}
4926 
4927 	verbose(env, "invalid bpf_context access off=%d size=%d\n", off, size);
4928 	return -EACCES;
4929 }
4930 
4931 static int check_ctx_access(struct bpf_verifier_env *env, int insn_idx, struct bpf_reg_state *reg, argno_t argno,
4932 			    int off, int access_size, enum bpf_access_type t,
4933 			    struct bpf_insn_access_aux *info)
4934 {
4935 	/*
4936 	 * Program types that don't rewrite ctx accesses can safely
4937 	 * dereference ctx pointers with fixed offsets.
4938 	 */
4939 	bool var_off_ok = is_var_ctx_off_allowed(env->prog);
4940 	bool fixed_off_ok = !env->ops->convert_ctx_access;
4941 	int err;
4942 
4943 	if (var_off_ok)
4944 		err = check_mem_region_access(env, reg, argno, off, access_size, U16_MAX, false);
4945 	else
4946 		err = __check_ptr_off_reg(env, reg, argno, fixed_off_ok);
4947 	if (err)
4948 		return err;
4949 	off += reg_umax(reg);
4950 
4951 	err = __check_ctx_access(env, insn_idx, off, access_size, t, info);
4952 	if (err)
4953 		verbose_linfo(env, insn_idx, "; ");
4954 	return err;
4955 }
4956 
4957 static int check_flow_keys_access(struct bpf_verifier_env *env,
4958 				  struct bpf_reg_state *reg, argno_t argno,
4959 				  int off, int size)
4960 {
4961 	/* Only a constant offset is allowed here; fold it into off. */
4962 	if (!tnum_is_const(reg->var_off)) {
4963 		char tn_buf[48];
4964 
4965 		tnum_strn(tn_buf, sizeof(tn_buf), reg->var_off);
4966 		verbose(env, "%s invalid variable offset to flow keys: off=%d, var_off=%s\n",
4967 			reg_arg_name(env, argno), off, tn_buf);
4968 		return -EACCES;
4969 	}
4970 	off += reg->var_off.value;
4971 
4972 	if (size < 0 || off < 0 ||
4973 	    (u64)off + size > sizeof(struct bpf_flow_keys)) {
4974 		verbose(env, "invalid access to flow keys off=%d size=%d\n",
4975 			off, size);
4976 		return -EACCES;
4977 	}
4978 	return 0;
4979 }
4980 
4981 static int check_sock_access(struct bpf_verifier_env *env, int insn_idx,
4982 			     struct bpf_reg_state *reg, argno_t argno, int off, int size,
4983 			     enum bpf_access_type t)
4984 {
4985 	struct bpf_insn_access_aux info = {};
4986 	bool valid;
4987 
4988 	if (reg_smin(reg) < 0) {
4989 		verbose(env, "%s min value is negative, either use unsigned index or do a if (index >=0) check.\n",
4990 			reg_arg_name(env, argno));
4991 		return -EACCES;
4992 	}
4993 
4994 	switch (reg->type) {
4995 	case PTR_TO_SOCK_COMMON:
4996 		valid = bpf_sock_common_is_valid_access(off, size, t, &info);
4997 		break;
4998 	case PTR_TO_SOCKET:
4999 		valid = bpf_sock_is_valid_access(off, size, t, &info);
5000 		break;
5001 	case PTR_TO_TCP_SOCK:
5002 		valid = bpf_tcp_sock_is_valid_access(off, size, t, &info);
5003 		break;
5004 	case PTR_TO_XDP_SOCK:
5005 		valid = bpf_xdp_sock_is_valid_access(off, size, t, &info);
5006 		break;
5007 	default:
5008 		valid = false;
5009 	}
5010 
5011 	if (valid) {
5012 		env->insn_aux_data[insn_idx].ctx_field_size =
5013 			info.ctx_field_size;
5014 		return 0;
5015 	}
5016 
5017 	verbose(env, "%s invalid %s access off=%d size=%d\n",
5018 		reg_arg_name(env, argno), reg_type_str(env, reg->type), off, size);
5019 
5020 	return -EACCES;
5021 }
5022 
5023 static bool is_pointer_value(struct bpf_verifier_env *env, int regno)
5024 {
5025 	return __is_pointer_value(env->allow_ptr_leaks, reg_state(env, regno));
5026 }
5027 
5028 static bool is_ctx_reg(struct bpf_verifier_env *env, int regno)
5029 {
5030 	const struct bpf_reg_state *reg = reg_state(env, regno);
5031 
5032 	return reg->type == PTR_TO_CTX;
5033 }
5034 
5035 static bool is_sk_reg(struct bpf_verifier_env *env, int regno)
5036 {
5037 	const struct bpf_reg_state *reg = reg_state(env, regno);
5038 
5039 	return type_is_sk_pointer(reg->type);
5040 }
5041 
5042 static bool is_pkt_reg(struct bpf_verifier_env *env, int regno)
5043 {
5044 	const struct bpf_reg_state *reg = reg_state(env, regno);
5045 
5046 	return type_is_pkt_pointer(reg->type);
5047 }
5048 
5049 static bool is_flow_key_reg(struct bpf_verifier_env *env, int regno)
5050 {
5051 	const struct bpf_reg_state *reg = reg_state(env, regno);
5052 
5053 	/* Separate to is_ctx_reg() since we still want to allow BPF_ST here. */
5054 	return reg->type == PTR_TO_FLOW_KEYS;
5055 }
5056 
5057 static bool is_arena_reg(struct bpf_verifier_env *env, int regno)
5058 {
5059 	const struct bpf_reg_state *reg = reg_state(env, regno);
5060 
5061 	return reg->type == PTR_TO_ARENA;
5062 }
5063 
5064 static bool is_load_acq_unsafe(struct bpf_verifier_env *env, int regno,
5065 			       struct bpf_insn *insn)
5066 {
5067 	const struct bpf_reg_state *reg = reg_state(env, regno);
5068 
5069 	/*
5070 	 * A BPF_LOAD_ACQ is not rewritten to a BPF_PROBE_MEM load by the
5071 	 * verifier, unlike a regular BPF_LDX. The JIT would emit a plain load
5072 	 * with no exception table entry, so a fault (e.g. NULL deref) crashes
5073 	 * the kernel instead of being handled. Reject the source pointer types
5074 	 * that would have needed that protection, the remaining ones stay
5075 	 * allowed.
5076 	 */
5077 	return insn->imm == BPF_LOAD_ACQ && bpf_may_fault_on_deref(reg->type);
5078 }
5079 
5080 /* Return false if @regno contains a pointer whose type isn't supported for
5081  * atomic instruction @insn.
5082  */
5083 static bool atomic_ptr_type_ok(struct bpf_verifier_env *env, int regno,
5084 			       struct bpf_insn *insn)
5085 {
5086 	if (is_ctx_reg(env, regno))
5087 		return false;
5088 	if (is_pkt_reg(env, regno))
5089 		return false;
5090 	if (is_flow_key_reg(env, regno))
5091 		return false;
5092 	if (is_sk_reg(env, regno))
5093 		return false;
5094 	if (is_arena_reg(env, regno))
5095 		return bpf_jit_supports_insn(insn, true);
5096 	if (is_load_acq_unsafe(env, regno, insn))
5097 		return false;
5098 	return true;
5099 }
5100 
5101 static u32 *reg2btf_ids[__BPF_REG_TYPE_MAX] = {
5102 #ifdef CONFIG_NET
5103 	[PTR_TO_SOCKET] = &btf_sock_ids[BTF_SOCK_TYPE_SOCK],
5104 	[PTR_TO_SOCK_COMMON] = &btf_sock_ids[BTF_SOCK_TYPE_SOCK_COMMON],
5105 	[PTR_TO_TCP_SOCK] = &btf_sock_ids[BTF_SOCK_TYPE_TCP],
5106 #endif
5107 	[CONST_PTR_TO_MAP] = btf_bpf_map_id,
5108 };
5109 
5110 static enum bpf_reg_type lookup_reg2btf_ids(u32 ref_id)
5111 {
5112 	enum bpf_reg_type type;
5113 
5114 	for (type = 0; type < __BPF_REG_TYPE_MAX; type++) {
5115 		if (reg2btf_ids[type] && *reg2btf_ids[type] == ref_id)
5116 			return type;
5117 	}
5118 
5119 	return NOT_INIT;
5120 }
5121 
5122 static bool is_trusted_reg(struct bpf_verifier_env *env, const struct bpf_reg_state *reg)
5123 {
5124 	/* A referenced register is always trusted. */
5125 	if (reg_is_referenced(env, reg))
5126 		return true;
5127 
5128 	/* Types listed in the reg2btf_ids are always trusted */
5129 	if (reg2btf_ids[base_type(reg->type)] &&
5130 	    !bpf_type_has_unsafe_modifiers(reg->type))
5131 		return true;
5132 
5133 	/* If a register is not referenced, it is trusted if it has the
5134 	 * MEM_ALLOC or PTR_TRUSTED type modifiers, and no others. Some of the
5135 	 * other type modifiers may be safe, but we elect to take an opt-in
5136 	 * approach here as some (e.g. PTR_UNTRUSTED and PTR_MAYBE_NULL) are
5137 	 * not.
5138 	 *
5139 	 * Eventually, we should make PTR_TRUSTED the single source of truth
5140 	 * for whether a register is trusted.
5141 	 */
5142 	return type_flag(reg->type) & BPF_REG_TRUSTED_MODIFIERS &&
5143 	       !bpf_type_has_unsafe_modifiers(reg->type);
5144 }
5145 
5146 static bool is_rcu_reg(const struct bpf_reg_state *reg)
5147 {
5148 	return reg->type & MEM_RCU;
5149 }
5150 
5151 static void clear_trusted_flags(enum bpf_type_flag *flag)
5152 {
5153 	*flag &= ~(BPF_REG_TRUSTED_MODIFIERS | MEM_RCU);
5154 }
5155 
5156 static int check_pkt_ptr_alignment(struct bpf_verifier_env *env,
5157 				   const struct bpf_reg_state *reg,
5158 				   int off, int size, bool strict)
5159 {
5160 	struct tnum reg_off;
5161 	int ip_align;
5162 
5163 	/* Byte size accesses are always allowed. */
5164 	if (!strict || size == 1)
5165 		return 0;
5166 
5167 	/* For platforms that do not have a Kconfig enabling
5168 	 * CONFIG_HAVE_EFFICIENT_UNALIGNED_ACCESS the value of
5169 	 * NET_IP_ALIGN is universally set to '2'.  And on platforms
5170 	 * that do set CONFIG_HAVE_EFFICIENT_UNALIGNED_ACCESS, we get
5171 	 * to this code only in strict mode where we want to emulate
5172 	 * the NET_IP_ALIGN==2 checking.  Therefore use an
5173 	 * unconditional IP align value of '2'.
5174 	 */
5175 	ip_align = 2;
5176 
5177 	reg_off = tnum_add(reg->var_off, tnum_const(ip_align + off));
5178 	if (!tnum_is_aligned(reg_off, size)) {
5179 		char tn_buf[48];
5180 
5181 		tnum_strn(tn_buf, sizeof(tn_buf), reg->var_off);
5182 		verbose(env,
5183 			"misaligned packet access off %d+%s+%d size %d\n",
5184 			ip_align, tn_buf, off, size);
5185 		return -EACCES;
5186 	}
5187 
5188 	return 0;
5189 }
5190 
5191 static int check_generic_ptr_alignment(struct bpf_verifier_env *env,
5192 				       const struct bpf_reg_state *reg,
5193 				       const char *pointer_desc,
5194 				       int off, int size, bool strict)
5195 {
5196 	struct tnum reg_off;
5197 
5198 	/* Byte size accesses are always allowed. */
5199 	if (!strict || size == 1)
5200 		return 0;
5201 
5202 	reg_off = tnum_add(reg->var_off, tnum_const(off));
5203 	if (!tnum_is_aligned(reg_off, size)) {
5204 		char tn_buf[48];
5205 
5206 		tnum_strn(tn_buf, sizeof(tn_buf), reg->var_off);
5207 		verbose(env, "misaligned %saccess off %s+%d size %d\n",
5208 			pointer_desc, tn_buf, off, size);
5209 		return -EACCES;
5210 	}
5211 
5212 	return 0;
5213 }
5214 
5215 static int check_ptr_alignment(struct bpf_verifier_env *env,
5216 			       const struct bpf_reg_state *reg, int off,
5217 			       int size, bool strict_alignment_once)
5218 {
5219 	bool strict = env->strict_alignment || strict_alignment_once;
5220 	const char *pointer_desc = "";
5221 
5222 	switch (reg->type) {
5223 	case PTR_TO_PACKET:
5224 	case PTR_TO_PACKET_META:
5225 		/* Special case, because of NET_IP_ALIGN. Given metadata sits
5226 		 * right in front, treat it the very same way.
5227 		 */
5228 		return check_pkt_ptr_alignment(env, reg, off, size, strict);
5229 	case PTR_TO_FLOW_KEYS:
5230 		pointer_desc = "flow keys ";
5231 		break;
5232 	case PTR_TO_MAP_KEY:
5233 		pointer_desc = "key ";
5234 		break;
5235 	case PTR_TO_MAP_VALUE:
5236 		pointer_desc = "value ";
5237 		if (reg->map_ptr->map_type == BPF_MAP_TYPE_INSN_ARRAY)
5238 			strict = true;
5239 		break;
5240 	case PTR_TO_CTX:
5241 		pointer_desc = "context ";
5242 		break;
5243 	case PTR_TO_STACK:
5244 		pointer_desc = "stack ";
5245 		/* The stack spill tracking logic in check_stack_write_fixed_off()
5246 		 * and check_stack_read_fixed_off() relies on stack accesses being
5247 		 * aligned.
5248 		 */
5249 		strict = true;
5250 		break;
5251 	case PTR_TO_SOCKET:
5252 		pointer_desc = "sock ";
5253 		break;
5254 	case PTR_TO_SOCK_COMMON:
5255 		pointer_desc = "sock_common ";
5256 		break;
5257 	case PTR_TO_TCP_SOCK:
5258 		pointer_desc = "tcp_sock ";
5259 		break;
5260 	case PTR_TO_XDP_SOCK:
5261 		pointer_desc = "xdp_sock ";
5262 		break;
5263 	case PTR_TO_ARENA:
5264 		return 0;
5265 	default:
5266 		break;
5267 	}
5268 	return check_generic_ptr_alignment(env, reg, pointer_desc, off, size,
5269 					   strict);
5270 }
5271 
5272 static enum priv_stack_mode bpf_enable_priv_stack(struct bpf_prog *prog)
5273 {
5274 	if (!bpf_jit_supports_private_stack())
5275 		return NO_PRIV_STACK;
5276 
5277 	/* bpf_prog_check_recur() checks all prog types that use bpf trampoline
5278 	 * while kprobe/tp/perf_event/raw_tp don't use trampoline hence checked
5279 	 * explicitly.
5280 	 */
5281 	switch (prog->type) {
5282 	case BPF_PROG_TYPE_KPROBE:
5283 	case BPF_PROG_TYPE_TRACEPOINT:
5284 	case BPF_PROG_TYPE_PERF_EVENT:
5285 	case BPF_PROG_TYPE_RAW_TRACEPOINT:
5286 		return PRIV_STACK_ADAPTIVE;
5287 	case BPF_PROG_TYPE_TRACING:
5288 	case BPF_PROG_TYPE_LSM:
5289 	case BPF_PROG_TYPE_STRUCT_OPS:
5290 		if (prog->aux->priv_stack_requested || bpf_prog_check_recur(prog))
5291 			return PRIV_STACK_ADAPTIVE;
5292 		fallthrough;
5293 	default:
5294 		break;
5295 	}
5296 
5297 	return NO_PRIV_STACK;
5298 }
5299 
5300 static int round_up_stack_depth(struct bpf_verifier_env *env, int stack_depth)
5301 {
5302 	if (env->prog->jit_requested)
5303 		return round_up(stack_depth, 16);
5304 
5305 	/* round up to 32-bytes, since this is granularity
5306 	 * of interpreter stack size
5307 	 */
5308 	return round_up(max_t(u32, stack_depth, 1), 32);
5309 }
5310 
5311 /* temporary state used for call frame depth calculation */
5312 struct bpf_subprog_call_depth_info {
5313 	int ret_insn; /* caller instruction where we return to. */
5314 	int caller; /* caller subprogram idx */
5315 	int frame; /* # of consecutive static call stack frames on top of stack */
5316 };
5317 
5318 /* starting from main bpf function walk all instructions of the function
5319  * and recursively walk all callees that given function can call.
5320  * Ignore jump and exit insns.
5321  */
5322 static int check_max_stack_depth_subprog(struct bpf_verifier_env *env, int idx,
5323 					 struct bpf_subprog_call_depth_info *dinfo,
5324 					 bool priv_stack_supported)
5325 {
5326 	struct bpf_subprog_info *subprog = env->subprog_info;
5327 	struct bpf_insn *insn = env->prog->insnsi;
5328 	int depth = 0, frame = 0, i, subprog_end, subprog_depth;
5329 	bool tail_call_reachable = false;
5330 	int total;
5331 	int tmp;
5332 
5333 	/* no caller idx */
5334 	dinfo[idx].caller = -1;
5335 
5336 	i = subprog[idx].start;
5337 	if (!priv_stack_supported)
5338 		subprog[idx].priv_stack_mode = NO_PRIV_STACK;
5339 process_func:
5340 	if (subprog[idx].has_ld_abs) {
5341 		for (tmp = idx; tmp >= 0; tmp = dinfo[tmp].caller) {
5342 			if (subprog[tmp].is_cb) {
5343 				verbose(env, "cannot use BPF_LD_[ABS|IND] within callback\n");
5344 				return -EINVAL;
5345 			}
5346 		}
5347 	}
5348 
5349 	/* protect against potential stack overflow that might happen when
5350 	 * bpf2bpf calls get combined with tailcalls. Limit the caller's stack
5351 	 * depth for such case down to 256 so that the worst case scenario
5352 	 * would result in 8k stack size (32 which is tailcall limit * 256 =
5353 	 * 8k).
5354 	 *
5355 	 * To get the idea what might happen, see an example:
5356 	 * func1 -> sub rsp, 128
5357 	 *  subfunc1 -> sub rsp, 256
5358 	 *  tailcall1 -> add rsp, 256
5359 	 *   func2 -> sub rsp, 192 (total stack size = 128 + 192 = 320)
5360 	 *   subfunc2 -> sub rsp, 64
5361 	 *   subfunc22 -> sub rsp, 128
5362 	 *   tailcall2 -> add rsp, 128
5363 	 *    func3 -> sub rsp, 32 (total stack size 128 + 192 + 64 + 32 = 416)
5364 	 *
5365 	 * tailcall will unwind the current stack frame but it will not get rid
5366 	 * of caller's stack as shown on the example above.
5367 	 */
5368 	if (idx && subprog[idx].has_tail_call && depth >= 256) {
5369 		verbose(env,
5370 			"tail_calls are not allowed when call stack of previous frames is %d bytes. Too large\n",
5371 			depth);
5372 		return -EACCES;
5373 	}
5374 
5375 	subprog_depth = round_up_stack_depth(env, subprog[idx].stack_depth);
5376 	if (IS_ENABLED(CONFIG_X86_64) && subprog[idx].stack_arg_cnt) {
5377 		/* x86-64 uses R9 for both private stack frame pointer and arg6. */
5378 		subprog[idx].priv_stack_mode = NO_PRIV_STACK;
5379 	} else if (priv_stack_supported) {
5380 		/* Request private stack support only if the subprog stack
5381 		 * depth is no less than BPF_PRIV_STACK_MIN_SIZE. This is to
5382 		 * avoid jit penalty if the stack usage is small.
5383 		 */
5384 		if (subprog[idx].priv_stack_mode == PRIV_STACK_UNKNOWN &&
5385 		    subprog_depth >= BPF_PRIV_STACK_MIN_SIZE)
5386 			subprog[idx].priv_stack_mode = PRIV_STACK_ADAPTIVE;
5387 	}
5388 
5389 	if (subprog[idx].priv_stack_mode == PRIV_STACK_ADAPTIVE) {
5390 		if (subprog_depth > env->max_stack_depth)
5391 			env->max_stack_depth = subprog_depth;
5392 		if (subprog_depth > MAX_BPF_STACK) {
5393 			verbose(env, "stack size of subprog %d is %d. Too large\n",
5394 				idx, subprog_depth);
5395 			return -EACCES;
5396 		}
5397 	} else {
5398 		depth += subprog_depth;
5399 		if (depth > env->max_stack_depth)
5400 			env->max_stack_depth = depth;
5401 		if (depth > MAX_BPF_STACK) {
5402 			total = 0;
5403 			for (tmp = idx; tmp >= 0; tmp = dinfo[tmp].caller)
5404 				total++;
5405 
5406 			verbose(env, "combined stack size of %d calls is %d. Too large\n",
5407 				total, depth);
5408 			return -EACCES;
5409 		}
5410 	}
5411 continue_func:
5412 	subprog_end = subprog[idx + 1].start;
5413 	for (; i < subprog_end; i++) {
5414 		int next_insn, sidx;
5415 
5416 		if (bpf_pseudo_kfunc_call(insn + i) && !insn[i].off) {
5417 			bool err = false;
5418 
5419 			if (!bpf_is_throw_kfunc(insn + i))
5420 				continue;
5421 			for (tmp = idx; tmp >= 0 && !err; tmp = dinfo[tmp].caller) {
5422 				if (subprog[tmp].is_cb) {
5423 					err = true;
5424 					break;
5425 				}
5426 			}
5427 			if (!err)
5428 				continue;
5429 			verbose(env,
5430 				"bpf_throw kfunc (insn %d) cannot be called from callback subprog %d\n",
5431 				i, idx);
5432 			return -EINVAL;
5433 		}
5434 
5435 		if (!bpf_pseudo_call(insn + i) && !bpf_pseudo_func(insn + i))
5436 			continue;
5437 		/* remember insn and function to return to */
5438 
5439 		/* find the callee */
5440 		next_insn = i + insn[i].imm + 1;
5441 		sidx = bpf_find_subprog(env, next_insn);
5442 		if (verifier_bug_if(sidx < 0, env, "callee not found at insn %d", next_insn))
5443 			return -EFAULT;
5444 		if (subprog[sidx].is_async_cb) {
5445 			/* async callbacks don't increase bpf prog stack size unless called directly */
5446 			if (!bpf_pseudo_call(insn + i))
5447 				continue;
5448 			if (subprog[sidx].is_exception_cb) {
5449 				verbose(env, "insn %d cannot call exception cb directly", i);
5450 				return -EINVAL;
5451 			}
5452 		}
5453 
5454 		/* store caller info for after we return from callee */
5455 		dinfo[idx].frame = frame;
5456 		dinfo[idx].ret_insn = i + 1;
5457 
5458 		/* push caller idx into callee's dinfo */
5459 		dinfo[sidx].caller = idx;
5460 
5461 		i = next_insn;
5462 
5463 		idx = sidx;
5464 		if (!priv_stack_supported)
5465 			subprog[idx].priv_stack_mode = NO_PRIV_STACK;
5466 
5467 		/* sync tail_call_reachable with callee state on entry */
5468 		tail_call_reachable = subprog[idx].has_tail_call;
5469 
5470 		frame = bpf_subprog_is_global(env, idx) ? 0 : frame + 1;
5471 		if (frame >= MAX_CALL_FRAMES) {
5472 			verbose(env, "the call stack of %d frames is too deep !\n",
5473 				frame);
5474 			return -E2BIG;
5475 		}
5476 		goto process_func;
5477 	}
5478 	/* if tail call got detected across bpf2bpf calls then mark each of the
5479 	 * currently present subprog frames as tail call reachable subprogs;
5480 	 * this info will be utilized by JIT so that we will be preserving the
5481 	 * tail call counter throughout bpf2bpf calls combined with tailcalls
5482 	 */
5483 	if (tail_call_reachable) {
5484 		for (tmp = idx; tmp >= 0; tmp = dinfo[tmp].caller) {
5485 			if (subprog[tmp].is_cb) {
5486 				verbose(env, "cannot tail call within callback\n");
5487 				return -EINVAL;
5488 			}
5489 			if (subprog[tmp].stack_arg_cnt) {
5490 				verbose(env, "tail_calls are not allowed in programs with stack args\n");
5491 				return -EINVAL;
5492 			}
5493 			subprog[tmp].tail_call_reachable = true;
5494 		}
5495 	} else if (!idx && subprog[0].has_tail_call && subprog[0].stack_arg_cnt) {
5496 		verbose(env, "tail_calls are not allowed in programs with stack args\n");
5497 		return -EINVAL;
5498 	}
5499 
5500 	if (subprog[0].tail_call_reachable)
5501 		env->prog->aux->tail_call_reachable = true;
5502 
5503 	/* end of for() loop means the last insn of the 'subprog'
5504 	 * was reached. Doesn't matter whether it was JA or EXIT
5505 	 */
5506 	if (frame == 0 && dinfo[idx].caller < 0)
5507 		return 0;
5508 	if (subprog[idx].priv_stack_mode != PRIV_STACK_ADAPTIVE)
5509 		depth -= round_up_stack_depth(env, subprog[idx].stack_depth);
5510 
5511 	/* pop caller idx from callee */
5512 	idx = dinfo[idx].caller;
5513 
5514 	/* retrieve caller state from its frame */
5515 	frame = dinfo[idx].frame;
5516 	i = dinfo[idx].ret_insn;
5517 
5518 	/* reset tail_call_reachable to the parent's actual state */
5519 	tail_call_reachable = subprog[idx].tail_call_reachable;
5520 
5521 	goto continue_func;
5522 }
5523 
5524 static int check_max_stack_depth(struct bpf_verifier_env *env)
5525 {
5526 	enum priv_stack_mode priv_stack_mode = PRIV_STACK_UNKNOWN;
5527 	struct bpf_subprog_call_depth_info *dinfo;
5528 	struct bpf_subprog_info *si = env->subprog_info;
5529 	bool priv_stack_supported;
5530 	int ret;
5531 
5532 	dinfo = kvzalloc_objs(*dinfo, env->subprog_cnt, GFP_KERNEL_ACCOUNT);
5533 	if (!dinfo)
5534 		return -ENOMEM;
5535 
5536 	for (int i = 0; i < env->subprog_cnt; i++) {
5537 		if (si[i].has_tail_call) {
5538 			priv_stack_mode = NO_PRIV_STACK;
5539 			break;
5540 		}
5541 	}
5542 
5543 	if (priv_stack_mode == PRIV_STACK_UNKNOWN)
5544 		priv_stack_mode = bpf_enable_priv_stack(env->prog);
5545 
5546 	/* All async_cb subprogs use normal kernel stack. If a particular
5547 	 * subprog appears in both main prog and async_cb subtree, that
5548 	 * subprog will use normal kernel stack to avoid potential nesting.
5549 	 * The reverse subprog traversal ensures when main prog subtree is
5550 	 * checked, the subprogs appearing in async_cb subtrees are already
5551 	 * marked as using normal kernel stack, so stack size checking can
5552 	 * be done properly.
5553 	 */
5554 	for (int i = env->subprog_cnt - 1; i >= 0; i--) {
5555 		if (!i || si[i].is_async_cb) {
5556 			priv_stack_supported = !i && priv_stack_mode == PRIV_STACK_ADAPTIVE;
5557 			ret = check_max_stack_depth_subprog(env, i, dinfo,
5558 					priv_stack_supported);
5559 			if (ret < 0) {
5560 				kvfree(dinfo);
5561 				return ret;
5562 			}
5563 		}
5564 	}
5565 
5566 	for (int i = 0; i < env->subprog_cnt; i++) {
5567 		if (si[i].priv_stack_mode == PRIV_STACK_ADAPTIVE) {
5568 			env->prog->aux->jits_use_priv_stack = true;
5569 			break;
5570 		}
5571 	}
5572 
5573 	kvfree(dinfo);
5574 
5575 	return 0;
5576 }
5577 
5578 static int __check_buffer_access(struct bpf_verifier_env *env,
5579 				 const char *buf_info,
5580 				 const struct bpf_reg_state *reg,
5581 				 argno_t argno, int off, int size,
5582 				 u32 *access_end)
5583 {
5584 	s64 start;
5585 
5586 	if (!tnum_is_const(reg->var_off)) {
5587 		char tn_buf[48];
5588 
5589 		tnum_strn(tn_buf, sizeof(tn_buf), reg->var_off);
5590 		verbose(env,
5591 			"%s invalid variable buffer offset: off=%d, var_off=%s\n",
5592 			reg_arg_name(env, argno), off, tn_buf);
5593 		return -EACCES;
5594 	}
5595 
5596 	start = (s64)reg->var_off.value + off;
5597 	if (start < 0) {
5598 		verbose(env,
5599 			"%s invalid negative %s buffer offset: off=%d, var_off=%lld\n",
5600 			reg_arg_name(env, argno), buf_info, off, (s64)reg->var_off.value);
5601 		return -EACCES;
5602 	}
5603 
5604 	*access_end = start + size;
5605 	return 0;
5606 }
5607 
5608 static int check_tp_buffer_access(struct bpf_verifier_env *env,
5609 				  const struct bpf_reg_state *reg,
5610 				  argno_t argno, int off, int size)
5611 {
5612 	u32 access_end;
5613 	int err;
5614 
5615 	err = __check_buffer_access(env, "tracepoint", reg, argno, off, size, &access_end);
5616 	if (err)
5617 		return err;
5618 
5619 	env->prog->aux->max_tp_access = max(access_end, env->prog->aux->max_tp_access);
5620 
5621 	return 0;
5622 }
5623 
5624 static int check_buffer_access(struct bpf_verifier_env *env,
5625 			       const struct bpf_reg_state *reg,
5626 			       argno_t argno, int off, int size,
5627 			       bool zero_size_allowed,
5628 			       u32 *max_access)
5629 {
5630 	const char *buf_info = type_is_rdonly_mem(reg->type) ? "rdonly" : "rdwr";
5631 	u32 access_end;
5632 	int err;
5633 
5634 	err = __check_buffer_access(env, buf_info, reg, argno, off, size, &access_end);
5635 	if (err)
5636 		return err;
5637 
5638 	*max_access = max(access_end, *max_access);
5639 
5640 	return 0;
5641 }
5642 
5643 /* BPF architecture zero extends alu32 ops into 64-bit registesr */
5644 static void zext_32_to_64(struct bpf_reg_state *reg)
5645 {
5646 	reg->var_off = tnum_subreg(reg->var_off);
5647 	reg_set_urange64(reg, reg_u32_min(reg), reg_u32_max(reg));
5648 }
5649 
5650 /* truncate register to smaller size (in bytes)
5651  * must be called with size < BPF_REG_SIZE
5652  */
5653 static void coerce_reg_to_size(struct bpf_reg_state *reg, int size)
5654 {
5655 	u64 mask;
5656 
5657 	/* clear high bits in bit representation */
5658 	reg->var_off = tnum_cast(reg->var_off, size);
5659 
5660 	/* fix arithmetic bounds */
5661 	mask = ((u64)1 << (size * 8)) - 1;
5662 	if ((reg_umin(reg) & ~mask) == (reg_umax(reg) & ~mask))
5663 		reg_set_urange64(reg, reg_umin(reg) & mask, reg_umax(reg) & mask);
5664 	else
5665 		reg_set_urange64(reg, 0, mask);
5666 
5667 	/* If size is smaller than 32bit register the 32bit register
5668 	 * values are also truncated so we push 64-bit bounds into
5669 	 * 32-bit bounds. Above were truncated < 32-bits already.
5670 	 */
5671 	if (size < 4)
5672 		__mark_reg32_unbounded(reg);
5673 
5674 	reg_bounds_sync(reg);
5675 }
5676 
5677 static void set_sext64_default_val(struct bpf_reg_state *reg, int size)
5678 {
5679 	if (size == 1) {
5680 		reg_set_srange64(reg, S8_MIN, S8_MAX);
5681 		reg_set_srange32(reg, S8_MIN, S8_MAX);
5682 	} else if (size == 2) {
5683 		reg_set_srange64(reg, S16_MIN, S16_MAX);
5684 		reg_set_srange32(reg, S16_MIN, S16_MAX);
5685 	} else {
5686 		/* size == 4 */
5687 		reg_set_srange64(reg, S32_MIN, S32_MAX);
5688 		reg_set_srange32(reg, S32_MIN, S32_MAX);
5689 	}
5690 	reg->var_off = tnum_unknown;
5691 }
5692 
5693 static void coerce_reg_to_size_sx(struct bpf_reg_state *reg, int size)
5694 {
5695 	s64 init_s64_max, init_s64_min, s64_max, s64_min, u64_cval;
5696 	u64 top_smax_value, top_smin_value;
5697 	u64 num_bits = size * 8;
5698 
5699 	if (tnum_is_const(reg->var_off)) {
5700 		u64_cval = reg->var_off.value;
5701 		if (size == 1)
5702 			reg->var_off = tnum_const((s8)u64_cval);
5703 		else if (size == 2)
5704 			reg->var_off = tnum_const((s16)u64_cval);
5705 		else
5706 			/* size == 4 */
5707 			reg->var_off = tnum_const((s32)u64_cval);
5708 
5709 		u64_cval = reg->var_off.value;
5710 		reg->r64 = cnum64_from_urange(u64_cval, u64_cval);
5711 		reg->r32 = cnum32_from_urange((u32)u64_cval, (u32)u64_cval);
5712 		return;
5713 	}
5714 
5715 	top_smax_value = ((u64)reg_smax(reg) >> num_bits) << num_bits;
5716 	top_smin_value = ((u64)reg_smin(reg) >> num_bits) << num_bits;
5717 
5718 	if (top_smax_value != top_smin_value)
5719 		goto out;
5720 
5721 	/* find the s64_min and s64_min after sign extension */
5722 	if (size == 1) {
5723 		init_s64_max = (s8)reg_smax(reg);
5724 		init_s64_min = (s8)reg_smin(reg);
5725 	} else if (size == 2) {
5726 		init_s64_max = (s16)reg_smax(reg);
5727 		init_s64_min = (s16)reg_smin(reg);
5728 	} else {
5729 		init_s64_max = (s32)reg_smax(reg);
5730 		init_s64_min = (s32)reg_smin(reg);
5731 	}
5732 
5733 	s64_max = max(init_s64_max, init_s64_min);
5734 	s64_min = min(init_s64_max, init_s64_min);
5735 
5736 	/* both of s64_max/s64_min positive or negative */
5737 	if ((s64_max >= 0) == (s64_min >= 0)) {
5738 		reg_set_srange64(reg, s64_min, s64_max);
5739 		reg_set_srange32(reg, s64_min, s64_max);
5740 		reg->var_off = tnum_range(s64_min, s64_max);
5741 		return;
5742 	}
5743 
5744 out:
5745 	set_sext64_default_val(reg, size);
5746 }
5747 
5748 static void set_sext32_default_val(struct bpf_reg_state *reg, int size)
5749 {
5750 	if (size == 1)
5751 		reg_set_srange32(reg, S8_MIN, S8_MAX);
5752 	else
5753 		/* size == 2 */
5754 		reg_set_srange32(reg, S16_MIN, S16_MAX);
5755 	reg->var_off = tnum_subreg(tnum_unknown);
5756 }
5757 
5758 static void coerce_subreg_to_size_sx(struct bpf_reg_state *reg, int size)
5759 {
5760 	s32 init_s32_max, init_s32_min, s32_max, s32_min, u32_val;
5761 	u32 top_smax_value, top_smin_value;
5762 	u32 num_bits = size * 8;
5763 
5764 	if (tnum_is_const(reg->var_off)) {
5765 		u32_val = reg->var_off.value;
5766 		if (size == 1)
5767 			reg->var_off = tnum_const((s8)u32_val);
5768 		else
5769 			reg->var_off = tnum_const((s16)u32_val);
5770 
5771 		u32_val = reg->var_off.value;
5772 		reg_set_srange32(reg, u32_val, u32_val);
5773 		return;
5774 	}
5775 
5776 	top_smax_value = ((u32)reg_s32_max(reg) >> num_bits) << num_bits;
5777 	top_smin_value = ((u32)reg_s32_min(reg) >> num_bits) << num_bits;
5778 
5779 	if (top_smax_value != top_smin_value)
5780 		goto out;
5781 
5782 	/* find the s32_min and s32_min after sign extension */
5783 	if (size == 1) {
5784 		init_s32_max = (s8)reg_s32_max(reg);
5785 		init_s32_min = (s8)reg_s32_min(reg);
5786 	} else {
5787 		/* size == 2 */
5788 		init_s32_max = (s16)reg_s32_max(reg);
5789 		init_s32_min = (s16)reg_s32_min(reg);
5790 	}
5791 	s32_max = max(init_s32_max, init_s32_min);
5792 	s32_min = min(init_s32_max, init_s32_min);
5793 
5794 	if ((s32_min >= 0) == (s32_max >= 0)) {
5795 		reg_set_srange32(reg, s32_min, s32_max);
5796 		reg->var_off = tnum_subreg(tnum_range(s32_min, s32_max));
5797 		return;
5798 	}
5799 
5800 out:
5801 	set_sext32_default_val(reg, size);
5802 }
5803 
5804 bool bpf_map_is_rdonly(const struct bpf_map *map)
5805 {
5806 	/* A map is considered read-only if the following condition are true:
5807 	 *
5808 	 * 1) BPF program side cannot change any of the map content. The
5809 	 *    BPF_F_RDONLY_PROG flag is throughout the lifetime of a map
5810 	 *    and was set at map creation time.
5811 	 * 2) The map value(s) have been initialized from user space by a
5812 	 *    loader and then "frozen", such that no new map update/delete
5813 	 *    operations from syscall side are possible for the rest of
5814 	 *    the map's lifetime from that point onwards.
5815 	 * 3) Any parallel/pending map update/delete operations from syscall
5816 	 *    side have been completed. Only after that point, it's safe to
5817 	 *    assume that map value(s) are immutable.
5818 	 */
5819 	return (map->map_flags & BPF_F_RDONLY_PROG) &&
5820 	       READ_ONCE(map->frozen) &&
5821 	       !bpf_map_write_active(map);
5822 }
5823 
5824 int bpf_map_direct_read(struct bpf_map *map, int off, int size, u64 *val,
5825 			bool is_ldsx)
5826 {
5827 	void *ptr;
5828 	u64 addr;
5829 	int err;
5830 
5831 	if (map->map_type == BPF_MAP_TYPE_INSN_ARRAY || map->map_type == BPF_MAP_TYPE_PERCPU_ARRAY)
5832 		return -EINVAL;
5833 	err = map->ops->map_direct_value_addr(map, &addr, off);
5834 	if (err)
5835 		return err;
5836 	ptr = (void *)(long)addr + off;
5837 
5838 	switch (size) {
5839 	case sizeof(u8):
5840 		*val = is_ldsx ? (s64)*(s8 *)ptr : (u64)*(u8 *)ptr;
5841 		break;
5842 	case sizeof(u16):
5843 		*val = is_ldsx ? (s64)*(s16 *)ptr : (u64)*(u16 *)ptr;
5844 		break;
5845 	case sizeof(u32):
5846 		*val = is_ldsx ? (s64)*(s32 *)ptr : (u64)*(u32 *)ptr;
5847 		break;
5848 	case sizeof(u64):
5849 		*val = *(u64 *)ptr;
5850 		break;
5851 	default:
5852 		return -EINVAL;
5853 	}
5854 	return 0;
5855 }
5856 
5857 #define BTF_TYPE_SAFE_RCU(__type)  __PASTE(__type, __safe_rcu)
5858 #define BTF_TYPE_SAFE_RCU_OR_NULL(__type)  __PASTE(__type, __safe_rcu_or_null)
5859 #define BTF_TYPE_SAFE_TRUSTED(__type)  __PASTE(__type, __safe_trusted)
5860 #define BTF_TYPE_SAFE_TRUSTED_OR_NULL(__type)  __PASTE(__type, __safe_trusted_or_null)
5861 
5862 /*
5863  * Allow list few fields as RCU trusted or full trusted.
5864  * This logic doesn't allow mix tagging and will be removed once GCC supports
5865  * btf_type_tag.
5866  */
5867 
5868 /* RCU trusted: these fields are trusted in RCU CS and never NULL */
5869 BTF_TYPE_SAFE_RCU(struct task_struct) {
5870 	const cpumask_t *cpus_ptr;
5871 	struct css_set __rcu *cgroups;
5872 	struct task_struct __rcu *real_parent;
5873 	struct task_struct *group_leader;
5874 };
5875 
5876 BTF_TYPE_SAFE_RCU(struct cgroup) {
5877 	/* cgrp->kn is always accessible as documented in kernel/cgroup/cgroup.c */
5878 	struct kernfs_node *kn;
5879 };
5880 
5881 BTF_TYPE_SAFE_RCU(struct css_set) {
5882 	struct cgroup *dfl_cgrp;
5883 };
5884 
5885 BTF_TYPE_SAFE_RCU(struct cgroup_subsys_state) {
5886 	struct cgroup *cgroup;
5887 };
5888 
5889 /* RCU trusted: these fields are trusted in RCU CS and can be NULL */
5890 BTF_TYPE_SAFE_RCU_OR_NULL(struct mm_struct) {
5891 	struct file __rcu *exe_file;
5892 #ifdef CONFIG_MEMCG
5893 	struct task_struct __rcu *owner;
5894 #endif
5895 };
5896 
5897 /* skb->sk, req->sk are not RCU protected, but we mark them as such
5898  * because bpf prog accessible sockets are SOCK_RCU_FREE.
5899  */
5900 BTF_TYPE_SAFE_RCU_OR_NULL(struct sk_buff) {
5901 	struct sock *sk;
5902 };
5903 
5904 BTF_TYPE_SAFE_RCU_OR_NULL(struct request_sock) {
5905 	struct sock *sk;
5906 };
5907 
5908 /* full trusted: these fields are trusted even outside of RCU CS and never NULL */
5909 BTF_TYPE_SAFE_TRUSTED(struct bpf_iter_meta) {
5910 	struct seq_file *seq;
5911 };
5912 
5913 BTF_TYPE_SAFE_TRUSTED(struct bpf_iter__task) {
5914 	struct bpf_iter_meta *meta;
5915 	struct task_struct *task;
5916 };
5917 
5918 BTF_TYPE_SAFE_TRUSTED(struct linux_binprm) {
5919 	struct file *file;
5920 };
5921 
5922 BTF_TYPE_SAFE_TRUSTED(struct file) {
5923 	struct inode *f_inode;
5924 };
5925 
5926 BTF_TYPE_SAFE_TRUSTED_OR_NULL(struct dentry) {
5927 	struct inode *d_inode;
5928 };
5929 
5930 BTF_TYPE_SAFE_TRUSTED_OR_NULL(struct socket) {
5931 	struct sock *sk;
5932 };
5933 
5934 BTF_TYPE_SAFE_TRUSTED_OR_NULL(struct vm_area_struct) {
5935 	struct mm_struct *vm_mm;
5936 	struct file *vm_file;
5937 };
5938 
5939 static bool type_is_rcu(struct bpf_verifier_env *env,
5940 			struct bpf_reg_state *reg,
5941 			const char *field_name, u32 btf_id)
5942 {
5943 	BTF_TYPE_EMIT(BTF_TYPE_SAFE_RCU(struct task_struct));
5944 	BTF_TYPE_EMIT(BTF_TYPE_SAFE_RCU(struct cgroup));
5945 	BTF_TYPE_EMIT(BTF_TYPE_SAFE_RCU(struct css_set));
5946 	BTF_TYPE_EMIT(BTF_TYPE_SAFE_RCU(struct cgroup_subsys_state));
5947 
5948 	return btf_nested_type_is_trusted(&env->log, reg, field_name, btf_id, "__safe_rcu");
5949 }
5950 
5951 static bool type_is_rcu_or_null(struct bpf_verifier_env *env,
5952 				struct bpf_reg_state *reg,
5953 				const char *field_name, u32 btf_id)
5954 {
5955 	BTF_TYPE_EMIT(BTF_TYPE_SAFE_RCU_OR_NULL(struct mm_struct));
5956 	BTF_TYPE_EMIT(BTF_TYPE_SAFE_RCU_OR_NULL(struct sk_buff));
5957 	BTF_TYPE_EMIT(BTF_TYPE_SAFE_RCU_OR_NULL(struct request_sock));
5958 
5959 	return btf_nested_type_is_trusted(&env->log, reg, field_name, btf_id, "__safe_rcu_or_null");
5960 }
5961 
5962 static bool type_is_trusted(struct bpf_verifier_env *env,
5963 			    struct bpf_reg_state *reg,
5964 			    const char *field_name, u32 btf_id)
5965 {
5966 	BTF_TYPE_EMIT(BTF_TYPE_SAFE_TRUSTED(struct bpf_iter_meta));
5967 	BTF_TYPE_EMIT(BTF_TYPE_SAFE_TRUSTED(struct bpf_iter__task));
5968 	BTF_TYPE_EMIT(BTF_TYPE_SAFE_TRUSTED(struct linux_binprm));
5969 	BTF_TYPE_EMIT(BTF_TYPE_SAFE_TRUSTED(struct file));
5970 
5971 	return btf_nested_type_is_trusted(&env->log, reg, field_name, btf_id, "__safe_trusted");
5972 }
5973 
5974 static bool type_is_trusted_or_null(struct bpf_verifier_env *env,
5975 				    struct bpf_reg_state *reg,
5976 				    const char *field_name, u32 btf_id)
5977 {
5978 	BTF_TYPE_EMIT(BTF_TYPE_SAFE_TRUSTED_OR_NULL(struct socket));
5979 	BTF_TYPE_EMIT(BTF_TYPE_SAFE_TRUSTED_OR_NULL(struct dentry));
5980 	BTF_TYPE_EMIT(BTF_TYPE_SAFE_TRUSTED_OR_NULL(struct vm_area_struct));
5981 
5982 	return btf_nested_type_is_trusted(&env->log, reg, field_name, btf_id,
5983 					  "__safe_trusted_or_null");
5984 }
5985 
5986 static int check_ptr_to_btf_access(struct bpf_verifier_env *env,
5987 				   struct bpf_reg_state *regs, struct bpf_reg_state *reg,
5988 				   argno_t argno, int off, int size,
5989 				   enum bpf_access_type atype,
5990 				   int value_regno)
5991 {
5992 	const struct btf_type *t = btf_type_by_id(reg->btf, reg->btf_id);
5993 	const char *tname = btf_name_by_offset(reg->btf, t->name_off);
5994 	const char *field_name = NULL;
5995 	enum bpf_type_flag flag = 0;
5996 	u32 btf_id = 0;
5997 	int ret;
5998 
5999 	if (!env->allow_ptr_leaks) {
6000 		verbose(env,
6001 			"'struct %s' access is allowed only to CAP_PERFMON and CAP_SYS_ADMIN\n",
6002 			tname);
6003 		return -EPERM;
6004 	}
6005 	if (!env->prog->gpl_compatible && btf_is_kernel(reg->btf)) {
6006 		verbose(env,
6007 			"Cannot access kernel 'struct %s' from non-GPL compatible program\n",
6008 			tname);
6009 		return -EINVAL;
6010 	}
6011 
6012 	if (!tnum_is_const(reg->var_off)) {
6013 		char tn_buf[48];
6014 
6015 		tnum_strn(tn_buf, sizeof(tn_buf), reg->var_off);
6016 		verbose(env,
6017 			"%s is ptr_%s invalid variable offset: off=%d, var_off=%s\n",
6018 			reg_arg_name(env, argno), tname, off, tn_buf);
6019 		return -EACCES;
6020 	}
6021 
6022 	off += reg->var_off.value;
6023 
6024 	if (off < 0) {
6025 		verbose(env,
6026 			"%s is ptr_%s invalid negative access: off=%d\n",
6027 			reg_arg_name(env, argno), tname, off);
6028 		return -EACCES;
6029 	}
6030 
6031 	if (reg->type & MEM_USER) {
6032 		verbose(env,
6033 			"%s is ptr_%s access user memory: off=%d\n",
6034 			reg_arg_name(env, argno), tname, off);
6035 		return -EACCES;
6036 	}
6037 
6038 	if (reg->type & MEM_PERCPU) {
6039 		verbose(env,
6040 			"%s is ptr_%s access percpu memory: off=%d\n",
6041 			reg_arg_name(env, argno), tname, off);
6042 		return -EACCES;
6043 	}
6044 
6045 	if (atype != BPF_READ && bpf_may_fault_on_deref(reg->type)) {
6046 		verbose(env, "only read is supported\n");
6047 		return -EACCES;
6048 	}
6049 
6050 	if (env->ops->btf_struct_access && !type_is_alloc(reg->type) && atype == BPF_WRITE) {
6051 		if (!btf_is_kernel(reg->btf)) {
6052 			verifier_bug(env, "reg->btf must be kernel btf");
6053 			return -EFAULT;
6054 		}
6055 		ret = env->ops->btf_struct_access(&env->log, reg, off, size);
6056 		if (ret < 0)
6057 			verbose(env,
6058 				"%s cannot write into ptr_%s at off=%d size=%d\n",
6059 				reg_arg_name(env, argno), tname, off, size);
6060 	} else {
6061 		/* Writes are permitted with default btf_struct_access for
6062 		 * program allocated objects (which always have id > 0).
6063 		 */
6064 		if (atype != BPF_READ && !type_is_ptr_alloc_obj(reg->type)) {
6065 			verbose(env, "only read is supported\n");
6066 			return -EACCES;
6067 		}
6068 
6069 		/*
6070 		 * A fault-prone allocated object may still be read through a
6071 		 * BPF_PROBE_MEM load after its lifetime protection ends. Writes
6072 		 * through such pointers were rejected above.
6073 		 */
6074 		if (type_is_alloc(reg->type) && !bpf_may_fault_on_deref(reg->type) &&
6075 		    !type_is_non_owning_ref(reg->type) &&
6076 		    !(reg->type & MEM_RCU) && !reg_is_referenced(env, reg)) {
6077 			verifier_bug(env, "allocated object must have a referenced id");
6078 			return -EFAULT;
6079 		}
6080 
6081 		ret = btf_struct_access(&env->log, reg, off, size, atype, &btf_id, &flag, &field_name);
6082 	}
6083 
6084 	if (ret < 0)
6085 		return ret;
6086 
6087 	if (ret != PTR_TO_BTF_ID) {
6088 		/* just mark; */
6089 
6090 	} else if (type_flag(reg->type) & PTR_UNTRUSTED) {
6091 		/* If this is an untrusted pointer, all pointers formed by walking it
6092 		 * also inherit the untrusted flag.
6093 		 */
6094 		flag = PTR_UNTRUSTED;
6095 
6096 	} else if (is_trusted_reg(env, reg) || is_rcu_reg(reg)) {
6097 		/* By default any pointer obtained from walking a trusted pointer is no
6098 		 * longer trusted, unless the field being accessed has explicitly been
6099 		 * marked as inheriting its parent's state of trust (either full or RCU).
6100 		 * For example:
6101 		 * 'cgroups' pointer is untrusted if task->cgroups dereference
6102 		 * happened in a sleepable program outside of bpf_rcu_read_lock()
6103 		 * section. In a non-sleepable program it's trusted while in RCU CS (aka MEM_RCU).
6104 		 * Note bpf_rcu_read_unlock() converts MEM_RCU pointers to PTR_UNTRUSTED.
6105 		 *
6106 		 * A regular RCU-protected pointer with __rcu tag can also be deemed
6107 		 * trusted if we are in an RCU CS. Such pointer can be NULL.
6108 		 */
6109 		if (type_is_trusted(env, reg, field_name, btf_id)) {
6110 			flag |= PTR_TRUSTED;
6111 		} else if (type_is_trusted_or_null(env, reg, field_name, btf_id)) {
6112 			flag |= PTR_TRUSTED | PTR_MAYBE_NULL;
6113 		} else if (in_rcu_cs(env) && !type_may_be_null(reg->type)) {
6114 			if (type_is_rcu(env, reg, field_name, btf_id)) {
6115 				/* ignore __rcu tag and mark it MEM_RCU */
6116 				flag |= MEM_RCU;
6117 			} else if (flag & MEM_RCU ||
6118 				   type_is_rcu_or_null(env, reg, field_name, btf_id)) {
6119 				/* __rcu tagged pointers can be NULL */
6120 				flag |= MEM_RCU | PTR_MAYBE_NULL;
6121 
6122 				/* We always trust them */
6123 				if (type_is_rcu_or_null(env, reg, field_name, btf_id) &&
6124 				    flag & PTR_UNTRUSTED)
6125 					flag &= ~PTR_UNTRUSTED;
6126 			} else if (flag & (MEM_PERCPU | MEM_USER)) {
6127 				/* keep as-is */
6128 			} else {
6129 				/* walking unknown pointers yields old deprecated PTR_TO_BTF_ID */
6130 				clear_trusted_flags(&flag);
6131 			}
6132 		} else {
6133 			/*
6134 			 * If not in RCU CS or MEM_RCU pointer can be NULL then
6135 			 * aggressively mark as untrusted otherwise such
6136 			 * pointers will be plain PTR_TO_BTF_ID without flags
6137 			 * and will be allowed to be passed into helpers for
6138 			 * compat reasons.
6139 			 */
6140 			flag = PTR_UNTRUSTED;
6141 		}
6142 	} else {
6143 		/* Old compat. Deprecated */
6144 		clear_trusted_flags(&flag);
6145 	}
6146 
6147 	if (atype == BPF_READ && value_regno >= 0) {
6148 		ret = mark_btf_ld_reg(env, regs, value_regno, ret, reg->btf, btf_id, flag);
6149 		if (ret < 0)
6150 			return ret;
6151 	}
6152 
6153 	return 0;
6154 }
6155 
6156 static int check_ptr_to_map_access(struct bpf_verifier_env *env,
6157 				   struct bpf_reg_state *regs, struct bpf_reg_state *reg,
6158 				   argno_t argno, int off, int size,
6159 				   enum bpf_access_type atype,
6160 				   int value_regno)
6161 {
6162 	struct bpf_map *map = reg->map_ptr;
6163 	struct bpf_reg_state map_reg;
6164 	enum bpf_type_flag flag = 0;
6165 	const struct btf_type *t;
6166 	const char *tname;
6167 	u32 btf_id;
6168 	int ret;
6169 
6170 	if (!btf_vmlinux) {
6171 		verbose(env, "map_ptr access not supported without CONFIG_DEBUG_INFO_BTF\n");
6172 		return -ENOTSUPP;
6173 	}
6174 
6175 	if (!map->ops->map_btf_id || !*map->ops->map_btf_id) {
6176 		verbose(env, "map_ptr access not supported for map type %d\n",
6177 			map->map_type);
6178 		return -ENOTSUPP;
6179 	}
6180 
6181 	t = btf_type_by_id(btf_vmlinux, *map->ops->map_btf_id);
6182 	tname = btf_name_by_offset(btf_vmlinux, t->name_off);
6183 
6184 	if (!env->allow_ptr_leaks) {
6185 		verbose(env,
6186 			"'struct %s' access is allowed only to CAP_PERFMON and CAP_SYS_ADMIN\n",
6187 			tname);
6188 		return -EPERM;
6189 	}
6190 
6191 	if (off < 0) {
6192 		verbose(env, "%s is %s invalid negative access: off=%d\n",
6193 			reg_arg_name(env, argno), tname, off);
6194 		return -EACCES;
6195 	}
6196 
6197 	if (atype != BPF_READ) {
6198 		verbose(env, "only read from %s is supported\n", tname);
6199 		return -EACCES;
6200 	}
6201 
6202 	/* Simulate access to a PTR_TO_BTF_ID */
6203 	memset(&map_reg, 0, sizeof(map_reg));
6204 	ret = mark_btf_ld_reg(env, &map_reg, 0, PTR_TO_BTF_ID,
6205 			      btf_vmlinux, *map->ops->map_btf_id, 0);
6206 	if (ret < 0)
6207 		return ret;
6208 	ret = btf_struct_access(&env->log, &map_reg, off, size, atype, &btf_id, &flag, NULL);
6209 	if (ret < 0)
6210 		return ret;
6211 
6212 	if (value_regno >= 0) {
6213 		ret = mark_btf_ld_reg(env, regs, value_regno, ret, btf_vmlinux, btf_id, flag);
6214 		if (ret < 0)
6215 			return ret;
6216 	}
6217 
6218 	return 0;
6219 }
6220 
6221 /* Check that the stack access at the given offset is within bounds. The
6222  * maximum valid offset is -1.
6223  *
6224  * The minimum valid offset is -MAX_BPF_STACK for writes, and
6225  * -state->allocated_stack for reads.
6226  */
6227 static int check_stack_slot_within_bounds(struct bpf_verifier_env *env,
6228                                           s64 off,
6229                                           struct bpf_func_state *state,
6230                                           enum bpf_access_type t)
6231 {
6232 	int min_valid_off;
6233 
6234 	if (t == BPF_WRITE || env->allow_uninit_stack)
6235 		min_valid_off = -MAX_BPF_STACK;
6236 	else
6237 		min_valid_off = -state->allocated_stack;
6238 
6239 	if (off < min_valid_off || off > -1)
6240 		return -EACCES;
6241 	return 0;
6242 }
6243 
6244 /* Check that the stack access at 'regno + off' falls within the maximum stack
6245  * bounds.
6246  *
6247  * 'off' includes `regno->offset`, but not its dynamic part (if any).
6248  */
6249 static int check_stack_access_within_bounds(
6250 		struct bpf_verifier_env *env, struct bpf_reg_state *reg,
6251 		argno_t argno, int off, int access_size,
6252 		enum bpf_access_type type)
6253 {
6254 	struct bpf_func_state *state = bpf_func(env, reg);
6255 	s64 min_off, max_off;
6256 	int err;
6257 	char *err_extra;
6258 
6259 	if (type == BPF_READ)
6260 		err_extra = " read from";
6261 	else
6262 		err_extra = " write to";
6263 
6264 	if (tnum_is_const(reg->var_off)) {
6265 		min_off = (s64)reg->var_off.value + off;
6266 		max_off = min_off + access_size;
6267 	} else {
6268 		if (reg_smax(reg) >= BPF_MAX_VAR_OFF ||
6269 		    reg_smin(reg) <= -BPF_MAX_VAR_OFF) {
6270 			verbose(env, "invalid unbounded variable-offset%s stack %s\n",
6271 				err_extra, reg_arg_name(env, argno));
6272 			return -EACCES;
6273 		}
6274 		min_off = reg_smin(reg) + off;
6275 		max_off = reg_smax(reg) + off + access_size;
6276 	}
6277 
6278 	err = check_stack_slot_within_bounds(env, min_off, state, type);
6279 	if (!err && max_off > 0)
6280 		err = -EINVAL; /* out of stack access into non-negative offsets */
6281 	if (!err && access_size < 0)
6282 		/* access_size should not be negative (or overflow an int); others checks
6283 		 * along the way should have prevented such an access.
6284 		 */
6285 		err = -EFAULT; /* invalid negative access size; integer overflow? */
6286 
6287 	if (err) {
6288 		if (tnum_is_const(reg->var_off)) {
6289 			verbose(env, "invalid%s stack %s off=%lld size=%d\n",
6290 				err_extra, reg_arg_name(env, argno), min_off, access_size);
6291 		} else {
6292 			char tn_buf[48];
6293 
6294 			tnum_strn(tn_buf, sizeof(tn_buf), reg->var_off);
6295 			verbose(env, "invalid variable-offset%s stack %s var_off=%s off=%d size=%d\n",
6296 				err_extra, reg_arg_name(env, argno), tn_buf, off, access_size);
6297 		}
6298 		return err;
6299 	}
6300 
6301 	/* Note that there is no stack access with offset zero, so the needed stack
6302 	 * size is -min_off, not -min_off+1.
6303 	 */
6304 	return grow_stack_state(env, state, -min_off /* size */);
6305 }
6306 
6307 static bool get_func_retval_range(struct bpf_prog *prog,
6308 				  struct bpf_retval_range *range)
6309 {
6310 	if (prog->type == BPF_PROG_TYPE_LSM &&
6311 		prog->expected_attach_type == BPF_LSM_MAC &&
6312 		!bpf_lsm_get_retval_range(prog, range)) {
6313 		return true;
6314 	}
6315 	return false;
6316 }
6317 
6318 static void add_scalar_to_reg(struct bpf_reg_state *dst_reg, s64 val)
6319 {
6320 	struct bpf_reg_state fake_reg;
6321 
6322 	if (!val)
6323 		return;
6324 
6325 	fake_reg.type = SCALAR_VALUE;
6326 	__mark_reg_known(&fake_reg, val);
6327 
6328 	scalar32_min_max_add(dst_reg, &fake_reg);
6329 	scalar_min_max_add(dst_reg, &fake_reg);
6330 	dst_reg->var_off = tnum_add(dst_reg->var_off, fake_reg.var_off);
6331 
6332 	reg_bounds_sync(dst_reg);
6333 }
6334 
6335 static int check_map_mem_read(struct bpf_verifier_env *env, struct bpf_reg_state *reg, int off,
6336 			      int bpf_size, int value_regno, bool is_ldsx)
6337 {
6338 	struct bpf_reg_state *regs = cur_regs(env);
6339 	int size = bpf_size_to_bytes(bpf_size);
6340 	struct bpf_map *map = reg->map_ptr;
6341 
6342 	switch (map->map_type) {
6343 	case BPF_MAP_TYPE_INSN_ARRAY:
6344 		if (bpf_size != BPF_DW) {
6345 			verbose(env, "Invalid read of %d bytes from insn_array\n", size);
6346 			return -EACCES;
6347 		}
6348 		regs[value_regno] = *reg;
6349 		add_scalar_to_reg(&regs[value_regno], off);
6350 		regs[value_regno].type = PTR_TO_INSN;
6351 		return 0;
6352 	case BPF_MAP_TYPE_PERCPU_ARRAY:
6353 		goto reg_unknown;
6354 	default:
6355 		break;
6356 	}
6357 
6358 	/* If map is read-only, track its contents as scalars. */
6359 	if (tnum_is_const(reg->var_off) &&
6360 	    bpf_map_is_rdonly(map) &&
6361 	    map->ops->map_direct_value_addr) {
6362 		int map_off = off + reg->var_off.value;
6363 		u64 val = 0;
6364 		int err;
6365 
6366 		err = bpf_map_direct_read(map, map_off, size, &val, is_ldsx);
6367 		if (err)
6368 			return err;
6369 
6370 		regs[value_regno].type = SCALAR_VALUE;
6371 		__mark_reg_known(&regs[value_regno], val);
6372 		return 0;
6373 	}
6374 
6375 reg_unknown:
6376 	mark_reg_unknown(env, regs, value_regno);
6377 	return 0;
6378 }
6379 
6380 /* check whether memory at (regno + off) is accessible for t = (read | write)
6381  * if t==write, value_regno is a register which value is stored into memory
6382  * if t==read, value_regno is a register which will receive the value from memory
6383  * if t==write && value_regno==-1, some unknown value is stored into memory
6384  * if t==read && value_regno==-1, don't care what we read from memory
6385  */
6386 static int check_mem_access(struct bpf_verifier_env *env, int insn_idx, struct bpf_reg_state *reg, argno_t argno,
6387 			    int off, int bpf_size, enum bpf_access_type t,
6388 			    int value_regno, bool strict_alignment_once, bool is_ldsx)
6389 {
6390 	struct bpf_reg_state *regs = cur_regs(env);
6391 	int size, err = 0;
6392 
6393 	size = bpf_size_to_bytes(bpf_size);
6394 	if (size < 0)
6395 		return size;
6396 
6397 	err = check_ptr_alignment(env, reg, off, size, strict_alignment_once);
6398 	if (err)
6399 		return err;
6400 
6401 	if (reg->type == PTR_TO_MAP_KEY) {
6402 		if (t == BPF_WRITE) {
6403 			verbose(env, "write to change key %s not allowed\n",
6404 				reg_arg_name(env, argno));
6405 			return -EACCES;
6406 		}
6407 
6408 		err = check_mem_region_access(env, reg, argno, off, size,
6409 					      reg->map_ptr->key_size, false);
6410 		if (err)
6411 			return err;
6412 		if (value_regno >= 0)
6413 			mark_reg_unknown(env, regs, value_regno);
6414 	} else if (reg->type == PTR_TO_MAP_VALUE) {
6415 		struct btf_field *kptr_field = NULL;
6416 
6417 		if (t == BPF_WRITE && value_regno >= 0 &&
6418 		    is_pointer_value(env, value_regno)) {
6419 			verbose(env, "R%d leaks addr into map\n", value_regno);
6420 			return -EACCES;
6421 		}
6422 		err = check_map_access_type(env, reg, off, size, t);
6423 		if (err)
6424 			return err;
6425 		err = check_map_access(env, reg, argno, off, size, false, ACCESS_DIRECT);
6426 		if (err)
6427 			return err;
6428 		if (tnum_is_const(reg->var_off))
6429 			kptr_field = btf_record_find(reg->map_ptr->record,
6430 						     off + reg->var_off.value, BPF_KPTR | BPF_UPTR);
6431 		if (kptr_field) {
6432 			err = check_map_kptr_access(env, value_regno, insn_idx, kptr_field);
6433 		} else if (t == BPF_READ && value_regno >= 0) {
6434 			err = check_map_mem_read(env, reg, off, bpf_size, value_regno, is_ldsx);
6435 		}
6436 	} else if (base_type(reg->type) == PTR_TO_MEM) {
6437 		bool rdonly_mem = type_is_rdonly_mem(reg->type);
6438 		bool rdonly_untrusted = rdonly_mem && (reg->type & PTR_UNTRUSTED);
6439 
6440 		if (type_may_be_null(reg->type)) {
6441 			verbose(env, "%s invalid mem access '%s'\n", reg_arg_name(env, argno),
6442 				reg_type_str(env, reg->type));
6443 			bpf_diag_invalid_deref(env, insn_idx, reg_from_argno(argno),
6444 					       reg_arg_name(env, argno), reg,
6445 						      BPF_DIAG_DEREF_NULLABLE_PTR, 0);
6446 			return -EACCES;
6447 		}
6448 
6449 		if (t == BPF_WRITE && rdonly_mem) {
6450 			verbose(env, "%s cannot write into %s\n",
6451 				reg_arg_name(env, argno), reg_type_str(env, reg->type));
6452 			return -EACCES;
6453 		}
6454 
6455 		if (t == BPF_WRITE && value_regno >= 0 &&
6456 		    is_pointer_value(env, value_regno)) {
6457 			verbose(env, "R%d leaks addr into mem\n", value_regno);
6458 			return -EACCES;
6459 		}
6460 
6461 		if (rdonly_untrusted && !env->allow_ptr_leaks) {
6462 			verbose(env, "%s access is allowed only to CAP_PERFMON and CAP_SYS_ADMIN\n",
6463 				reg_type_str(env, reg->type));
6464 			bpf_diag_policy(env, insn_idx, "read from untrusted read-only memory",
6465 					"the access requires CAP_PERFMON",
6466 					"Load the program with CAP_PERFMON, or avoid dereferencing untrusted pointers.");
6467 			return -EPERM;
6468 		}
6469 
6470 		/*
6471 		 * Accesses to untrusted PTR_TO_MEM are done through probe
6472 		 * instructions, hence no need to check bounds in that case.
6473 		 */
6474 		if (!rdonly_untrusted)
6475 			err = check_mem_region_access(env, reg, argno, off, size,
6476 						      reg->mem_size, false);
6477 		if (!err && value_regno >= 0 && (t == BPF_READ || rdonly_mem))
6478 			mark_reg_unknown(env, regs, value_regno);
6479 	} else if (reg->type == PTR_TO_CTX) {
6480 		struct bpf_insn_access_aux info = {
6481 			.reg_type = SCALAR_VALUE,
6482 			.is_ldsx = is_ldsx,
6483 			.log = &env->log,
6484 		};
6485 		struct bpf_retval_range range;
6486 
6487 		if (t == BPF_WRITE && value_regno >= 0 &&
6488 		    is_pointer_value(env, value_regno)) {
6489 			verbose(env, "R%d leaks addr into ctx\n", value_regno);
6490 			return -EACCES;
6491 		}
6492 
6493 		err = check_ctx_access(env, insn_idx, reg, argno, off, size, t, &info);
6494 		if (!err && t == BPF_READ && value_regno >= 0) {
6495 			/* ctx access returns either a scalar, or a
6496 			 * PTR_TO_PACKET[_META,_END]. In the latter
6497 			 * case, we know the offset is zero.
6498 			 */
6499 			if (info.reg_type == SCALAR_VALUE) {
6500 				if (info.is_retval && get_func_retval_range(env->prog, &range)) {
6501 					mark_reg_unknown(env, regs, value_regno);
6502 					err = __mark_reg_s32_range(env, regs, value_regno,
6503 								   range.minval, range.maxval);
6504 					if (err)
6505 						return err;
6506 				} else {
6507 					mark_reg_unknown(env, regs, value_regno);
6508 				}
6509 			} else {
6510 				mark_reg_known_zero(env, regs,
6511 						    value_regno);
6512 				if (base_type(info.reg_type) == PTR_TO_BTF_ID) {
6513 					regs[value_regno].btf = info.btf;
6514 					regs[value_regno].btf_id = info.btf_id;
6515 					regs[value_regno].id = info.ref_id;
6516 				}
6517 				if (type_may_be_null(info.reg_type) && !regs[value_regno].id)
6518 					regs[value_regno].id = ++env->id_gen;
6519 			}
6520 			regs[value_regno].type = info.reg_type;
6521 		}
6522 
6523 	} else if (reg->type == PTR_TO_STACK) {
6524 		/* Basic bounds checks. */
6525 		err = check_stack_access_within_bounds(env, reg, argno, off, size, t);
6526 		if (err)
6527 			return err;
6528 
6529 		if (t == BPF_READ)
6530 			err = check_stack_read(env, reg, argno, off, size,
6531 					       value_regno);
6532 		else
6533 			err = check_stack_write(env, reg, off, size,
6534 						value_regno, insn_idx);
6535 	} else if (reg_is_pkt_pointer(reg)) {
6536 		if (t == BPF_WRITE && !may_access_direct_pkt_data(env, NULL, t)) {
6537 			verbose(env, "cannot write into packet\n");
6538 			return -EACCES;
6539 		}
6540 		if (t == BPF_WRITE && value_regno >= 0 &&
6541 		    is_pointer_value(env, value_regno)) {
6542 			verbose(env, "R%d leaks addr into packet\n",
6543 				value_regno);
6544 			return -EACCES;
6545 		}
6546 		err = check_packet_access(env, reg, argno, off, size, false);
6547 		if (!err && t == BPF_READ && value_regno >= 0)
6548 			mark_reg_unknown(env, regs, value_regno);
6549 	} else if (reg->type == PTR_TO_FLOW_KEYS) {
6550 		if (t == BPF_WRITE && value_regno >= 0 &&
6551 		    is_pointer_value(env, value_regno)) {
6552 			verbose(env, "R%d leaks addr into flow keys\n",
6553 				value_regno);
6554 			return -EACCES;
6555 		}
6556 
6557 		err = check_flow_keys_access(env, reg, argno, off, size);
6558 		if (!err && t == BPF_READ && value_regno >= 0)
6559 			mark_reg_unknown(env, regs, value_regno);
6560 	} else if (type_is_sk_pointer(reg->type)) {
6561 		if (t == BPF_WRITE) {
6562 			verbose(env, "%s cannot write into %s\n",
6563 				reg_arg_name(env, argno), reg_type_str(env, reg->type));
6564 			return -EACCES;
6565 		}
6566 		err = check_sock_access(env, insn_idx, reg, argno, off, size, t);
6567 		if (!err && value_regno >= 0)
6568 			mark_reg_unknown(env, regs, value_regno);
6569 	} else if (reg->type == PTR_TO_TP_BUFFER) {
6570 		err = check_tp_buffer_access(env, reg, argno, off, size);
6571 		if (!err && t == BPF_READ && value_regno >= 0)
6572 			mark_reg_unknown(env, regs, value_regno);
6573 	} else if (base_type(reg->type) == PTR_TO_BTF_ID &&
6574 		   !type_may_be_null(reg->type)) {
6575 		err = check_ptr_to_btf_access(env, regs, reg, argno, off, size, t,
6576 					      value_regno);
6577 	} else if (reg->type == CONST_PTR_TO_MAP) {
6578 		err = check_ptr_to_map_access(env, regs, reg, argno, off, size, t,
6579 					      value_regno);
6580 	} else if (base_type(reg->type) == PTR_TO_BUF &&
6581 		   !type_may_be_null(reg->type)) {
6582 		bool rdonly_mem = type_is_rdonly_mem(reg->type);
6583 		u32 *max_access;
6584 
6585 		if (rdonly_mem) {
6586 			if (t == BPF_WRITE) {
6587 				verbose(env, "%s cannot write into %s\n",
6588 					reg_arg_name(env, argno), reg_type_str(env, reg->type));
6589 				return -EACCES;
6590 			}
6591 			max_access = &env->prog->aux->max_rdonly_access;
6592 		} else {
6593 			max_access = &env->prog->aux->max_rdwr_access;
6594 		}
6595 
6596 		err = check_buffer_access(env, reg, argno, off, size, false,
6597 					  max_access);
6598 
6599 		if (!err && value_regno >= 0 && (rdonly_mem || t == BPF_READ))
6600 			mark_reg_unknown(env, regs, value_regno);
6601 	} else if (reg->type == PTR_TO_ARENA) {
6602 		if (t == BPF_READ && value_regno >= 0)
6603 			mark_reg_unknown(env, regs, value_regno);
6604 	} else {
6605 		enum bpf_diag_invalid_deref_kind kind = BPF_DIAG_DEREF_INVALID_PTR;
6606 
6607 		verbose(env, "%s invalid mem access '%s'\n", reg_arg_name(env, argno),
6608 			reg_type_str(env, reg->type));
6609 		if (reg->type == SCALAR_VALUE)
6610 			kind = BPF_DIAG_DEREF_SCALAR;
6611 		else if (type_may_be_null(reg->type))
6612 			kind = BPF_DIAG_DEREF_NULLABLE_PTR;
6613 		bpf_diag_invalid_deref(env, insn_idx, reg_from_argno(argno),
6614 				       reg_arg_name(env, argno), reg, kind, 0);
6615 		return -EACCES;
6616 	}
6617 
6618 	if (!err && size < BPF_REG_SIZE && value_regno >= 0 && t == BPF_READ &&
6619 	    regs[value_regno].type == SCALAR_VALUE) {
6620 		if (!is_ldsx) {
6621 			/* b/h/w load zero-extends, mark upper bits as known 0 */
6622 			coerce_reg_to_size(&regs[value_regno], size);
6623 		} else {
6624 			/*
6625 			 * Sign-extension can change the register value relative
6626 			 * to a scalar it is linked with by id (e.g. a zero-
6627 			 * extending fill of the same spilled stack slot), thus
6628 			 * drop the shared id in that case.
6629 			 */
6630 			bool no_sext = reg_umax(&regs[value_regno]) <
6631 					(1ULL << (size * BITS_PER_BYTE - 1));
6632 
6633 			coerce_reg_to_size_sx(&regs[value_regno], size);
6634 			if (!no_sext)
6635 				clear_scalar_id(&regs[value_regno]);
6636 		}
6637 	}
6638 	return err;
6639 }
6640 
6641 static int save_aux_ptr_type(struct bpf_verifier_env *env, enum bpf_reg_type type,
6642 			     bool allow_trust_mismatch);
6643 
6644 static int check_load_mem(struct bpf_verifier_env *env, struct bpf_insn *insn,
6645 			  bool strict_alignment_once, bool is_ldsx,
6646 			  bool allow_trust_mismatch, const char *ctx)
6647 {
6648 	struct bpf_verifier_state *vstate = env->cur_state;
6649 	struct bpf_func_state *state = vstate->frame[vstate->curframe];
6650 	struct bpf_reg_state *regs = cur_regs(env);
6651 	enum bpf_reg_type src_reg_type;
6652 	int err;
6653 
6654 	/* Handle stack arg read */
6655 	if (is_stack_arg_ldx(insn)) {
6656 		err = check_reg_arg(env, insn->dst_reg, DST_OP_NO_MARK);
6657 		if (err)
6658 			return err;
6659 		return check_stack_arg_read(env, state, insn->off, insn->dst_reg);
6660 	}
6661 
6662 	/* check src operand */
6663 	err = check_reg_arg(env, insn->src_reg, SRC_OP);
6664 	if (err)
6665 		return err;
6666 
6667 	/* check dst operand */
6668 	err = check_reg_arg(env, insn->dst_reg, DST_OP_NO_MARK);
6669 	if (err)
6670 		return err;
6671 
6672 	src_reg_type = regs[insn->src_reg].type;
6673 
6674 	/*
6675 	 * check_stack_read_fixed_off() may refine the modification's origin to
6676 	 * the source stack slot.
6677 	 */
6678 	bpf_diag_mod_begin(env, &regs[insn->dst_reg], NULL, BPF_DIAG_MOD_WRITE);
6679 	err = check_mem_access(env, env->insn_idx, regs + insn->src_reg, argno_from_reg(insn->src_reg), insn->off,
6680 			       BPF_SIZE(insn->code), BPF_READ, insn->dst_reg,
6681 			       strict_alignment_once, is_ldsx);
6682 	err = err ?: save_aux_ptr_type(env, src_reg_type,
6683 				       allow_trust_mismatch);
6684 	err = err ?: reg_bounds_sanity_check(env, &regs[insn->dst_reg], ctx);
6685 	if (!err)
6686 		bpf_diag_mod_end(env);
6687 
6688 	return err;
6689 }
6690 
6691 static int check_store_reg(struct bpf_verifier_env *env, struct bpf_insn *insn,
6692 			   bool strict_alignment_once)
6693 {
6694 	struct bpf_verifier_state *vstate = env->cur_state;
6695 	struct bpf_func_state *state = vstate->frame[vstate->curframe];
6696 	struct bpf_reg_state *regs = cur_regs(env);
6697 	enum bpf_reg_type dst_reg_type;
6698 	int err;
6699 
6700 	/* Handle stack arg write */
6701 	if (is_stack_arg_stx(insn)) {
6702 		err = check_reg_arg(env, insn->src_reg, SRC_OP);
6703 		if (err)
6704 			return err;
6705 		return check_stack_arg_write(env, state, insn->off, regs + insn->src_reg);
6706 	}
6707 
6708 	/* check src1 operand */
6709 	err = check_reg_arg(env, insn->src_reg, SRC_OP);
6710 	if (err)
6711 		return err;
6712 
6713 	/* check src2 operand */
6714 	err = check_reg_arg(env, insn->dst_reg, SRC_OP);
6715 	if (err)
6716 		return err;
6717 
6718 	dst_reg_type = regs[insn->dst_reg].type;
6719 
6720 	/* Check if (dst_reg + off) is writeable. */
6721 	err = check_mem_access(env, env->insn_idx, regs + insn->dst_reg, argno_from_reg(insn->dst_reg), insn->off,
6722 			       BPF_SIZE(insn->code), BPF_WRITE, insn->src_reg,
6723 			       strict_alignment_once, false);
6724 	err = err ?: save_aux_ptr_type(env, dst_reg_type, false);
6725 
6726 	return err;
6727 }
6728 
6729 static int check_atomic_rmw(struct bpf_verifier_env *env,
6730 			    struct bpf_insn *insn)
6731 {
6732 	struct bpf_reg_state *dst_reg;
6733 	int load_reg;
6734 	int err;
6735 
6736 	if (BPF_SIZE(insn->code) != BPF_W && BPF_SIZE(insn->code) != BPF_DW) {
6737 		verbose(env, "invalid atomic operand size\n");
6738 		return -EINVAL;
6739 	}
6740 
6741 	/* check src1 operand */
6742 	err = check_reg_arg(env, insn->src_reg, SRC_OP);
6743 	if (err)
6744 		return err;
6745 
6746 	/* check src2 operand */
6747 	err = check_reg_arg(env, insn->dst_reg, SRC_OP);
6748 	if (err)
6749 		return err;
6750 
6751 	if (insn->imm == BPF_CMPXCHG) {
6752 		/* Check comparison of R0 with memory location */
6753 		const u32 aux_reg = BPF_REG_0;
6754 
6755 		err = check_reg_arg(env, aux_reg, SRC_OP);
6756 		if (err)
6757 			return err;
6758 
6759 		if (is_pointer_value(env, aux_reg)) {
6760 			verbose(env, "R%d leaks addr into mem\n", aux_reg);
6761 			return -EACCES;
6762 		}
6763 	}
6764 
6765 	if (is_pointer_value(env, insn->src_reg)) {
6766 		verbose(env, "R%d leaks addr into mem\n", insn->src_reg);
6767 		return -EACCES;
6768 	}
6769 
6770 	if (!atomic_ptr_type_ok(env, insn->dst_reg, insn)) {
6771 		verbose(env, "BPF_ATOMIC stores into R%d %s is not allowed\n",
6772 			insn->dst_reg,
6773 			reg_type_str(env, reg_state(env, insn->dst_reg)->type));
6774 		return -EACCES;
6775 	}
6776 
6777 	load_reg = bpf_atomic_load_reg(insn);
6778 	if (load_reg >= 0) {
6779 		/* check and record load of old value */
6780 		err = check_reg_arg(env, load_reg, DST_OP);
6781 		if (err)
6782 			return err;
6783 	}
6784 
6785 	dst_reg = cur_regs(env) + insn->dst_reg;
6786 
6787 	/* Check whether we can read the memory, with second call for fetch
6788 	 * case to simulate the register fill.
6789 	 */
6790 	err = check_mem_access(env, env->insn_idx, dst_reg, argno_from_reg(insn->dst_reg), insn->off,
6791 			       BPF_SIZE(insn->code), BPF_READ, -1, true, false);
6792 	if (!err && load_reg >= 0) {
6793 		bpf_diag_mod_begin(env, cur_regs(env) + load_reg, NULL, BPF_DIAG_MOD_WRITE);
6794 		err = check_mem_access(env, env->insn_idx, dst_reg, argno_from_reg(insn->dst_reg),
6795 				       insn->off, BPF_SIZE(insn->code),
6796 				       BPF_READ, load_reg, true, false);
6797 		if (!err)
6798 			bpf_diag_mod_end(env);
6799 	}
6800 	if (err)
6801 		return err;
6802 
6803 	err = save_aux_ptr_type(env, dst_reg->type, false);
6804 	if (err)
6805 		return err;
6806 	/* Check whether we can write into the same memory. */
6807 	err = check_mem_access(env, env->insn_idx, dst_reg, argno_from_reg(insn->dst_reg), insn->off,
6808 			       BPF_SIZE(insn->code), BPF_WRITE, -1, true, false);
6809 	if (err)
6810 		return err;
6811 	return 0;
6812 }
6813 
6814 static int check_atomic_load(struct bpf_verifier_env *env,
6815 			     struct bpf_insn *insn)
6816 {
6817 	int err;
6818 
6819 	err = check_reg_arg(env, insn->src_reg, SRC_OP);
6820 	if (err)
6821 		return err;
6822 
6823 	if (!atomic_ptr_type_ok(env, insn->src_reg, insn)) {
6824 		verbose(env, "BPF_ATOMIC loads from R%d %s is not allowed\n",
6825 			insn->src_reg,
6826 			reg_type_str(env, reg_state(env, insn->src_reg)->type));
6827 		return -EACCES;
6828 	}
6829 
6830 	return check_load_mem(env, insn, true, false, false, "atomic_load");
6831 }
6832 
6833 static int check_atomic_store(struct bpf_verifier_env *env,
6834 			      struct bpf_insn *insn)
6835 {
6836 	int err;
6837 
6838 	err = check_store_reg(env, insn, true);
6839 	if (err)
6840 		return err;
6841 
6842 	if (!atomic_ptr_type_ok(env, insn->dst_reg, insn)) {
6843 		verbose(env, "BPF_ATOMIC stores into R%d %s is not allowed\n",
6844 			insn->dst_reg,
6845 			reg_type_str(env, reg_state(env, insn->dst_reg)->type));
6846 		return -EACCES;
6847 	}
6848 
6849 	return 0;
6850 }
6851 
6852 static int check_atomic(struct bpf_verifier_env *env, struct bpf_insn *insn)
6853 {
6854 	switch (insn->imm) {
6855 	case BPF_ADD:
6856 	case BPF_ADD | BPF_FETCH:
6857 	case BPF_AND:
6858 	case BPF_AND | BPF_FETCH:
6859 	case BPF_OR:
6860 	case BPF_OR | BPF_FETCH:
6861 	case BPF_XOR:
6862 	case BPF_XOR | BPF_FETCH:
6863 	case BPF_XCHG:
6864 	case BPF_CMPXCHG:
6865 		return check_atomic_rmw(env, insn);
6866 	case BPF_LOAD_ACQ:
6867 		if (BPF_SIZE(insn->code) == BPF_DW && BITS_PER_LONG != 64) {
6868 			verbose(env,
6869 				"64-bit load-acquires are only supported on 64-bit arches\n");
6870 			return -EOPNOTSUPP;
6871 		}
6872 		return check_atomic_load(env, insn);
6873 	case BPF_STORE_REL:
6874 		if (BPF_SIZE(insn->code) == BPF_DW && BITS_PER_LONG != 64) {
6875 			verbose(env,
6876 				"64-bit store-releases are only supported on 64-bit arches\n");
6877 			return -EOPNOTSUPP;
6878 		}
6879 		return check_atomic_store(env, insn);
6880 	default:
6881 		verbose(env, "BPF_ATOMIC uses invalid atomic opcode %02x\n",
6882 			insn->imm);
6883 		return -EINVAL;
6884 	}
6885 }
6886 
6887 /* When register 'regno' is used to read the stack (either directly or through
6888  * a helper function) make sure that it's within stack boundary and, depending
6889  * on the access type and privileges, that all elements of the stack are
6890  * initialized.
6891  *
6892  * All registers that have been spilled on the stack in the slots within the
6893  * read offsets are marked as read.
6894  */
6895 static int check_stack_range_initialized(
6896 		struct bpf_verifier_env *env, struct bpf_reg_state *reg, argno_t argno, int off,
6897 		int access_size, bool zero_size_allowed,
6898 		enum bpf_access_type type, struct bpf_call_arg_meta *meta)
6899 {
6900 	struct bpf_func_state *state = bpf_func(env, reg);
6901 	int err, min_off, max_off, i, j, slot, spi;
6902 	/* Some accesses can write anything into the stack, others are
6903 	 * read-only.
6904 	 */
6905 	bool clobber = type == BPF_WRITE;
6906 	/*
6907 	 * Negative access_size signals global subprog arg check where
6908 	 * STACK_POISON slots are acceptable. static stack liveness
6909 	 * might have determined that subprog doesn't read them,
6910 	 * but BTF based global subprog validation isn't accurate enough.
6911 	 */
6912 	bool allow_poison = access_size < 0 || clobber;
6913 	/* The call will initialize the memory; uninitialized stack allowed */
6914 	bool raw_mode = meta && meta->arg_raw_mem.regno == reg_from_argno(argno);
6915 
6916 	access_size = abs(access_size);
6917 
6918 	if (access_size == 0 && !zero_size_allowed) {
6919 		verbose(env, "invalid zero-sized read\n");
6920 		return -EACCES;
6921 	}
6922 
6923 	err = check_stack_access_within_bounds(env, reg, argno, off, access_size, type);
6924 	if (err)
6925 		return err;
6926 
6927 	if (tnum_is_const(reg->var_off)) {
6928 		min_off = max_off = reg->var_off.value + off;
6929 	} else {
6930 		/* Variable offset is prohibited for unprivileged mode for
6931 		 * simplicity since it requires corresponding support in
6932 		 * Spectre masking for stack ALU.
6933 		 * See also retrieve_ptr_limit().
6934 		 */
6935 		if (!env->bypass_spec_v1) {
6936 			char tn_buf[48];
6937 
6938 			tnum_strn(tn_buf, sizeof(tn_buf), reg->var_off);
6939 			verbose(env, "%s variable offset stack access prohibited for !root, var_off=%s\n",
6940 				reg_arg_name(env, argno), tn_buf);
6941 			return -EACCES;
6942 		}
6943 		/* Only initialized buffer on stack is allowed to be accessed
6944 		 * with variable offset. With uninitialized buffer it's hard to
6945 		 * guarantee that whole memory is marked as initialized on
6946 		 * helper return since specific bounds are unknown what may
6947 		 * cause uninitialized stack leaking.
6948 		 */
6949 		raw_mode = false;
6950 
6951 		min_off = reg_smin(reg) + off;
6952 		max_off = reg_smax(reg) + off;
6953 	}
6954 
6955 	if (raw_mode) {
6956 		meta->arg_raw_mem.size = access_size;
6957 		return 0;
6958 	}
6959 
6960 	for (i = min_off; i < max_off + access_size; i++) {
6961 		u8 *stype;
6962 
6963 		slot = -i - 1;
6964 		spi = slot / BPF_REG_SIZE;
6965 		if (state->allocated_stack <= slot) {
6966 			verbose(env, "allocated_stack too small\n");
6967 			return -EFAULT;
6968 		}
6969 
6970 		stype = &state->stack[spi].slot_type[slot % BPF_REG_SIZE];
6971 		if (*stype == STACK_MISC)
6972 			goto mark;
6973 		if ((*stype == STACK_ZERO) ||
6974 		    (*stype == STACK_INVALID && env->allow_uninit_stack)) {
6975 			if (clobber) {
6976 				/* helper can write anything into the stack */
6977 				*stype = STACK_MISC;
6978 			}
6979 			goto mark;
6980 		}
6981 
6982 		if (bpf_is_spilled_reg(&state->stack[spi]) &&
6983 		    (state->stack[spi].spilled_ptr.type == SCALAR_VALUE ||
6984 		     env->allow_ptr_leaks)) {
6985 			if (clobber) {
6986 				__mark_reg_unknown(env, &state->stack[spi].spilled_ptr);
6987 				for (j = 0; j < BPF_REG_SIZE; j++)
6988 					scrub_spilled_slot(&state->stack[spi].slot_type[j]);
6989 			}
6990 			goto mark;
6991 		}
6992 
6993 		if (*stype == STACK_POISON) {
6994 			if (allow_poison)
6995 				goto mark;
6996 			verbose(env, "reading from stack %s off %d+%d size %d, slot poisoned by dead code elimination\n",
6997 				reg_arg_name(env, argno), min_off, i - min_off, access_size);
6998 		} else if (tnum_is_const(reg->var_off)) {
6999 			verbose(env, "invalid read from stack %s off %d+%d size %d\n",
7000 				reg_arg_name(env, argno), min_off, i - min_off, access_size);
7001 		} else {
7002 			char tn_buf[48];
7003 
7004 			tnum_strn(tn_buf, sizeof(tn_buf), reg->var_off);
7005 			verbose(env, "invalid read from stack %s var_off %s+%d size %d\n",
7006 				reg_arg_name(env, argno), tn_buf, i - min_off, access_size);
7007 		}
7008 		return -EACCES;
7009 mark:
7010 		;
7011 	}
7012 	return 0;
7013 }
7014 
7015 static int check_helper_mem_access(struct bpf_verifier_env *env, struct bpf_reg_state *reg,
7016 				   argno_t argno, int access_size,
7017 				   enum bpf_access_type access_type, bool zero_size_allowed,
7018 				   struct bpf_call_arg_meta *meta, bool *known_memory)
7019 {
7020 	struct bpf_reg_state *regs = cur_regs(env);
7021 	u32 *max_access;
7022 
7023 	if (known_memory)
7024 		*known_memory = true;
7025 
7026 	switch (base_type(reg->type)) {
7027 	case PTR_TO_PACKET:
7028 	case PTR_TO_PACKET_META:
7029 		return check_packet_access(env, reg, argno, 0, access_size,
7030 					   zero_size_allowed);
7031 	case PTR_TO_MAP_KEY:
7032 		if (access_type == BPF_WRITE) {
7033 			verbose(env, "%s cannot write into %s\n",
7034 				reg_arg_name(env, argno), reg_type_str(env, reg->type));
7035 			return -EACCES;
7036 		}
7037 		return check_mem_region_access(env, reg, argno, 0, access_size,
7038 					       reg->map_ptr->key_size, false);
7039 	case PTR_TO_MAP_VALUE:
7040 		if (check_map_access_type(env, reg, 0, access_size, access_type))
7041 			return -EACCES;
7042 		return check_map_access(env, reg, argno, 0, access_size,
7043 					zero_size_allowed, ACCESS_HELPER);
7044 	case PTR_TO_MEM:
7045 		if (type_is_rdonly_mem(reg->type)) {
7046 			if (access_type == BPF_WRITE) {
7047 				verbose(env, "%s cannot write into %s\n",
7048 					reg_arg_name(env, argno), reg_type_str(env, reg->type));
7049 				return -EACCES;
7050 			}
7051 		}
7052 		return check_mem_region_access(env, reg, argno, 0,
7053 					       access_size, reg->mem_size,
7054 					       zero_size_allowed);
7055 	case PTR_TO_BUF:
7056 		if (type_is_rdonly_mem(reg->type)) {
7057 			if (access_type == BPF_WRITE) {
7058 				verbose(env, "%s cannot write into %s\n",
7059 					reg_arg_name(env, argno), reg_type_str(env, reg->type));
7060 				return -EACCES;
7061 			}
7062 
7063 			max_access = &env->prog->aux->max_rdonly_access;
7064 		} else {
7065 			max_access = &env->prog->aux->max_rdwr_access;
7066 		}
7067 		return check_buffer_access(env, reg, argno, 0,
7068 					   access_size, zero_size_allowed,
7069 					   max_access);
7070 	case PTR_TO_STACK:
7071 		return check_stack_range_initialized(
7072 				env, reg,
7073 				argno, 0, access_size,
7074 				zero_size_allowed, access_type, meta);
7075 	case PTR_TO_BTF_ID:
7076 		return check_ptr_to_btf_access(env, regs, reg, argno, 0,
7077 					       access_size, access_type, -1);
7078 	case PTR_TO_CTX:
7079 		/* Only permit reading or writing syscall context using helper calls. */
7080 		if (is_var_ctx_off_allowed(env->prog)) {
7081 			int err = check_mem_region_access(env, reg, argno, 0, access_size, U16_MAX,
7082 							  zero_size_allowed);
7083 			if (err)
7084 				return err;
7085 			if (env->prog->aux->max_ctx_offset < reg_umax(reg) + access_size)
7086 				env->prog->aux->max_ctx_offset = reg_umax(reg) + access_size;
7087 			return 0;
7088 		}
7089 		fallthrough;
7090 	default: /* scalar_value or invalid ptr */
7091 		/* Allow zero-byte read from NULL, regardless of pointer type */
7092 		if (zero_size_allowed && access_size == 0 &&
7093 		    bpf_register_is_null(reg))
7094 			return 0;
7095 		if (known_memory && base_type(reg->type) != PTR_TO_CTX)
7096 			*known_memory = false;
7097 
7098 		verbose(env, "%s type=%s ", reg_arg_name(env, argno),
7099 			reg_type_str(env, reg->type));
7100 		verbose(env, "expected=%s\n", reg_type_str(env, PTR_TO_STACK));
7101 		return -EACCES;
7102 	}
7103 }
7104 
7105 enum bpf_mem_size_failure {
7106 	BPF_MEM_SIZE_FAIL_NONE,
7107 	BPF_MEM_SIZE_FAIL_MEMORY,
7108 	BPF_MEM_SIZE_FAIL_SIZE,
7109 };
7110 
7111 /* verify arguments to helpers or kfuncs consisting of a pointer and an access
7112  * size.
7113  *
7114  * @mem_reg contains the pointer, @size_reg contains the access size.
7115  */
7116 static int check_mem_size_reg(struct bpf_verifier_env *env,
7117 			      struct bpf_reg_state *mem_reg,
7118 			      struct bpf_reg_state *size_reg, argno_t mem_argno,
7119 			      argno_t size_argno, u32 access_type,
7120 			      bool zero_size_allowed,
7121 			      struct bpf_call_arg_meta *meta,
7122 			      enum bpf_mem_size_failure *failure)
7123 {
7124 	int err = 0;
7125 
7126 	if (failure)
7127 		*failure = BPF_MEM_SIZE_FAIL_NONE;
7128 
7129 	/* This is used to refine r0 return value bounds for helpers
7130 	 * that enforce this value as an upper bound on return values.
7131 	 * See do_refine_retval_range() for helpers that can refine
7132 	 * the return value. C type of helper is u32 so we pull register
7133 	 * bound from umax_value however, if negative verifier errors
7134 	 * out. Only upper bounds can be learned because retval is an
7135 	 * int type and negative retvals are allowed.
7136 	 */
7137 	meta->msize_max_value = reg_umax(size_reg);
7138 
7139 	/* The register is SCALAR_VALUE; the access check happens using
7140 	 * its boundaries. For unprivileged variable accesses, disable
7141 	 * raw mode so that the program is required to initialize all
7142 	 * the memory that the helper could just partially fill up.
7143 	 */
7144 	if (!tnum_is_const(size_reg->var_off))
7145 		meta = NULL;
7146 
7147 	if (reg_smin(size_reg) < 0) {
7148 		verbose(env, "%s min value is negative, either use unsigned or 'var &= const'\n",
7149 			reg_arg_name(env, size_argno));
7150 		err = -EACCES;
7151 		goto size_error;
7152 	}
7153 
7154 	if (reg_umin(size_reg) == 0 && !zero_size_allowed) {
7155 		verbose(env, "%s invalid zero-sized read: u64=[%lld,%lld]\n",
7156 			reg_arg_name(env, size_argno), reg_umin(size_reg), reg_umax(size_reg));
7157 		err = -EACCES;
7158 		goto size_error;
7159 	}
7160 
7161 	if (reg_umax(size_reg) >= BPF_MAX_VAR_SIZ) {
7162 		verbose(env, "%s unbounded memory access, use 'var &= const' or 'if (var < const)'\n",
7163 			reg_arg_name(env, size_argno));
7164 		err = -EACCES;
7165 		goto size_error;
7166 	}
7167 
7168 	if (access_type & BPF_READ)
7169 		err = check_helper_mem_access(env, mem_reg, mem_argno, reg_umax(size_reg),
7170 					      BPF_READ, zero_size_allowed, meta, NULL);
7171 	if (!err && access_type & BPF_WRITE)
7172 		err = check_helper_mem_access(env, mem_reg, mem_argno, reg_umax(size_reg),
7173 					      BPF_WRITE, zero_size_allowed, meta, NULL);
7174 	if (err && failure)
7175 		*failure = BPF_MEM_SIZE_FAIL_MEMORY;
7176 
7177 	if (!err)
7178 		err = mark_arg_precision(env, size_argno);
7179 
7180 	return err;
7181 
7182 size_error:
7183 	if (failure)
7184 		*failure = BPF_MEM_SIZE_FAIL_SIZE;
7185 	return err;
7186 }
7187 
7188 static int check_mem_reg(struct bpf_verifier_env *env, struct bpf_reg_state *reg,
7189 			 argno_t argno, u32 mem_size, enum bpf_access_type access_type,
7190 			 struct bpf_call_arg_meta *meta, bool *known_memory)
7191 {
7192 	int size, err = 0;
7193 
7194 	if (bpf_register_is_null(reg))
7195 		return mark_arg_precision(env, argno);
7196 	if (known_memory)
7197 		*known_memory = true;
7198 
7199 	if (mem_size > S32_MAX) {
7200 		verbose(env, "%s memory size %u is too large\n",
7201 			reg_arg_name(env, argno), mem_size);
7202 		return -EACCES;
7203 	}
7204 
7205 	/*
7206 	 * Only a global subprog (meta == NULL) may read poisoned stack slots:
7207 	 * its static stack liveness proved the callee body skips them.
7208 	 */
7209 	size = (!meta && base_type(reg->type) == PTR_TO_STACK) ? -(int)mem_size : mem_size;
7210 
7211 	if (access_type & BPF_READ)
7212 		err = check_helper_mem_access(env, reg, argno, size, BPF_READ, true, meta,
7213 					      known_memory);
7214 	if (!err && (access_type & BPF_WRITE))
7215 		err = check_helper_mem_access(env, reg, argno, size, BPF_WRITE, true, meta,
7216 					      known_memory);
7217 
7218 	return err;
7219 }
7220 
7221 static int process_const_alloc_mem_size(struct bpf_verifier_env *env, struct bpf_reg_state *reg,
7222 					argno_t argno, struct ret_mem_desc *ret_mem)
7223 {
7224 	int regno = reg_from_argno(argno);
7225 	int err;
7226 
7227 	if (ret_mem->found) {
7228 		verifier_bug(env, "only one allocation size argument permitted");
7229 		return -EFAULT;
7230 	}
7231 
7232 	if (!tnum_is_const(reg->var_off)) {
7233 		verbose(env, "%s is not a const\n", reg_arg_name(env, argno));
7234 		return -EINVAL;
7235 	}
7236 
7237 	if (reg->var_off.value > U32_MAX) {
7238 		verbose(env, "%s allocation size exceeds u32 max\n", reg_arg_name(env, argno));
7239 		return -EINVAL;
7240 	}
7241 
7242 	if (regno >= 0)
7243 		err = mark_chain_precision(env, regno);
7244 	else
7245 		err = mark_stack_arg_precision(env, arg_idx_from_argno(argno));
7246 	if (err)
7247 		return err;
7248 
7249 	ret_mem->size = reg->var_off.value;
7250 	ret_mem->found = true;
7251 
7252 	return 0;
7253 }
7254 
7255 static int process_const_arg(struct bpf_verifier_env *env, struct bpf_reg_state *reg,
7256 			     argno_t argno, struct bpf_call_arg_meta *meta)
7257 {
7258 	int regno = reg_from_argno(argno);
7259 	int err;
7260 
7261 	if (meta->arg_constant.found) {
7262 		verifier_bug(env, "only one constant argument permitted");
7263 		return -EFAULT;
7264 	}
7265 
7266 	if (!tnum_is_const(reg->var_off)) {
7267 		verbose(env, "%s must be a known constant\n", reg_arg_name(env, argno));
7268 		return -EINVAL;
7269 	}
7270 
7271 	if (regno >= 0)
7272 		err = mark_chain_precision(env, regno);
7273 	else
7274 		err = mark_stack_arg_precision(env, arg_idx_from_argno(argno));
7275 	if (err < 0)
7276 		return err;
7277 
7278 	meta->arg_constant.found = true;
7279 	meta->arg_constant.value = reg->var_off.value;
7280 
7281 	return 0;
7282 }
7283 
7284 enum {
7285 	PROCESS_SPIN_LOCK = (1 << 0),
7286 	PROCESS_RES_LOCK  = (1 << 1),
7287 	PROCESS_LOCK_IRQ  = (1 << 2),
7288 };
7289 
7290 /* Implementation details:
7291  * bpf_map_lookup returns PTR_TO_MAP_VALUE_OR_NULL.
7292  * bpf_obj_new returns PTR_TO_BTF_ID | MEM_ALLOC | PTR_MAYBE_NULL.
7293  * Two bpf_map_lookups (even with the same key) will have different reg->id.
7294  * Two separate bpf_obj_new will also have different reg->id.
7295  * For traditional PTR_TO_MAP_VALUE or PTR_TO_BTF_ID | MEM_ALLOC, the verifier
7296  * clears reg->id after value_or_null->value transition, since the verifier only
7297  * cares about the range of access to valid map value pointer and doesn't care
7298  * about actual address of the map element.
7299  * For maps with 'struct bpf_spin_lock' inside map value the verifier keeps
7300  * reg->id > 0 after value_or_null->value transition. By doing so
7301  * two bpf_map_lookups will be considered two different pointers that
7302  * point to different bpf_spin_locks. Likewise for pointers to allocated objects
7303  * returned from bpf_obj_new.
7304  * The verifier allows taking only one bpf_spin_lock at a time to avoid
7305  * dead-locks.
7306  * Since only one bpf_spin_lock is allowed the checks are simpler than
7307  * reg_is_refcounted() logic. The verifier needs to remember only
7308  * one spin_lock instead of array of acquired_refs.
7309  * env->cur_state->active_locks remembers which map value element or allocated
7310  * object got locked and clears it after bpf_spin_unlock.
7311  */
7312 static int process_spin_lock(struct bpf_verifier_env *env, struct bpf_reg_state *reg, argno_t argno, int flags)
7313 {
7314 	bool is_lock = flags & PROCESS_SPIN_LOCK, is_res_lock = flags & PROCESS_RES_LOCK;
7315 	const char *lock_str = is_res_lock ? "bpf_res_spin" : "bpf_spin";
7316 	struct bpf_verifier_state *cur = env->cur_state;
7317 	struct bpf_reference_state *lock;
7318 	bool is_const = tnum_is_const(reg->var_off);
7319 	bool is_irq = flags & PROCESS_LOCK_IRQ;
7320 	u64 val = reg->var_off.value;
7321 	struct bpf_map *map = NULL;
7322 	struct btf *btf = NULL;
7323 	struct btf_record *rec;
7324 	u32 spin_lock_off;
7325 	int err;
7326 
7327 	if (!is_const) {
7328 		verbose(env,
7329 			"%s doesn't have constant offset. %s_lock has to be at the constant offset\n",
7330 			reg_arg_name(env, argno), lock_str);
7331 		return -EINVAL;
7332 	}
7333 	if (reg->type == PTR_TO_MAP_VALUE) {
7334 		map = reg->map_ptr;
7335 		if (!map->btf) {
7336 			verbose(env,
7337 				"map '%s' has to have BTF in order to use %s_lock\n",
7338 				map->name, lock_str);
7339 			return -EINVAL;
7340 		}
7341 	} else {
7342 		btf = reg->btf;
7343 	}
7344 
7345 	rec = reg_btf_record(reg);
7346 	if (!btf_record_has_field(rec, is_res_lock ? BPF_RES_SPIN_LOCK : BPF_SPIN_LOCK)) {
7347 		verbose(env, "%s '%s' has no valid %s_lock\n", map ? "map" : "local",
7348 			map ? map->name : "kptr", lock_str);
7349 		return -EINVAL;
7350 	}
7351 	spin_lock_off = is_res_lock ? rec->res_spin_lock_off : rec->spin_lock_off;
7352 	if (spin_lock_off != val) {
7353 		verbose(env, "off %lld doesn't point to 'struct %s_lock' that is at %d\n",
7354 			val, lock_str, spin_lock_off);
7355 		return -EINVAL;
7356 	}
7357 	if (is_lock) {
7358 		void *ptr;
7359 		int type;
7360 
7361 		if (map)
7362 			ptr = map;
7363 		else
7364 			ptr = btf;
7365 
7366 		if (!is_res_lock && cur->active_locks) {
7367 			lock = find_lock_state(cur, REF_TYPE_LOCK, 0, NULL);
7368 			if (lock) {
7369 				verbose(env,
7370 					"Locking two bpf_spin_locks are not allowed\n");
7371 				bpf_diag_lock(
7372 					env, env->insn_idx, "nested spin lock",
7373 					"This path already holds a bpf_spin_lock. The verifier allows only one regular BPF spin lock at a time.",
7374 					"Unlock the current bpf_spin_lock before taking another one.", lock);
7375 				return -EINVAL;
7376 			}
7377 		} else if (is_res_lock && cur->active_locks) {
7378 			lock = find_lock_state(cur, REF_TYPE_RES_LOCK | REF_TYPE_RES_LOCK_IRQ,
7379 					       reg->id, ptr);
7380 			if (lock) {
7381 				verbose(env, "Acquiring the same lock again, AA deadlock detected\n");
7382 				bpf_diag_lock(
7383 					env, env->insn_idx, "recursive resource spin lock",
7384 					"This path already holds the same resource spin lock. Taking it again would deadlock.",
7385 					"Avoid reacquiring the same resource spin lock before it is unlocked.", lock);
7386 				return -EINVAL;
7387 			}
7388 		}
7389 
7390 		if (is_res_lock && is_irq)
7391 			type = REF_TYPE_RES_LOCK_IRQ;
7392 		else if (is_res_lock)
7393 			type = REF_TYPE_RES_LOCK;
7394 		else
7395 			type = REF_TYPE_LOCK;
7396 		err = acquire_lock_state(env, env->insn_idx, type, reg->id, ptr);
7397 		if (err < 0) {
7398 			verbose(env, "Failed to acquire lock state\n");
7399 			return err;
7400 		}
7401 	} else {
7402 		void *ptr;
7403 		int type;
7404 
7405 		if (map)
7406 			ptr = map;
7407 		else
7408 			ptr = btf;
7409 
7410 		if (!cur->active_locks) {
7411 			verbose(env, "%s_unlock without taking a lock\n", lock_str);
7412 			bpf_diag_res(
7413 				env, env->insn_idx, "unlock without lock",
7414 				"This unlock operation has no matching active lock on the current path.",
7415 				"Take the matching lock before this unlock, or remove the unmatched unlock path.");
7416 			return -EINVAL;
7417 		}
7418 
7419 		if (is_res_lock && is_irq)
7420 			type = REF_TYPE_RES_LOCK_IRQ;
7421 		else if (is_res_lock)
7422 			type = REF_TYPE_RES_LOCK;
7423 		else
7424 			type = REF_TYPE_LOCK;
7425 
7426 		lock = find_lock_state(cur, type, reg->id, ptr);
7427 		if (!lock) {
7428 			verbose(env, "%s_unlock of different lock\n", lock_str);
7429 			lock = find_lock_state(cur, REF_TYPE_LOCK_MASK, cur->active_lock_id,
7430 					       cur->active_lock_ptr);
7431 			bpf_diag_lock(
7432 				env, env->insn_idx, "unlock of a different lock",
7433 				"This unlock does not match any active lock with the same tracked identity on the current path.",
7434 				"Unlock the same lock object that was most recently acquired.", lock);
7435 			return -EINVAL;
7436 		}
7437 		if (reg->id != cur->active_lock_id || ptr != cur->active_lock_ptr) {
7438 			verbose(env, "%s_unlock cannot be out of order\n", lock_str);
7439 			lock = find_lock_state(cur, REF_TYPE_LOCK_MASK, cur->active_lock_id,
7440 					       cur->active_lock_ptr);
7441 			bpf_diag_lock(
7442 				env, env->insn_idx, "unlock out of order",
7443 				"Locks must be released in last-in, first-out order, but this unlock does not match the currently active lock.",
7444 				"Release nested locks in the reverse order they were acquired.", lock);
7445 			return -EINVAL;
7446 		}
7447 		if (release_lock_state(env, type, reg->id, ptr)) {
7448 			verbose(env, "%s_unlock of different lock\n", lock_str);
7449 			bpf_diag_lock(
7450 				env, env->insn_idx, "unlock of a different lock",
7451 				"The verifier could not release a lock state matching this unlock operation.",
7452 				"Pass the same lock object and lock kind that were used for the matching lock operation.",
7453 				lock);
7454 			return -EINVAL;
7455 		}
7456 		/*
7457 		 * Invalidate non-owning refs before RCU demotion clears their
7458 		 * NON_OWN_REF flag.
7459 		 */
7460 		invalidate_non_owning_refs(env);
7461 
7462 		if (!in_rcu_cs(env))
7463 			invalidate_rcu_protected_refs(env);
7464 	}
7465 	return 0;
7466 }
7467 
7468 /* Check if @regno is a pointer to a specific field in a map value */
7469 static int check_map_field_pointer(struct bpf_verifier_env *env, struct bpf_reg_state *reg, argno_t argno,
7470 				   enum btf_field_type field_type,
7471 				   struct bpf_map_desc *map_desc)
7472 {
7473 	bool is_const = tnum_is_const(reg->var_off);
7474 	struct bpf_map *map = reg->map_ptr;
7475 	u64 val = reg->var_off.value;
7476 	const char *struct_name = btf_field_type_name(field_type);
7477 	int field_off = -1;
7478 
7479 	if (!is_const) {
7480 		verbose(env,
7481 			"%s doesn't have constant offset. %s has to be at the constant offset\n",
7482 			reg_arg_name(env, argno), struct_name);
7483 		return -EINVAL;
7484 	}
7485 	if (!map->btf) {
7486 		verbose(env, "map '%s' has to have BTF in order to use %s\n", map->name,
7487 			struct_name);
7488 		return -EINVAL;
7489 	}
7490 	if (!btf_record_has_field(map->record, field_type)) {
7491 		verbose(env, "map '%s' has no valid %s\n", map->name, struct_name);
7492 		return -EINVAL;
7493 	}
7494 	switch (field_type) {
7495 	case BPF_TIMER:
7496 		field_off = map->record->timer_off;
7497 		break;
7498 	case BPF_TASK_WORK:
7499 		field_off = map->record->task_work_off;
7500 		break;
7501 	case BPF_WORKQUEUE:
7502 		field_off = map->record->wq_off;
7503 		break;
7504 	default:
7505 		verifier_bug(env, "unsupported BTF field type: %s\n", struct_name);
7506 		return -EINVAL;
7507 	}
7508 	if (field_off != val) {
7509 		verbose(env, "off %lld doesn't point to 'struct %s' that is at %d\n",
7510 			val, struct_name, field_off);
7511 		return -EINVAL;
7512 	}
7513 	if (map_desc->ptr) {
7514 		verifier_bug(env, "Two map pointers in a %s helper", struct_name);
7515 		return -EFAULT;
7516 	}
7517 	map_desc->uid = reg->map_uid;
7518 	map_desc->ptr = map;
7519 	return 0;
7520 }
7521 
7522 static int process_timer_func(struct bpf_verifier_env *env, struct bpf_reg_state *reg, argno_t argno,
7523 			      struct bpf_map_desc *map)
7524 {
7525 	if (IS_ENABLED(CONFIG_PREEMPT_RT)) {
7526 		verbose(env, "bpf_timer cannot be used for PREEMPT_RT.\n");
7527 		return -EOPNOTSUPP;
7528 	}
7529 	return check_map_field_pointer(env, reg, argno, BPF_TIMER, map);
7530 }
7531 
7532 static int process_kptr_func(struct bpf_verifier_env *env, int regno,
7533 			     struct bpf_call_arg_meta *meta)
7534 {
7535 	struct bpf_reg_state *reg = reg_state(env, regno);
7536 	struct btf_field *kptr_field;
7537 	struct bpf_map *map_ptr;
7538 	struct btf_record *rec;
7539 	u32 kptr_off;
7540 
7541 	if (type_is_ptr_alloc_obj(reg->type)) {
7542 		rec = reg_btf_record(reg);
7543 	} else { /* PTR_TO_MAP_VALUE */
7544 		map_ptr = reg->map_ptr;
7545 		if (!map_ptr->btf) {
7546 			verbose(env, "map '%s' has to have BTF in order to use bpf_kptr_xchg\n",
7547 				map_ptr->name);
7548 			return -EINVAL;
7549 		}
7550 		rec = map_ptr->record;
7551 		meta->map.ptr = map_ptr;
7552 	}
7553 
7554 	if (!tnum_is_const(reg->var_off)) {
7555 		verbose(env,
7556 			"R%d doesn't have constant offset. kptr has to be at the constant offset\n",
7557 			regno);
7558 		return -EINVAL;
7559 	}
7560 
7561 	if (!btf_record_has_field(rec, BPF_KPTR)) {
7562 		verbose(env, "R%d has no valid kptr\n", regno);
7563 		return -EINVAL;
7564 	}
7565 
7566 	kptr_off = reg->var_off.value;
7567 	kptr_field = btf_record_find(rec, kptr_off, BPF_KPTR);
7568 	if (!kptr_field) {
7569 		verbose(env, "off=%d doesn't point to kptr\n", kptr_off);
7570 		return -EACCES;
7571 	}
7572 	if (kptr_field->type != BPF_KPTR_REF && kptr_field->type != BPF_KPTR_PERCPU) {
7573 		verbose(env, "off=%d kptr isn't referenced kptr\n", kptr_off);
7574 		return -EACCES;
7575 	}
7576 	meta->kptr_field = kptr_field;
7577 	return 0;
7578 }
7579 
7580 static void bpf_diag_call_arg(struct bpf_verifier_env *env, u32 insn_idx, argno_t argno,
7581 			      const char *call_name, const char *reason, const char *suggestion);
7582 __printf(6, 7) static void bpf_diag_call_arg_fmt(struct bpf_verifier_env *env, u32 insn_idx,
7583 						 argno_t argno, const char *call_name,
7584 						 const char *suggestion, const char *fmt, ...);
7585 
7586 /*
7587  * Validate dynptr arguments for helper, kfunc and subprog.
7588  *
7589  * @dynptr is both input and output. It is populated when the argument is
7590  * tagged with MEM_UNINIT (i.e., the dynptr argument that will be constructed)
7591  * and consumed when the argument is expecting to be an initialized dynptr.
7592  * @parent_id is used to track the referenced parent object (e.g., file or skb in
7593  * qdisc program) when constructing a dynptr.
7594  *
7595  * There are two register types representing a bpf_dynptr, one is PTR_TO_STACK
7596  * which points to a stack slot, and the other is CONST_PTR_TO_DYNPTR.
7597  *
7598  * In both cases we deal with the first 8 bytes, but need to mark the next 8
7599  * bytes as STACK_DYNPTR in case of PTR_TO_STACK. In case of
7600  * CONST_PTR_TO_DYNPTR, we are guaranteed to get the beginning of the object.
7601  *
7602  * Mutability of bpf_dynptr is at two levels: the dynptr and the memory the
7603  * dynptr points to. At the first level, the verifier will make sure a
7604  * CONST_PTR_TO_DYNPTR cannot be reinitialized or destroyed. The mutability of
7605  * a dynptr's view (i.e., start and offset) is not tracked as there is not such
7606  * use case. The second level is tracked using the upper bit of bpf_dynptr->size
7607  * and checked dynamically during runtime.
7608  */
7609 static int process_dynptr_func(struct bpf_verifier_env *env, struct bpf_reg_state *reg,
7610 			       argno_t argno, int insn_idx, const char *call_name,
7611 			       enum bpf_arg_type arg_type,
7612 			       struct ref_obj_desc *ref_obj, struct bpf_dynptr_desc *dynptr)
7613 {
7614 	int spi, err = 0;
7615 
7616 	if (reg->type != PTR_TO_STACK && reg->type != CONST_PTR_TO_DYNPTR) {
7617 		verbose(env,
7618 			"%s expected pointer to stack or const struct bpf_dynptr\n",
7619 			reg_arg_name(env, argno));
7620 		bpf_diag_call_arg_fmt(
7621 			env, insn_idx, argno, call_name,
7622 			"Pass the address of a stack dynptr object, or use a const dynptr pointer returned by the verifier-supported path.",
7623 			"a dynptr argument must be a pointer to a dynptr stack slot or a verifier-provided const struct bpf_dynptr, but %s is %s",
7624 			reg_arg_name(env, argno), bpf_diag_reg_type_plain(env, reg->type));
7625 		return -EINVAL;
7626 	}
7627 
7628 	/*  MEM_UNINIT - Points to memory that is an appropriate candidate for
7629 	 *		 constructing a mutable bpf_dynptr object.
7630 	 *
7631 	 *		 Currently, this is only possible with PTR_TO_STACK
7632 	 *		 pointing to a region of at least 16 bytes which doesn't
7633 	 *		 contain an existing bpf_dynptr.
7634 	 *
7635 	 *  OBJ_RELEASE - Points to a initialized bpf_dynptr that will be
7636 	 *		  destroyed.
7637 	 *
7638 	 *  None       - Points to a initialized dynptr that cannot be
7639 	 *		 reinitialized or destroyed. However, the view of the
7640 	 *		 dynptr and the memory it points to may be mutated.
7641 	 */
7642 	if (arg_type & MEM_UNINIT) {
7643 		int i;
7644 
7645 		if (!is_dynptr_reg_valid_uninit(env, reg)) {
7646 			verbose(env, "Dynptr has to be an uninitialized dynptr\n");
7647 			bpf_diag_res(
7648 				env, insn_idx, "dynptr is already initialized",
7649 				"This kfunc constructs a dynptr and requires an uninitialized dynptr stack slot, but the selected slot already holds dynptr state.",
7650 				"Use a fresh stack dynptr slot, or release/destroy the existing dynptr before reusing the slot.");
7651 			return -EINVAL;
7652 		}
7653 
7654 		/* we write BPF_DW bits (8 bytes) at a time */
7655 		for (i = 0; i < BPF_DYNPTR_SIZE; i += 8) {
7656 			err = check_mem_access(env, insn_idx, reg, argno,
7657 					       i, BPF_DW, BPF_WRITE, -1, false, false);
7658 			if (err)
7659 				return err;
7660 		}
7661 
7662 		err = mark_stack_slots_dynptr(env, reg, arg_type, insn_idx, ref_obj, dynptr);
7663 	} else /* OBJ_RELEASE and None case from above */ {
7664 		/* For the reg->type == PTR_TO_STACK case, bpf_dynptr is never const */
7665 		if (reg->type == CONST_PTR_TO_DYNPTR && (arg_type & OBJ_RELEASE)) {
7666 			verbose(env, "CONST_PTR_TO_DYNPTR cannot be released\n");
7667 			bpf_diag_res(
7668 				env, insn_idx, "const dynptr release",
7669 				"This release operation was given a const dynptr. Const dynptr values are verifier-provided views and cannot be released by the program.",
7670 				"Release only mutable dynptrs that the program initialized or reserved.");
7671 			return -EINVAL;
7672 		}
7673 
7674 		if (!is_dynptr_reg_valid_init(env, reg)) {
7675 			verbose(env, "Expected an initialized dynptr as %s\n",
7676 				reg_arg_name(env, argno));
7677 			bpf_diag_res(
7678 				env, insn_idx, "uninitialized dynptr use",
7679 				"This operation requires an initialized dynptr, but the stack slot does not currently hold a valid dynptr on this path.",
7680 				"Initialize the dynptr on every path before this call, and avoid overwriting or releasing it before this use.");
7681 			return -EINVAL;
7682 		}
7683 
7684 		/* Fold modifiers (in this case, OBJ_RELEASE) when checking expected type */
7685 		if (!is_dynptr_type_expected(env, reg, arg_type & ~OBJ_RELEASE)) {
7686 			enum bpf_dynptr_type expected_type = arg_to_dynptr_type(arg_type);
7687 			enum bpf_dynptr_type actual_type = dynptr_reg_type(env, reg);
7688 
7689 			verbose(env, "Expected a dynptr of type %s as %s\n",
7690 				dynptr_type_str(expected_type), reg_arg_name(env, argno));
7691 			bpf_diag_call_arg_fmt(
7692 				env, insn_idx, argno, call_name,
7693 				"Use a dynptr constructor that matches this operation, or call an operation that accepts the dynptr's current type.",
7694 				"the dynptr is initialized with backing object type %s, but this operation expects dynptr type %s",
7695 				dynptr_type_str(actual_type), dynptr_type_str(expected_type));
7696 			return -EINVAL;
7697 		}
7698 
7699 		if (reg->type != CONST_PTR_TO_DYNPTR) {
7700 			struct bpf_func_state *state = bpf_func(env, reg);
7701 
7702 			spi = dynptr_get_spi(env, reg);
7703 			if (spi < 0)
7704 				return spi;
7705 
7706 			mark_stack_slots_scratched(env, spi, BPF_DYNPTR_NR_SLOTS);
7707 
7708 			reg = &state->stack[spi].spilled_ptr;
7709 		}
7710 
7711 		if (dynptr) {
7712 			dynptr->type = reg->dynptr.type;
7713 			dynptr->id = reg->id;
7714 			dynptr->parent_id = reg->parent_id;
7715 		}
7716 	}
7717 	return err;
7718 }
7719 
7720 static bool is_iter_kfunc(struct bpf_call_arg_meta *meta)
7721 {
7722 	return meta->kfunc_flags & (KF_ITER_NEW | KF_ITER_NEXT | KF_ITER_DESTROY);
7723 }
7724 
7725 static bool is_iter_new_kfunc(struct bpf_call_arg_meta *meta)
7726 {
7727 	return meta->kfunc_flags & KF_ITER_NEW;
7728 }
7729 
7730 static bool is_iter_destroy_kfunc(struct bpf_call_arg_meta *meta)
7731 {
7732 	return meta->kfunc_flags & KF_ITER_DESTROY;
7733 }
7734 
7735 static bool is_kfunc_arg_iter(struct bpf_call_arg_meta *meta, int arg_idx,
7736 			      const struct btf_param *arg)
7737 {
7738 	/* btf_check_iter_kfuncs() guarantees that first argument of any iter
7739 	 * kfunc is iter state pointer
7740 	 */
7741 	if (is_iter_kfunc(meta))
7742 		return arg_idx == 0;
7743 
7744 	/* iter passed as an argument to a generic kfunc */
7745 	return btf_param_match_suffix(meta->btf, arg, "__iter");
7746 }
7747 
7748 static int process_iter_arg(struct bpf_verifier_env *env, struct bpf_reg_state *reg, argno_t argno, int insn_idx,
7749 			    struct bpf_call_arg_meta *meta)
7750 {
7751 	struct bpf_func_state *state = bpf_func(env, reg);
7752 	const struct btf_type *t;
7753 	u32 arg_idx = arg_idx_from_argno(argno);
7754 	int spi, err, i, nr_slots, btf_id;
7755 
7756 	if (reg->type != PTR_TO_STACK) {
7757 		verbose(env, "%s expected pointer to an iterator on stack\n",
7758 			reg_arg_name(env, argno));
7759 		bpf_diag_call_arg_fmt(
7760 			env, insn_idx, argno, meta->func_name,
7761 			"Pass the address of a stack iterator object for iterator new, next, and destroy calls.",
7762 			"iterator state must live in verifier-tracked stack memory, but %s is %s",
7763 			reg_arg_name(env, argno), bpf_diag_reg_type_plain(env, reg->type));
7764 		return -EINVAL;
7765 	}
7766 
7767 	/* For iter_{new,next,destroy} functions, btf_check_iter_kfuncs()
7768 	 * ensures struct convention, so we wouldn't need to do any BTF
7769 	 * validation here. But given iter state can be passed as a parameter
7770 	 * to any kfunc, if arg has "__iter" suffix, we need to be a bit more
7771 	 * conservative here.
7772 	 */
7773 	btf_id = btf_check_iter_arg(meta->btf, meta->func_proto, arg_idx);
7774 	if (btf_id < 0) {
7775 		verbose(env, "expected valid iter pointer as %s\n",
7776 			reg_arg_name(env, argno));
7777 		bpf_diag_call_arg(
7778 			env, insn_idx, argno, meta->func_name,
7779 			"the kfunc expects a recognized iterator state pointer, but this argument does not match a valid iterator type",
7780 			"Pass the exact iterator state type expected by this kfunc.");
7781 		return -EINVAL;
7782 	}
7783 	t = btf_type_by_id(meta->btf, btf_id);
7784 	nr_slots = t->size / BPF_REG_SIZE;
7785 
7786 	if (is_iter_new_kfunc(meta)) {
7787 		/* bpf_iter_<type>_new() expects pointer to uninit iter state */
7788 		if (!is_iter_reg_valid_uninit(env, reg, nr_slots)) {
7789 			verbose(env, "expected uninitialized iter_%s as %s\n",
7790 				iter_type_str(meta->btf, btf_id), reg_arg_name(env, argno));
7791 			bpf_diag_res(
7792 				env, insn_idx, "iterator is already initialized",
7793 				"Iterator creation requires an uninitialized iterator stack object, but this stack range already contains iterator state.",
7794 				"Use a fresh iterator stack slot, or destroy the existing iterator before reusing the slot.");
7795 			return -EINVAL;
7796 		}
7797 
7798 		for (i = 0; i < nr_slots * 8; i += BPF_REG_SIZE) {
7799 			err = check_mem_access(env, insn_idx, reg, argno,
7800 					       i, BPF_DW, BPF_WRITE, -1, false, false);
7801 			if (err)
7802 				return err;
7803 		}
7804 
7805 		err = mark_stack_slots_iter(env, meta, reg, insn_idx, meta->btf, btf_id, nr_slots);
7806 		if (err)
7807 			return err;
7808 	} else {
7809 		/* iter_next() or iter_destroy(), as well as any kfunc
7810 		 * accepting iter argument, expect initialized iter state
7811 		 */
7812 		err = is_iter_reg_valid_init(env, reg, meta->btf, btf_id, nr_slots);
7813 		switch (err) {
7814 		case 0:
7815 			break;
7816 		case -EINVAL:
7817 			verbose(env, "expected an initialized iter_%s as %s\n",
7818 				iter_type_str(meta->btf, btf_id), reg_arg_name(env, argno));
7819 			bpf_diag_res(
7820 				env, insn_idx, "uninitialized iterator use",
7821 				"This iterator operation requires an initialized iterator state object, but the stack range does not contain a live iterator on this path.",
7822 				"Call the matching iterator new kfunc on every path before calling next or destroy, and do not destroy the iterator before this use.");
7823 			return err;
7824 		case -EPROTO:
7825 			verbose(env, "expected an RCU CS when using %s\n", meta->func_name);
7826 			bpf_diag_ctx_required(
7827 				env, insn_idx, meta->func_name, BPF_DIAG_CONTEXT_RCU,
7828 				"Wrap iterator use in bpf_rcu_read_lock() and bpf_rcu_read_unlock(), keeping all exit paths balanced.");
7829 			return err;
7830 		default:
7831 			return err;
7832 		}
7833 
7834 		spi = iter_get_spi(env, reg, nr_slots);
7835 		if (spi < 0)
7836 			return spi;
7837 
7838 		mark_stack_slots_scratched(env, spi, nr_slots);
7839 
7840 		/* remember meta->iter info for process_iter_next_call() */
7841 		meta->iter.spi = spi;
7842 		meta->iter.frameno = reg->frameno;
7843 		update_ref_obj(&meta->ref_obj, &state->stack[spi].spilled_ptr);
7844 
7845 		if (is_iter_destroy_kfunc(meta)) {
7846 			err = unmark_stack_slots_iter(env, reg, nr_slots);
7847 			if (err)
7848 				return err;
7849 		}
7850 	}
7851 
7852 	return 0;
7853 }
7854 
7855 /* Look for a previous loop entry at insn_idx: nearest parent state
7856  * stopped at insn_idx with callsites matching those in cur->frame.
7857  */
7858 static struct bpf_verifier_state *find_prev_entry(struct bpf_verifier_env *env,
7859 						  struct bpf_verifier_state *cur,
7860 						  int insn_idx)
7861 {
7862 	struct bpf_verifier_state_list *sl;
7863 	struct bpf_verifier_state *st;
7864 	struct list_head *pos, *head;
7865 
7866 	/* Explored states are pushed in stack order, most recent states come first */
7867 	head = bpf_explored_state(env, insn_idx);
7868 	list_for_each(pos, head) {
7869 		sl = container_of(pos, struct bpf_verifier_state_list, node);
7870 		/* If st->branches != 0 state is a part of current DFS verification path,
7871 		 * hence cur & st for a loop.
7872 		 */
7873 		st = &sl->state;
7874 		if (st->insn_idx == insn_idx && st->branches && same_callsites(st, cur) &&
7875 		    st->dfs_depth < cur->dfs_depth)
7876 			return st;
7877 	}
7878 
7879 	return NULL;
7880 }
7881 
7882 /*
7883  * Check if scalar registers are exact for the purpose of not widening.
7884  * More lenient than regs_exact()
7885  */
7886 static bool scalars_exact_for_widen(const struct bpf_reg_state *rold,
7887 				    const struct bpf_reg_state *rcur)
7888 {
7889 	return !memcmp(rold, rcur, offsetof(struct bpf_reg_state, id));
7890 }
7891 
7892 static void maybe_widen_reg(struct bpf_verifier_env *env,
7893 			    struct bpf_reg_state *rold, struct bpf_reg_state *rcur)
7894 {
7895 	if (rold->type != SCALAR_VALUE)
7896 		return;
7897 	if (rold->type != rcur->type)
7898 		return;
7899 	if (rold->precise || rcur->precise || scalars_exact_for_widen(rold, rcur))
7900 		return;
7901 	__mark_reg_unknown(env, rcur);
7902 }
7903 
7904 static int widen_imprecise_scalars(struct bpf_verifier_env *env,
7905 				   struct bpf_verifier_state *old,
7906 				   struct bpf_verifier_state *cur)
7907 {
7908 	struct bpf_func_state *fold, *fcur;
7909 	int i, fr, num_slots;
7910 
7911 	for (fr = old->curframe; fr >= 0; fr--) {
7912 		fold = old->frame[fr];
7913 		fcur = cur->frame[fr];
7914 
7915 		for (i = 0; i < MAX_BPF_REG; i++)
7916 			maybe_widen_reg(env,
7917 					&fold->regs[i],
7918 					&fcur->regs[i]);
7919 
7920 		num_slots = min(fold->allocated_stack / BPF_REG_SIZE,
7921 				fcur->allocated_stack / BPF_REG_SIZE);
7922 		for (i = 0; i < num_slots; i++) {
7923 			if (!bpf_is_spilled_reg(&fold->stack[i]) ||
7924 			    !bpf_is_spilled_reg(&fcur->stack[i]))
7925 				continue;
7926 
7927 			maybe_widen_reg(env,
7928 					&fold->stack[i].spilled_ptr,
7929 					&fcur->stack[i].spilled_ptr);
7930 		}
7931 	}
7932 	return 0;
7933 }
7934 
7935 static struct bpf_reg_state *get_iter_from_state(struct bpf_verifier_state *cur_st,
7936 						 struct bpf_call_arg_meta *meta)
7937 {
7938 	int iter_frameno = meta->iter.frameno;
7939 	int iter_spi = meta->iter.spi;
7940 
7941 	return &cur_st->frame[iter_frameno]->stack[iter_spi].spilled_ptr;
7942 }
7943 
7944 /* process_iter_next_call() is called when verifier gets to iterator's next
7945  * "method" (e.g., bpf_iter_num_next() for numbers iterator) call. We'll refer
7946  * to it as just "iter_next()" in comments below.
7947  *
7948  * BPF verifier relies on a crucial contract for any iter_next()
7949  * implementation: it should *eventually* return NULL, and once that happens
7950  * it should keep returning NULL. That is, once iterator exhausts elements to
7951  * iterate, it should never reset or spuriously return new elements.
7952  *
7953  * With the assumption of such contract, process_iter_next_call() simulates
7954  * a fork in the verifier state to validate loop logic correctness and safety
7955  * without having to simulate infinite amount of iterations.
7956  *
7957  * In current state, we first assume that iter_next() returned NULL and
7958  * iterator state is set to DRAINED (BPF_ITER_STATE_DRAINED). In such
7959  * conditions we should not form an infinite loop and should eventually reach
7960  * exit.
7961  *
7962  * Besides that, we also fork current state and enqueue it for later
7963  * verification. In a forked state we keep iterator state as ACTIVE
7964  * (BPF_ITER_STATE_ACTIVE) and assume non-NULL return from iter_next(). We
7965  * also bump iteration depth to prevent erroneous infinite loop detection
7966  * later on (see iter_active_depths_differ() comment for details). In this
7967  * state we assume that we'll eventually loop back to another iter_next()
7968  * calls (it could be in exactly same location or in some other instruction,
7969  * it doesn't matter, we don't make any unnecessary assumptions about this,
7970  * everything revolves around iterator state in a stack slot, not which
7971  * instruction is calling iter_next()). When that happens, we either will come
7972  * to iter_next() with equivalent state and can conclude that next iteration
7973  * will proceed in exactly the same way as we just verified, so it's safe to
7974  * assume that loop converges. If not, we'll go on another iteration
7975  * simulation with a different input state, until all possible starting states
7976  * are validated or we reach maximum number of instructions limit.
7977  *
7978  * This way, we will either exhaustively discover all possible input states
7979  * that iterator loop can start with and eventually will converge, or we'll
7980  * effectively regress into bounded loop simulation logic and either reach
7981  * maximum number of instructions if loop is not provably convergent, or there
7982  * is some statically known limit on number of iterations (e.g., if there is
7983  * an explicit `if n > 100 then break;` statement somewhere in the loop).
7984  *
7985  * Iteration convergence logic in is_state_visited() relies on exact
7986  * states comparison, which ignores read and precision marks.
7987  * This is necessary because read and precision marks are not finalized
7988  * while in the loop. Exact comparison might preclude convergence for
7989  * simple programs like below:
7990  *
7991  *     i = 0;
7992  *     while(iter_next(&it))
7993  *       i++;
7994  *
7995  * At each iteration step i++ would produce a new distinct state and
7996  * eventually instruction processing limit would be reached.
7997  *
7998  * To avoid such behavior speculatively forget (widen) range for
7999  * imprecise scalar registers, if those registers were not precise at the
8000  * end of the previous iteration and do not match exactly.
8001  *
8002  * This is a conservative heuristic that allows to verify wide range of programs,
8003  * however it precludes verification of programs that conjure an
8004  * imprecise value on the first loop iteration and use it as precise on a second.
8005  * For example, the following safe program would fail to verify:
8006  *
8007  *     struct bpf_num_iter it;
8008  *     int arr[10];
8009  *     int i = 0, a = 0;
8010  *     bpf_iter_num_new(&it, 0, 10);
8011  *     while (bpf_iter_num_next(&it)) {
8012  *       if (a == 0) {
8013  *         a = 1;
8014  *         i = 7; // Because i changed verifier would forget
8015  *                // it's range on second loop entry.
8016  *       } else {
8017  *         arr[i] = 42; // This would fail to verify.
8018  *       }
8019  *     }
8020  *     bpf_iter_num_destroy(&it);
8021  */
8022 static int process_iter_next_call(struct bpf_verifier_env *env, int insn_idx,
8023 				  struct bpf_call_arg_meta *meta)
8024 {
8025 	struct bpf_verifier_state *cur_st = env->cur_state, *queued_st, *prev_st;
8026 	struct bpf_func_state *cur_fr = cur_st->frame[cur_st->curframe], *queued_fr;
8027 	struct bpf_reg_state *cur_iter, *queued_iter;
8028 
8029 	BTF_TYPE_EMIT(struct bpf_iter);
8030 
8031 	cur_iter = get_iter_from_state(cur_st, meta);
8032 
8033 	if (cur_iter->iter.state != BPF_ITER_STATE_ACTIVE &&
8034 	    cur_iter->iter.state != BPF_ITER_STATE_DRAINED) {
8035 		verifier_bug(env, "unexpected iterator state %d (%s)",
8036 			     cur_iter->iter.state, iter_state_str(cur_iter->iter.state));
8037 		return -EFAULT;
8038 	}
8039 
8040 	if (cur_iter->iter.state == BPF_ITER_STATE_ACTIVE) {
8041 		/* Because iter_next() call is a checkpoint is_state_visitied()
8042 		 * should guarantee parent state with same call sites and insn_idx.
8043 		 */
8044 		if (!cur_st->parent || cur_st->parent->insn_idx != insn_idx ||
8045 		    !same_callsites(cur_st->parent, cur_st)) {
8046 			verifier_bug(env, "bad parent state for iter next call");
8047 			return -EFAULT;
8048 		}
8049 		/* Note cur_st->parent in the call below, it is necessary to skip
8050 		 * checkpoint created for cur_st by is_state_visited()
8051 		 * right at this instruction.
8052 		 */
8053 		prev_st = find_prev_entry(env, cur_st->parent, insn_idx);
8054 		/* branch out active iter state */
8055 		queued_st = push_stack(env, insn_idx + 1, insn_idx, false);
8056 		if (IS_ERR(queued_st))
8057 			return PTR_ERR(queued_st);
8058 
8059 		queued_iter = get_iter_from_state(queued_st, meta);
8060 		queued_iter->iter.state = BPF_ITER_STATE_ACTIVE;
8061 		queued_iter->iter.depth++;
8062 		if (prev_st)
8063 			widen_imprecise_scalars(env, prev_st, queued_st);
8064 
8065 		queued_fr = queued_st->frame[queued_st->curframe];
8066 		mark_ptr_not_null_reg(&queued_fr->regs[BPF_REG_0]);
8067 	}
8068 
8069 	/* switch to DRAINED state, but keep the depth unchanged */
8070 	/* mark current iter state as drained and assume returned NULL */
8071 	cur_iter->iter.state = BPF_ITER_STATE_DRAINED;
8072 	__mark_reg_const_zero(env, &cur_fr->regs[BPF_REG_0]);
8073 
8074 	return 0;
8075 }
8076 
8077 static bool arg_type_is_mem_size(enum bpf_arg_type type)
8078 {
8079 	return type == ARG_MEM_SIZE || type == ARG_MEM_SIZE_OR_ZERO;
8080 }
8081 
8082 static bool arg_type_is_raw_mem(enum bpf_arg_type type)
8083 {
8084 	/*
8085 	 * A map value output buffer (e.g. bpf_map_pop_elem) is also a raw
8086 	 * (uninitialized) memory argument, and like ARG_PTR_TO_MEM it may be
8087 	 * passed as a PTR_TO_STACK that reaches check_stack_range_initialized().
8088 	 */
8089 	return (base_type(type) == ARG_PTR_TO_MEM ||
8090 		base_type(type) == ARG_PTR_TO_MAP_VALUE) &&
8091 	       type & MEM_UNINIT;
8092 }
8093 
8094 static bool arg_type_is_release(enum bpf_arg_type type)
8095 {
8096 	return type & OBJ_RELEASE;
8097 }
8098 
8099 static bool arg_type_is_dynptr(enum bpf_arg_type type)
8100 {
8101 	return base_type(type) == ARG_PTR_TO_DYNPTR;
8102 }
8103 
8104 static int resolve_map_arg_type(struct bpf_verifier_env *env,
8105 				 const struct bpf_call_arg_meta *meta,
8106 				 enum bpf_arg_type *arg_type)
8107 {
8108 	if (!meta->map.ptr) {
8109 		/* kernel subsystem misconfigured verifier */
8110 		verifier_bug(env, "invalid map_ptr to access map->type");
8111 		return -EFAULT;
8112 	}
8113 
8114 	switch (meta->map.ptr->map_type) {
8115 	case BPF_MAP_TYPE_SOCKMAP:
8116 	case BPF_MAP_TYPE_SOCKHASH:
8117 		if (*arg_type == ARG_PTR_TO_MAP_VALUE) {
8118 			*arg_type = ARG_PTR_TO_BTF_ID_SOCK_COMMON;
8119 		} else {
8120 			verbose(env, "invalid arg_type for sockmap/sockhash\n");
8121 			return -EINVAL;
8122 		}
8123 		break;
8124 	case BPF_MAP_TYPE_BLOOM_FILTER:
8125 		if (meta->func_id == BPF_FUNC_map_peek_elem)
8126 			*arg_type = ARG_PTR_TO_MAP_VALUE;
8127 		break;
8128 	default:
8129 		break;
8130 	}
8131 	return 0;
8132 }
8133 
8134 struct bpf_reg_types {
8135 	const enum bpf_reg_type types[10];
8136 	u32 *btf_id;
8137 };
8138 
8139 static const struct bpf_reg_types sock_types = {
8140 	.types = {
8141 		PTR_TO_SOCK_COMMON,
8142 		PTR_TO_SOCKET,
8143 		PTR_TO_TCP_SOCK,
8144 		PTR_TO_XDP_SOCK,
8145 	},
8146 };
8147 
8148 #ifdef CONFIG_NET
8149 static const struct bpf_reg_types btf_id_sock_common_types = {
8150 	.types = {
8151 		PTR_TO_SOCK_COMMON,
8152 		PTR_TO_SOCKET,
8153 		PTR_TO_TCP_SOCK,
8154 		PTR_TO_XDP_SOCK,
8155 		PTR_TO_BTF_ID,
8156 		PTR_TO_BTF_ID | PTR_TRUSTED,
8157 	},
8158 	.btf_id = &btf_sock_ids[BTF_SOCK_TYPE_SOCK_COMMON],
8159 };
8160 #endif
8161 
8162 static const struct bpf_reg_types mem_types = {
8163 	.types = {
8164 		PTR_TO_STACK,
8165 		PTR_TO_PACKET,
8166 		PTR_TO_PACKET_META,
8167 		PTR_TO_MAP_KEY,
8168 		PTR_TO_MAP_VALUE,
8169 		PTR_TO_MEM,
8170 		PTR_TO_MEM | MEM_RINGBUF,
8171 		PTR_TO_BUF,
8172 		PTR_TO_BTF_ID | PTR_TRUSTED,
8173 		PTR_TO_CTX,
8174 	},
8175 };
8176 
8177 static const struct bpf_reg_types spin_lock_types = {
8178 	.types = {
8179 		PTR_TO_MAP_VALUE,
8180 		PTR_TO_BTF_ID | MEM_ALLOC,
8181 	}
8182 };
8183 
8184 static const struct bpf_reg_types fullsock_types = { .types = { PTR_TO_SOCKET } };
8185 static const struct bpf_reg_types scalar_types = { .types = { SCALAR_VALUE } };
8186 static const struct bpf_reg_types context_types = { .types = { PTR_TO_CTX } };
8187 static const struct bpf_reg_types ringbuf_mem_types = { .types = { PTR_TO_MEM | MEM_RINGBUF } };
8188 static const struct bpf_reg_types const_map_ptr_types = { .types = { CONST_PTR_TO_MAP } };
8189 static const struct bpf_reg_types btf_ptr_types = {
8190 	.types = {
8191 		PTR_TO_BTF_ID,
8192 		PTR_TO_BTF_ID | PTR_TRUSTED,
8193 		PTR_TO_BTF_ID | MEM_RCU,
8194 	},
8195 };
8196 static const struct bpf_reg_types percpu_btf_ptr_types = {
8197 	.types = {
8198 		PTR_TO_BTF_ID | MEM_PERCPU,
8199 		PTR_TO_BTF_ID | MEM_PERCPU | MEM_RCU,
8200 		PTR_TO_BTF_ID | MEM_PERCPU | PTR_TRUSTED,
8201 	}
8202 };
8203 static const struct bpf_reg_types func_ptr_types = { .types = { PTR_TO_FUNC } };
8204 static const struct bpf_reg_types stack_ptr_types = { .types = { PTR_TO_STACK } };
8205 static const struct bpf_reg_types const_str_ptr_types = { .types = { PTR_TO_MAP_VALUE } };
8206 static const struct bpf_reg_types timer_types = { .types = { PTR_TO_MAP_VALUE } };
8207 static const struct bpf_reg_types kptr_xchg_dest_types = {
8208 	.types = {
8209 		PTR_TO_MAP_VALUE,
8210 		PTR_TO_BTF_ID | MEM_ALLOC,
8211 		PTR_TO_BTF_ID | MEM_ALLOC | NON_OWN_REF,
8212 		PTR_TO_BTF_ID | MEM_ALLOC | NON_OWN_REF | MEM_RCU,
8213 	}
8214 };
8215 static const struct bpf_reg_types dynptr_types = {
8216 	.types = {
8217 		PTR_TO_STACK,
8218 		CONST_PTR_TO_DYNPTR,
8219 	}
8220 };
8221 
8222 static const struct bpf_reg_types *compatible_reg_types[__BPF_ARG_TYPE_MAX] = {
8223 	[ARG_PTR_TO_MAP_KEY]		= &mem_types,
8224 	[ARG_PTR_TO_MAP_VALUE]		= &mem_types,
8225 	[ARG_MEM_SIZE]			= &scalar_types,
8226 	[ARG_MEM_SIZE_OR_ZERO]		= &scalar_types,
8227 	[ARG_CONST_ALLOC_SIZE_OR_ZERO]	= &scalar_types,
8228 	[ARG_SCALAR]			= &scalar_types,
8229 	[ARG_CONST_MAP_PTR]		= &const_map_ptr_types,
8230 	[ARG_PTR_TO_CTX]		= &context_types,
8231 	[ARG_PTR_TO_SOCK_COMMON]	= &sock_types,
8232 #ifdef CONFIG_NET
8233 	[ARG_PTR_TO_BTF_ID_SOCK_COMMON]	= &btf_id_sock_common_types,
8234 #endif
8235 	[ARG_PTR_TO_SOCKET]		= &fullsock_types,
8236 	[ARG_PTR_TO_BTF_ID]		= &btf_ptr_types,
8237 	[ARG_PTR_TO_SPIN_LOCK]		= &spin_lock_types,
8238 	[ARG_PTR_TO_MEM]		= &mem_types,
8239 	[ARG_PTR_TO_RINGBUF_MEM]	= &ringbuf_mem_types,
8240 	[ARG_PTR_TO_PERCPU_BTF_ID]	= &percpu_btf_ptr_types,
8241 	[ARG_PTR_TO_FUNC]		= &func_ptr_types,
8242 	[ARG_PTR_TO_STACK]		= &stack_ptr_types,
8243 	[ARG_PTR_TO_CONST_STR]		= &const_str_ptr_types,
8244 	[ARG_PTR_TO_TIMER]		= &timer_types,
8245 	[ARG_KPTR_XCHG_DEST]		= &kptr_xchg_dest_types,
8246 	[ARG_PTR_TO_DYNPTR]		= &dynptr_types,
8247 };
8248 
8249 static void bpf_diag_call_arg(struct bpf_verifier_env *env, u32 insn_idx, argno_t argno,
8250 			      const char *call_name, const char *reason,
8251 			      const char *suggestion)
8252 {
8253 	int arg = arg_from_argno(argno);
8254 	int regno = reg_from_argno(argno);
8255 	int stack_slot = -1;
8256 
8257 	if (arg < 0 && regno >= BPF_REG_1 && regno <= BPF_REG_5)
8258 		arg = regno;
8259 	if (arg > MAX_BPF_FUNC_REG_ARGS)
8260 		stack_slot = arg - MAX_BPF_FUNC_REG_ARGS - 1;
8261 
8262 	bpf_diag_call_type(env, insn_idx, arg, regno, stack_slot,
8263 			   call_name && *call_name ? call_name : "call",
8264 			   reg_arg_name(env, argno), reason, suggestion);
8265 }
8266 
8267 static const char *bpf_diag_arg_name(struct bpf_verifier_env *env, argno_t argno)
8268 {
8269 	return bpf_diag_fmt(env, "%s", reg_arg_name(env, argno));
8270 }
8271 
8272 __printf(6, 7) static void bpf_diag_call_arg_fmt(struct bpf_verifier_env *env, u32 insn_idx,
8273 						 argno_t argno, const char *call_name,
8274 						 const char *suggestion, const char *fmt, ...)
8275 {
8276 	const char *reason;
8277 	va_list args;
8278 
8279 	va_start(args, fmt);
8280 	reason = bpf_diag_vfmt(env, fmt, args);
8281 	va_end(args);
8282 
8283 	bpf_diag_call_arg(env, insn_idx, argno, call_name, reason, suggestion);
8284 }
8285 
8286 static const char *bpf_diag_expected_reg_types(struct bpf_verifier_env *env,
8287 					       const enum bpf_reg_type *types, int count)
8288 {
8289 	size_t len = 0, size = 1;
8290 	char *buf;
8291 	int i;
8292 
8293 	for (i = 0; i < count; i++)
8294 		size += strlen(reg_type_str(env, types[i])) + (i ? 2 : 0);
8295 
8296 	buf = bpf_diag_fmt_buf(env, size);
8297 	if (!buf)
8298 		return "";
8299 
8300 	for (i = 0; i < count; i++)
8301 		len += scnprintf(buf + len, size - len, "%s%s", i ? ", " : "",
8302 				 reg_type_str(env, types[i]));
8303 	return buf;
8304 }
8305 
8306 static int check_reg_type(struct bpf_verifier_env *env, struct bpf_reg_state *reg, argno_t argno,
8307 			  enum bpf_arg_type arg_type, const u32 *arg_btf_id,
8308 			  struct bpf_call_arg_meta *meta, const char *call_name)
8309 {
8310 	enum bpf_reg_type expected, type = reg->type;
8311 	const struct bpf_reg_types *compatible;
8312 	const char *actual, *accepted;
8313 	int i, j, err;
8314 
8315 	compatible = compatible_reg_types[base_type(arg_type)];
8316 	if (!compatible) {
8317 		verifier_bug(env, "unsupported arg type %d", arg_type);
8318 		return -EFAULT;
8319 	}
8320 
8321 	/* ARG_PTR_TO_MEM + RDONLY is compatible with PTR_TO_MEM and PTR_TO_MEM + RDONLY,
8322 	 * but ARG_PTR_TO_MEM is compatible only with PTR_TO_MEM and NOT with PTR_TO_MEM + RDONLY
8323 	 *
8324 	 * Same for MAYBE_NULL:
8325 	 *
8326 	 * ARG_PTR_TO_MEM + MAYBE_NULL is compatible with PTR_TO_MEM and PTR_TO_MEM + MAYBE_NULL,
8327 	 * but ARG_PTR_TO_MEM is compatible only with PTR_TO_MEM but NOT with PTR_TO_MEM + MAYBE_NULL
8328 	 *
8329 	 * ARG_PTR_TO_MEM is compatible with PTR_TO_MEM that is tagged with a dynptr type.
8330 	 *
8331 	 * Therefore we fold these flags depending on the arg_type before comparison.
8332 	 */
8333 	if (arg_type & MEM_RDONLY)
8334 		type &= ~MEM_RDONLY;
8335 	if (arg_type & PTR_MAYBE_NULL)
8336 		type &= ~PTR_MAYBE_NULL;
8337 	if (base_type(arg_type) == ARG_PTR_TO_MEM)
8338 		type &= ~DYNPTR_TYPE_FLAG_MASK;
8339 
8340 	/* Local kptr types are allowed as the source argument of bpf_kptr_xchg */
8341 	if (meta->func_id == BPF_FUNC_kptr_xchg && type_is_alloc(type) && reg_from_argno(argno) == BPF_REG_2) {
8342 		type &= ~MEM_ALLOC;
8343 		type &= ~MEM_PERCPU;
8344 	}
8345 
8346 	for (i = 0; i < ARRAY_SIZE(compatible->types); i++) {
8347 		expected = compatible->types[i];
8348 		if (expected == NOT_INIT)
8349 			break;
8350 
8351 		if (type == expected)
8352 			goto found;
8353 	}
8354 
8355 	verbose(env, "%s type=%s expected=", reg_arg_name(env, argno), reg_type_str(env, reg->type));
8356 	for (j = 0; j + 1 < i; j++)
8357 		verbose(env, "%s, ", reg_type_str(env, compatible->types[j]));
8358 	verbose(env, "%s\n", reg_type_str(env, compatible->types[j]));
8359 	actual = bpf_diag_fmt(env, "%s", reg_type_str(env, reg->type));
8360 	accepted = bpf_diag_expected_reg_types(env, compatible->types, i);
8361 	bpf_diag_call_arg_fmt(env, env->insn_idx, argno, call_name,
8362 			      "Pass a value with one of the accepted pointer or scalar types for this call.",
8363 			      "it has type %s, but this argument accepts %s",
8364 			      actual, accepted);
8365 	return -EACCES;
8366 
8367 found:
8368 	if (base_type(reg->type) != PTR_TO_BTF_ID)
8369 		return 0;
8370 
8371 	if (compatible == &mem_types) {
8372 		if (!(arg_type & MEM_RDONLY)) {
8373 			verbose(env,
8374 				"%s() may write into memory pointed by %s type=%s\n",
8375 				func_id_name(meta->func_id),
8376 				reg_arg_name(env, argno), reg_type_str(env, reg->type));
8377 			return -EACCES;
8378 		}
8379 		return 0;
8380 	}
8381 
8382 	switch ((int)reg->type) {
8383 	case PTR_TO_BTF_ID:
8384 	case PTR_TO_BTF_ID | PTR_TRUSTED:
8385 	case PTR_TO_BTF_ID | PTR_TRUSTED | PTR_MAYBE_NULL:
8386 	case PTR_TO_BTF_ID | MEM_RCU:
8387 	case PTR_TO_BTF_ID | PTR_MAYBE_NULL:
8388 	case PTR_TO_BTF_ID | PTR_MAYBE_NULL | MEM_RCU:
8389 	{
8390 		/* For bpf_sk_release, it needs to match against first member
8391 		 * 'struct sock_common', hence make an exception for it. This
8392 		 * allows bpf_sk_release to work for multiple socket types.
8393 		 */
8394 		bool strict_type_match = arg_type_is_release(arg_type) &&
8395 					 meta->func_id != BPF_FUNC_sk_release;
8396 
8397 		if (type_may_be_null(reg->type) &&
8398 		    (!type_may_be_null(arg_type) || arg_type_is_release(arg_type))) {
8399 			verbose(env, "Possibly NULL pointer passed to helper %s\n",
8400 				reg_arg_name(env, argno));
8401 			bpf_diag_call_arg(
8402 				env, env->insn_idx, argno, call_name,
8403 				"the pointer may be NULL, but this call requires a non-NULL pointer",
8404 				"Add a NULL check and make the call only on the non-NULL path.");
8405 			return -EACCES;
8406 		}
8407 
8408 		if (!arg_btf_id) {
8409 			if (!compatible->btf_id) {
8410 				verifier_bug(env, "missing arg compatible BTF ID");
8411 				return -EFAULT;
8412 			}
8413 			arg_btf_id = compatible->btf_id;
8414 		}
8415 
8416 		if (meta->func_id == BPF_FUNC_kptr_xchg) {
8417 			if (map_kptr_match_type(env, meta->kptr_field, reg, reg_from_argno(argno)))
8418 				return -EACCES;
8419 		} else {
8420 			if (arg_btf_id == BPF_PTR_POISON) {
8421 				verbose(env, "verifier internal error:");
8422 				verbose(env, "%s has non-overwritten BPF_PTR_POISON type\n",
8423 					reg_arg_name(env, argno));
8424 				return -EACCES;
8425 			}
8426 
8427 			err = __check_ptr_off_reg(env, reg, argno, true);
8428 			if (err)
8429 				return err;
8430 
8431 			if (!btf_struct_ids_match(&env->log, reg->btf, reg->btf_id,
8432 						  reg->var_off.value, btf_vmlinux, *arg_btf_id,
8433 						  strict_type_match, !type_is_alloc(reg->type))) {
8434 				verbose(env, "%s is of type %s but %s is expected\n",
8435 					reg_arg_name(env, argno),
8436 					btf_type_name(reg->btf, reg->btf_id),
8437 					btf_type_name(btf_vmlinux, *arg_btf_id));
8438 				return -EACCES;
8439 			}
8440 		}
8441 		break;
8442 	}
8443 	case PTR_TO_BTF_ID | MEM_ALLOC:
8444 	case PTR_TO_BTF_ID | MEM_PERCPU | MEM_ALLOC:
8445 	case PTR_TO_BTF_ID | MEM_ALLOC | NON_OWN_REF:
8446 	case PTR_TO_BTF_ID | MEM_ALLOC | NON_OWN_REF | MEM_RCU:
8447 		if (meta->func_id != BPF_FUNC_spin_lock && meta->func_id != BPF_FUNC_spin_unlock &&
8448 		    meta->func_id != BPF_FUNC_kptr_xchg) {
8449 			verifier_bug(env, "unimplemented handling of MEM_ALLOC");
8450 			return -EFAULT;
8451 		}
8452 		/* Check if local kptr in src arg matches kptr in dst arg */
8453 		if (meta->func_id == BPF_FUNC_kptr_xchg) {
8454 			int regno = reg_from_argno(argno);
8455 
8456 			if (regno == BPF_REG_2 &&
8457 			    map_kptr_match_type(env, meta->kptr_field, reg, regno))
8458 				return -EACCES;
8459 		}
8460 		break;
8461 	case PTR_TO_BTF_ID | MEM_PERCPU:
8462 	case PTR_TO_BTF_ID | MEM_PERCPU | MEM_RCU:
8463 	case PTR_TO_BTF_ID | MEM_PERCPU | PTR_TRUSTED:
8464 		/* Handled by helper specific checks */
8465 		break;
8466 	default:
8467 		verifier_bug(env, "invalid PTR_TO_BTF_ID register for type match");
8468 		return -EFAULT;
8469 	}
8470 	return 0;
8471 }
8472 
8473 static struct btf_field *
8474 reg_find_field_offset(const struct bpf_reg_state *reg, s32 off, u32 fields)
8475 {
8476 	struct btf_field *field;
8477 	struct btf_record *rec;
8478 
8479 	rec = reg_btf_record(reg);
8480 	if (!rec)
8481 		return NULL;
8482 
8483 	field = btf_record_find(rec, off, fields);
8484 	if (!field)
8485 		return NULL;
8486 
8487 	return field;
8488 }
8489 
8490 static int __check_func_arg_reg_off(struct bpf_verifier_env *env,
8491 				    const struct bpf_reg_state *reg, argno_t argno,
8492 				    enum bpf_arg_type arg_type,
8493 				    bool btf_id_fixed_off_ok)
8494 {
8495 	u32 type = reg->type;
8496 
8497 	/* When referenced register is passed to release function, its fixed
8498 	 * offset must be 0.
8499 	 *
8500 	 * We will check arg_type_is_release reg has id when storing
8501 	 * meta->release_regno.
8502 	 */
8503 	if (arg_type_is_release(arg_type)) {
8504 		/* ARG_PTR_TO_DYNPTR with OBJ_RELEASE is a bit special, as it
8505 		 * may not directly point to the object being released, but to
8506 		 * dynptr pointing to such object, which might be at some offset
8507 		 * on the stack. In that case, we simply to fallback to the
8508 		 * default handling.
8509 		 */
8510 		if (arg_type_is_dynptr(arg_type) && type == PTR_TO_STACK)
8511 			return 0;
8512 
8513 		/* Doing check_ptr_off_reg check for the offset will catch this
8514 		 * because fixed_off_ok is false, but checking here allows us
8515 		 * to give the user a better error message.
8516 		 */
8517 		if (!tnum_is_const(reg->var_off) || reg->var_off.value != 0) {
8518 			verbose(env, "%s must have zero offset when passed to release func or trusted arg to kfunc\n",
8519 				reg_arg_name(env, argno));
8520 			return -EINVAL;
8521 		}
8522 	}
8523 
8524 	switch (type) {
8525 	/* Pointer types where both fixed and variable offset is explicitly allowed: */
8526 	case PTR_TO_STACK:
8527 	case PTR_TO_PACKET:
8528 	case PTR_TO_PACKET_META:
8529 	case PTR_TO_MAP_KEY:
8530 	case PTR_TO_MAP_VALUE:
8531 	case PTR_TO_MEM:
8532 	case PTR_TO_MEM | MEM_RDONLY:
8533 	case PTR_TO_MEM | MEM_RINGBUF:
8534 	case PTR_TO_BUF:
8535 	case PTR_TO_BUF | MEM_RDONLY:
8536 	case PTR_TO_ARENA:
8537 	case SCALAR_VALUE:
8538 		return 0;
8539 	/* All the rest must be rejected, except PTR_TO_BTF_ID which allows
8540 	 * fixed offset.
8541 	 */
8542 	case PTR_TO_BTF_ID:
8543 	case PTR_TO_BTF_ID | MEM_ALLOC:
8544 	case PTR_TO_BTF_ID | PTR_TRUSTED:
8545 	case PTR_TO_BTF_ID | MEM_RCU:
8546 	case PTR_TO_BTF_ID | MEM_ALLOC | NON_OWN_REF:
8547 	case PTR_TO_BTF_ID | MEM_ALLOC | NON_OWN_REF | MEM_RCU:
8548 		/* When referenced PTR_TO_BTF_ID is passed to release function,
8549 		 * its fixed offset must be 0. In the other cases, fixed offset
8550 		 * can be non-zero unless the caller requires otherwise.
8551 		 * var_off always must be 0 for PTR_TO_BTF_ID, hence we still
8552 		 * need to do checks instead of returning.
8553 		 */
8554 		return __check_ptr_off_reg(env, reg, argno, btf_id_fixed_off_ok);
8555 	case PTR_TO_CTX:
8556 		/*
8557 		 * Allow fixed and variable offsets for syscall context, but
8558 		 * only when the argument is passed as memory, not ctx,
8559 		 * otherwise we may get modified ctx in tail called programs and
8560 		 * global subprogs (that may act as extension prog hooks).
8561 		 */
8562 		if (arg_type != ARG_PTR_TO_CTX && is_var_ctx_off_allowed(env->prog))
8563 			return 0;
8564 		fallthrough;
8565 	default:
8566 		return __check_ptr_off_reg(env, reg, argno, false);
8567 	}
8568 }
8569 
8570 static int check_func_arg_reg_off(struct bpf_verifier_env *env,
8571 				  const struct bpf_reg_state *reg, argno_t argno,
8572 				  enum bpf_arg_type arg_type)
8573 {
8574 	return __check_func_arg_reg_off(env, reg, argno, arg_type, true);
8575 }
8576 
8577 static int check_arg_const_str(struct bpf_verifier_env *env,
8578 			       struct bpf_reg_state *reg, argno_t argno)
8579 {
8580 	struct bpf_map *map = reg->map_ptr;
8581 	int err;
8582 	int map_off;
8583 	u64 map_addr;
8584 	char *str_ptr;
8585 
8586 	if (reg->type != PTR_TO_MAP_VALUE)
8587 		return -EINVAL;
8588 
8589 	if (map->map_type == BPF_MAP_TYPE_INSN_ARRAY) {
8590 		verbose(env, "%s points to insn_array map which cannot be used as const string\n",
8591 			reg_arg_name(env, argno));
8592 		return -EACCES;
8593 	}
8594 
8595 	if (map->map_type == BPF_MAP_TYPE_PERCPU_ARRAY) {
8596 		verbose(env, "%s points to percpu_array map which cannot be used as const string\n",
8597 			reg_arg_name(env, argno));
8598 		return -EACCES;
8599 	}
8600 
8601 	if (!bpf_map_is_rdonly(map)) {
8602 		verbose(env, "%s does not point to a readonly map'\n", reg_arg_name(env, argno));
8603 		return -EACCES;
8604 	}
8605 
8606 	if (!tnum_is_const(reg->var_off)) {
8607 		verbose(env, "%s is not a constant address'\n", reg_arg_name(env, argno));
8608 		return -EACCES;
8609 	}
8610 
8611 	if (!map->ops->map_direct_value_addr) {
8612 		verbose(env, "no direct value access support for this map type\n");
8613 		return -EACCES;
8614 	}
8615 
8616 	err = check_map_access(env, reg, argno, 0,
8617 			       map->value_size - reg->var_off.value, false,
8618 			       ACCESS_HELPER);
8619 	if (err)
8620 		return err;
8621 
8622 	map_off = reg->var_off.value;
8623 	err = map->ops->map_direct_value_addr(map, &map_addr, map_off);
8624 	if (err) {
8625 		verbose(env, "direct value access on string failed\n");
8626 		return err;
8627 	}
8628 
8629 	str_ptr = (char *)(long)(map_addr);
8630 	if (!strnchr(str_ptr + map_off, map->value_size - map_off, 0)) {
8631 		verbose(env, "string is not zero-terminated\n");
8632 		return -EINVAL;
8633 	}
8634 	return 0;
8635 }
8636 
8637 /* Returns constant key value in `value` if possible, else negative error */
8638 static int get_constant_map_key(struct bpf_verifier_env *env,
8639 				struct bpf_reg_state *key,
8640 				u32 key_size,
8641 				s64 *value)
8642 {
8643 	struct bpf_func_state *state = bpf_func(env, key);
8644 	struct bpf_reg_state *reg;
8645 	int slot, spi, off;
8646 	int spill_size = 0;
8647 	int zero_size = 0;
8648 	int stack_off;
8649 	int i, err;
8650 	u8 *stype;
8651 
8652 	if (!env->bpf_capable)
8653 		return -EOPNOTSUPP;
8654 	if (key->type != PTR_TO_STACK)
8655 		return -EOPNOTSUPP;
8656 	if (!tnum_is_const(key->var_off))
8657 		return -EOPNOTSUPP;
8658 
8659 	stack_off = key->var_off.value;
8660 	slot = -stack_off - 1;
8661 	spi = slot / BPF_REG_SIZE;
8662 	off = slot % BPF_REG_SIZE;
8663 	stype = state->stack[spi].slot_type;
8664 
8665 	/* First handle precisely tracked STACK_ZERO */
8666 	for (i = off; i >= 0 && stype[i] == STACK_ZERO; i--)
8667 		zero_size++;
8668 	if (zero_size >= key_size) {
8669 		*value = 0;
8670 		return 0;
8671 	}
8672 
8673 	/* Check that stack contains a scalar spill of expected size */
8674 	if (!bpf_is_spilled_scalar_reg(&state->stack[spi]))
8675 		return -EOPNOTSUPP;
8676 	for (i = off; i >= 0 && stype[i] == STACK_SPILL; i--)
8677 		spill_size++;
8678 	if (spill_size != key_size)
8679 		return -EOPNOTSUPP;
8680 
8681 	reg = &state->stack[spi].spilled_ptr;
8682 	if (!tnum_is_const(reg->var_off))
8683 		/* Stack value not statically known */
8684 		return -EOPNOTSUPP;
8685 
8686 	/* We are relying on a constant value. So mark as precise
8687 	 * to prevent pruning on it.
8688 	 */
8689 	bpf_bt_set_frame_slot(&env->bt, key->frameno, spi);
8690 	err = mark_chain_precision_batch(env, env->cur_state);
8691 	if (err < 0)
8692 		return err;
8693 
8694 	*value = reg->var_off.value;
8695 	return 0;
8696 }
8697 
8698 static bool can_elide_value_nullness(const struct bpf_map *map);
8699 
8700 static int process_map_ptr_arg(struct bpf_verifier_env *env, struct bpf_reg_state *reg,
8701 			       argno_t argno, struct bpf_call_arg_meta *meta)
8702 {
8703 	/* Use map_uid (which is unique id of inner map) to reject:
8704 	 * inner_map1 = bpf_map_lookup_elem(outer_map, key1)
8705 	 * inner_map2 = bpf_map_lookup_elem(outer_map, key2)
8706 	 * if (inner_map1 && inner_map2) {
8707 	 *     timer = bpf_map_lookup_elem(inner_map1);
8708 	 *     if (timer)
8709 	 *         // mismatch would have been allowed
8710 	 *         bpf_timer_init(timer, inner_map2);
8711 	 * }
8712 	 *
8713 	 * Comparing map_ptr is enough to distinguish normal and outer maps.
8714 	 */
8715 	if (meta->map.ptr &&
8716 	    (meta->map.ptr != reg->map_ptr || meta->map.uid != reg->map_uid)) {
8717 		argno_t obj_argno = argno_from_reg(reg_from_argno(argno) - 1);
8718 		struct btf_record *rec = meta->map.ptr->record;
8719 		const char *obj_name = "workqueue";
8720 
8721 		if (rec->timer_off >= 0)
8722 			obj_name = "timer";
8723 		else if (rec->task_work_off >= 0)
8724 			obj_name = "bpf_task_work";
8725 
8726 		verbose(env, "%s pointer in %s map_uid=%d ",
8727 			obj_name, reg_arg_name(env, obj_argno), meta->map.uid);
8728 		verbose(env, "doesn't match map pointer in %s map_uid=%d\n",
8729 			reg_arg_name(env, argno), reg->map_uid);
8730 		return -EINVAL;
8731 	}
8732 
8733 	meta->map.ptr = reg->map_ptr;
8734 	meta->map.uid = reg->map_uid;
8735 	return 0;
8736 }
8737 
8738 static int check_func_arg(struct bpf_verifier_env *env, u32 arg,
8739 			  struct bpf_call_arg_meta *meta,
8740 			  int insn_idx)
8741 {
8742 	const struct bpf_func_proto *fn = meta->fn;
8743 	u32 regno = BPF_REG_1 + arg;
8744 	struct bpf_reg_state *reg = reg_state(env, regno);
8745 	enum bpf_arg_type arg_type = fn->arg_type[arg];
8746 	argno_t argno = argno_from_reg(regno);
8747 	enum bpf_reg_type type = reg->type;
8748 	u32 *arg_btf_id = NULL;
8749 	u32 key_size;
8750 	int err = 0;
8751 
8752 	if (arg_type == ARG_DONTCARE)
8753 		return 0;
8754 
8755 	err = check_reg_arg(env, regno, SRC_OP);
8756 	if (err)
8757 		return err;
8758 
8759 	if (arg_type == ARG_ANYTHING) {
8760 		if (is_pointer_value(env, regno)) {
8761 			verbose(env, "R%d leaks addr into helper function\n",
8762 				regno);
8763 			return -EACCES;
8764 		}
8765 		return 0;
8766 	}
8767 
8768 	if (type_is_pkt_pointer(type) &&
8769 	    !may_access_direct_pkt_data(env, fn, BPF_READ)) {
8770 		verbose(env, "helper access to the packet is not allowed\n");
8771 		return -EACCES;
8772 	}
8773 
8774 	if (base_type(arg_type) == ARG_PTR_TO_MAP_VALUE) {
8775 		err = resolve_map_arg_type(env, meta, &arg_type);
8776 		if (err)
8777 			return err;
8778 	}
8779 
8780 	if (bpf_register_is_null(reg) && type_may_be_null(arg_type)) {
8781 		/* A NULL register has a SCALAR_VALUE type, so skip
8782 		 * type checking.
8783 		 */
8784 		err = mark_chain_precision(env, regno);
8785 		if (err)
8786 			return err;
8787 		goto skip_type_check;
8788 	}
8789 
8790 	/* arg_btf_id and arg_size are in a union. */
8791 	if (base_type(arg_type) == ARG_PTR_TO_BTF_ID ||
8792 	    base_type(arg_type) == ARG_PTR_TO_SPIN_LOCK)
8793 		arg_btf_id = fn->arg_btf_id[arg];
8794 
8795 	err = check_reg_type(env, reg, argno, arg_type, arg_btf_id, meta,
8796 			     func_id_name(meta->func_id));
8797 	if (err)
8798 		return err;
8799 
8800 	err = check_func_arg_reg_off(env, reg, argno, arg_type);
8801 	if (err)
8802 		return err;
8803 
8804 skip_type_check:
8805 	if (arg_type_is_release(arg_type) && !arg_type_is_dynptr(arg_type) &&
8806 	    !reg_is_referenced(env, reg) && !bpf_register_is_null(reg)) {
8807 		verbose(env, "release helper %s expects referenced PTR_TO_BTF_ID passed to %s\n",
8808 			func_id_name(meta->func_id), reg_arg_name(env, argno));
8809 		bpf_diag_call_arg(
8810 			env, insn_idx, argno, func_id_name(meta->func_id),
8811 			"release helpers require a value that owns a live resource returned by a matching acquire helper",
8812 			"Pass the resource-owning pointer returned by the matching acquire helper, and avoid calling the release helper after ownership has already been transferred or released.");
8813 		return -EINVAL;
8814 	}
8815 
8816 	if (reg_is_referenced(env, reg))
8817 		update_ref_obj(&meta->ref_obj, reg);
8818 
8819 	switch (base_type(arg_type)) {
8820 	case ARG_CONST_MAP_PTR:
8821 		/* bpf_map_xxx(map_ptr) call: remember that map_ptr */
8822 		err = process_map_ptr_arg(env, reg, argno, meta);
8823 		if (err)
8824 			return err;
8825 		break;
8826 	case ARG_PTR_TO_MAP_KEY:
8827 		/* bpf_map_xxx(..., map_ptr, ..., key) call:
8828 		 * check that [key, key + map->key_size) are within
8829 		 * stack limits and initialized
8830 		 */
8831 		if (!meta->map.ptr) {
8832 			/* in function declaration map_ptr must come before
8833 			 * map_key, so that it's verified and known before
8834 			 * we have to check map_key here. Otherwise it means
8835 			 * that kernel subsystem misconfigured verifier
8836 			 */
8837 			verifier_bug(env, "invalid map_ptr to access map->key");
8838 			return -EFAULT;
8839 		}
8840 		key_size = meta->map.ptr->key_size;
8841 		err = check_helper_mem_access(env, reg, argno, key_size, BPF_READ, false, NULL,
8842 					      NULL);
8843 		if (err)
8844 			return err;
8845 		if (can_elide_value_nullness(meta->map.ptr)) {
8846 			err = get_constant_map_key(env, reg, key_size, &meta->const_map_key);
8847 			if (err < 0) {
8848 				meta->const_map_key = -1;
8849 				if (err == -EOPNOTSUPP)
8850 					err = 0;
8851 				else
8852 					return err;
8853 			}
8854 		}
8855 		break;
8856 	case ARG_PTR_TO_MAP_VALUE:
8857 		if (type_may_be_null(arg_type) && bpf_register_is_null(reg))
8858 			return 0;
8859 
8860 		/* bpf_map_xxx(..., map_ptr, ..., value) call:
8861 		 * check [value, value + map->value_size) validity
8862 		 */
8863 		if (!meta->map.ptr) {
8864 			/* kernel subsystem misconfigured verifier */
8865 			verifier_bug(env, "invalid map_ptr to access map->value");
8866 			return -EFAULT;
8867 		}
8868 
8869 		/*
8870 		 * Disable raw mode for bpf_map_peek_elem() on a bloom filter. The helper reads
8871 		 * the value buffer as an input rather than filling it.
8872 		 */
8873 		if (meta->func_id == BPF_FUNC_map_peek_elem &&
8874 		    meta->map.ptr->map_type == BPF_MAP_TYPE_BLOOM_FILTER)
8875 			meta->arg_raw_mem.regno = 0;
8876 
8877 		err = check_helper_mem_access(env, reg, argno, meta->map.ptr->value_size,
8878 					      arg_type & MEM_WRITE ? BPF_WRITE : BPF_READ,
8879 					      false, meta, NULL);
8880 		break;
8881 	case ARG_PTR_TO_PERCPU_BTF_ID:
8882 		if (!reg->btf_id) {
8883 			verbose(env, "Helper has invalid btf_id in R%d\n", regno);
8884 			return -EACCES;
8885 		}
8886 		meta->ret_btf = reg->btf;
8887 		meta->ret_btf_id = reg->btf_id;
8888 		break;
8889 	case ARG_PTR_TO_SPIN_LOCK:
8890 		if (in_rbtree_lock_required_cb(env)) {
8891 			verbose(env, "can't spin_{lock,unlock} in rbtree cb\n");
8892 			return -EACCES;
8893 		}
8894 		if (meta->func_id == BPF_FUNC_spin_lock) {
8895 			err = process_spin_lock(env, reg, argno, PROCESS_SPIN_LOCK);
8896 			if (err)
8897 				return err;
8898 		} else if (meta->func_id == BPF_FUNC_spin_unlock) {
8899 			err = process_spin_lock(env, reg, argno, 0);
8900 			if (err)
8901 				return err;
8902 		} else {
8903 			verifier_bug(env, "spin lock arg on unexpected helper");
8904 			return -EFAULT;
8905 		}
8906 		break;
8907 	case ARG_PTR_TO_TIMER:
8908 		err = process_timer_func(env, reg, argno, &meta->map);
8909 		if (err)
8910 			return err;
8911 		break;
8912 	case ARG_PTR_TO_FUNC:
8913 		meta->subprogno = reg->subprogno;
8914 		break;
8915 	case ARG_PTR_TO_MEM:
8916 		/* The access to this pointer is only checked when we hit the
8917 		 * next is_mem_size argument below.
8918 		 */
8919 		if (arg_type & MEM_FIXED_SIZE) {
8920 			err = check_mem_reg(env, reg, argno_from_reg(regno), fn->arg_size[arg],
8921 					    arg_type & MEM_WRITE ? BPF_WRITE : BPF_READ, meta, NULL);
8922 			if (err)
8923 				return err;
8924 			if (arg_type & MEM_ALIGNED)
8925 				err = check_ptr_alignment(env, reg, 0, fn->arg_size[arg], true);
8926 		}
8927 		break;
8928 	case ARG_MEM_SIZE:
8929 		err = check_mem_size_reg(env, reg_state(env, regno - 1), reg,
8930 					 argno_from_reg(regno - 1), argno,
8931 					 fn->arg_type[arg - 1] & MEM_WRITE ? BPF_WRITE : BPF_READ,
8932 					 false, meta, NULL);
8933 		break;
8934 	case ARG_MEM_SIZE_OR_ZERO:
8935 		err = check_mem_size_reg(env, reg_state(env, regno - 1), reg,
8936 					 argno_from_reg(regno - 1), argno,
8937 					 fn->arg_type[arg - 1] & MEM_WRITE ? BPF_WRITE : BPF_READ,
8938 					 true, meta, NULL);
8939 		break;
8940 	case ARG_PTR_TO_DYNPTR:
8941 		err = process_dynptr_func(env, reg, argno, insn_idx, func_id_name(meta->func_id),
8942 					  arg_type, &meta->ref_obj, &meta->dynptr);
8943 		if (err)
8944 			return err;
8945 		break;
8946 	case ARG_CONST_ALLOC_SIZE_OR_ZERO:
8947 		err = process_const_alloc_mem_size(env, reg, argno, &meta->ret_mem);
8948 		if (err)
8949 			return err;
8950 		break;
8951 	case ARG_PTR_TO_CONST_STR:
8952 	{
8953 		err = check_arg_const_str(env, reg, argno);
8954 		if (err)
8955 			return err;
8956 		break;
8957 	}
8958 	case ARG_KPTR_XCHG_DEST:
8959 		err = process_kptr_func(env, regno, meta);
8960 		if (err)
8961 			return err;
8962 		break;
8963 	}
8964 
8965 	return err;
8966 }
8967 
8968 static bool may_update_sockmap(struct bpf_verifier_env *env, int func_id)
8969 {
8970 	enum bpf_attach_type eatype = env->prog->expected_attach_type;
8971 	enum bpf_prog_type type = resolve_prog_type(env->prog);
8972 
8973 	if (func_id != BPF_FUNC_map_update_elem &&
8974 	    func_id != BPF_FUNC_map_delete_elem)
8975 		return false;
8976 
8977 	/* It's not possible to get access to a locked struct sock in these
8978 	 * contexts, so updating is safe.
8979 	 */
8980 	switch (type) {
8981 	case BPF_PROG_TYPE_TRACING:
8982 		if (eatype == BPF_TRACE_ITER)
8983 			return true;
8984 		break;
8985 	case BPF_PROG_TYPE_SOCK_OPS:
8986 		/* map_update allowed only via dedicated helpers with event type checks */
8987 		if (func_id == BPF_FUNC_map_delete_elem)
8988 			return true;
8989 		break;
8990 	case BPF_PROG_TYPE_SK_REUSEPORT:
8991 	case BPF_PROG_TYPE_SK_LOOKUP:
8992 		return true;
8993 	default:
8994 		break;
8995 	}
8996 
8997 	verbose(env, "cannot update sockmap in this context\n");
8998 	return false;
8999 }
9000 
9001 bool bpf_allow_tail_call_in_subprogs(struct bpf_verifier_env *env)
9002 {
9003 	return env->prog->jit_requested &&
9004 	       bpf_jit_supports_subprog_tailcalls();
9005 }
9006 
9007 static int check_map_func_compatibility(struct bpf_verifier_env *env,
9008 					struct bpf_map *map, int func_id)
9009 {
9010 	if (!map)
9011 		return 0;
9012 
9013 	/* We need a two way check, first is from map perspective ... */
9014 	switch (map->map_type) {
9015 	case BPF_MAP_TYPE_PROG_ARRAY:
9016 		if (func_id != BPF_FUNC_tail_call)
9017 			goto error;
9018 		break;
9019 	case BPF_MAP_TYPE_PERF_EVENT_ARRAY:
9020 		if (func_id != BPF_FUNC_perf_event_read &&
9021 		    func_id != BPF_FUNC_perf_event_output &&
9022 		    func_id != BPF_FUNC_skb_output &&
9023 		    func_id != BPF_FUNC_perf_event_read_value &&
9024 		    func_id != BPF_FUNC_xdp_output)
9025 			goto error;
9026 		break;
9027 	case BPF_MAP_TYPE_RINGBUF:
9028 		if (func_id != BPF_FUNC_ringbuf_output &&
9029 		    func_id != BPF_FUNC_ringbuf_reserve &&
9030 		    func_id != BPF_FUNC_ringbuf_query &&
9031 		    func_id != BPF_FUNC_ringbuf_reserve_dynptr &&
9032 		    func_id != BPF_FUNC_ringbuf_submit_dynptr &&
9033 		    func_id != BPF_FUNC_ringbuf_discard_dynptr)
9034 			goto error;
9035 		break;
9036 	case BPF_MAP_TYPE_USER_RINGBUF:
9037 		if (func_id != BPF_FUNC_user_ringbuf_drain)
9038 			goto error;
9039 		break;
9040 	case BPF_MAP_TYPE_STACK_TRACE:
9041 		if (func_id != BPF_FUNC_get_stackid)
9042 			goto error;
9043 		break;
9044 	case BPF_MAP_TYPE_CGROUP_ARRAY:
9045 		if (func_id != BPF_FUNC_skb_under_cgroup &&
9046 		    func_id != BPF_FUNC_current_task_under_cgroup)
9047 			goto error;
9048 		break;
9049 	case BPF_MAP_TYPE_CGROUP_STORAGE:
9050 	case BPF_MAP_TYPE_PERCPU_CGROUP_STORAGE:
9051 		if (func_id != BPF_FUNC_get_local_storage)
9052 			goto error;
9053 		break;
9054 	case BPF_MAP_TYPE_DEVMAP:
9055 	case BPF_MAP_TYPE_DEVMAP_HASH:
9056 		if (func_id != BPF_FUNC_redirect_map &&
9057 		    func_id != BPF_FUNC_map_lookup_elem)
9058 			goto error;
9059 		break;
9060 	/* Restrict bpf side of cpumap and xskmap, open when use-cases
9061 	 * appear.
9062 	 */
9063 	case BPF_MAP_TYPE_CPUMAP:
9064 		if (func_id != BPF_FUNC_redirect_map)
9065 			goto error;
9066 		break;
9067 	case BPF_MAP_TYPE_XSKMAP:
9068 		if (func_id != BPF_FUNC_redirect_map &&
9069 		    func_id != BPF_FUNC_map_lookup_elem)
9070 			goto error;
9071 		break;
9072 	case BPF_MAP_TYPE_ARRAY_OF_MAPS:
9073 	case BPF_MAP_TYPE_HASH_OF_MAPS:
9074 		if (func_id != BPF_FUNC_map_lookup_elem)
9075 			goto error;
9076 		break;
9077 	case BPF_MAP_TYPE_SOCKMAP:
9078 		if (func_id != BPF_FUNC_sk_redirect_map &&
9079 		    func_id != BPF_FUNC_sock_map_update &&
9080 		    func_id != BPF_FUNC_msg_redirect_map &&
9081 		    func_id != BPF_FUNC_sk_select_reuseport &&
9082 		    func_id != BPF_FUNC_map_lookup_elem &&
9083 		    !may_update_sockmap(env, func_id))
9084 			goto error;
9085 		break;
9086 	case BPF_MAP_TYPE_SOCKHASH:
9087 		if (func_id != BPF_FUNC_sk_redirect_hash &&
9088 		    func_id != BPF_FUNC_sock_hash_update &&
9089 		    func_id != BPF_FUNC_msg_redirect_hash &&
9090 		    func_id != BPF_FUNC_sk_select_reuseport &&
9091 		    func_id != BPF_FUNC_map_lookup_elem &&
9092 		    !may_update_sockmap(env, func_id))
9093 			goto error;
9094 		break;
9095 	case BPF_MAP_TYPE_REUSEPORT_SOCKARRAY:
9096 		if (func_id != BPF_FUNC_sk_select_reuseport)
9097 			goto error;
9098 		break;
9099 	case BPF_MAP_TYPE_QUEUE:
9100 	case BPF_MAP_TYPE_STACK:
9101 		if (func_id != BPF_FUNC_map_peek_elem &&
9102 		    func_id != BPF_FUNC_map_pop_elem &&
9103 		    func_id != BPF_FUNC_map_push_elem)
9104 			goto error;
9105 		break;
9106 	case BPF_MAP_TYPE_SK_STORAGE:
9107 		if (func_id != BPF_FUNC_sk_storage_get &&
9108 		    func_id != BPF_FUNC_sk_storage_delete &&
9109 		    func_id != BPF_FUNC_kptr_xchg)
9110 			goto error;
9111 		break;
9112 	case BPF_MAP_TYPE_INODE_STORAGE:
9113 		if (func_id != BPF_FUNC_inode_storage_get &&
9114 		    func_id != BPF_FUNC_inode_storage_delete &&
9115 		    func_id != BPF_FUNC_kptr_xchg)
9116 			goto error;
9117 		break;
9118 	case BPF_MAP_TYPE_TASK_STORAGE:
9119 		if (func_id != BPF_FUNC_task_storage_get &&
9120 		    func_id != BPF_FUNC_task_storage_delete &&
9121 		    func_id != BPF_FUNC_kptr_xchg)
9122 			goto error;
9123 		break;
9124 	case BPF_MAP_TYPE_CGRP_STORAGE:
9125 		if (func_id != BPF_FUNC_cgrp_storage_get &&
9126 		    func_id != BPF_FUNC_cgrp_storage_delete &&
9127 		    func_id != BPF_FUNC_kptr_xchg)
9128 			goto error;
9129 		break;
9130 	case BPF_MAP_TYPE_BLOOM_FILTER:
9131 		if (func_id != BPF_FUNC_map_peek_elem &&
9132 		    func_id != BPF_FUNC_map_push_elem)
9133 			goto error;
9134 		break;
9135 	case BPF_MAP_TYPE_INSN_ARRAY:
9136 		goto error;
9137 	default:
9138 		break;
9139 	}
9140 
9141 	/* ... and second from the function itself. */
9142 	switch (func_id) {
9143 	case BPF_FUNC_tail_call:
9144 		if (map->map_type != BPF_MAP_TYPE_PROG_ARRAY)
9145 			goto error;
9146 		if (env->subprog_cnt > 1 && !bpf_allow_tail_call_in_subprogs(env)) {
9147 			verbose(env, "mixing of tail_calls and bpf-to-bpf calls is not supported\n");
9148 			return -EINVAL;
9149 		}
9150 		break;
9151 	case BPF_FUNC_perf_event_read:
9152 	case BPF_FUNC_perf_event_output:
9153 	case BPF_FUNC_perf_event_read_value:
9154 	case BPF_FUNC_skb_output:
9155 	case BPF_FUNC_xdp_output:
9156 		if (map->map_type != BPF_MAP_TYPE_PERF_EVENT_ARRAY)
9157 			goto error;
9158 		break;
9159 	case BPF_FUNC_ringbuf_output:
9160 	case BPF_FUNC_ringbuf_reserve:
9161 	case BPF_FUNC_ringbuf_query:
9162 	case BPF_FUNC_ringbuf_reserve_dynptr:
9163 	case BPF_FUNC_ringbuf_submit_dynptr:
9164 	case BPF_FUNC_ringbuf_discard_dynptr:
9165 		if (map->map_type != BPF_MAP_TYPE_RINGBUF)
9166 			goto error;
9167 		break;
9168 	case BPF_FUNC_user_ringbuf_drain:
9169 		if (map->map_type != BPF_MAP_TYPE_USER_RINGBUF)
9170 			goto error;
9171 		break;
9172 	case BPF_FUNC_get_stackid:
9173 		if (map->map_type != BPF_MAP_TYPE_STACK_TRACE)
9174 			goto error;
9175 		break;
9176 	case BPF_FUNC_current_task_under_cgroup:
9177 	case BPF_FUNC_skb_under_cgroup:
9178 		if (map->map_type != BPF_MAP_TYPE_CGROUP_ARRAY)
9179 			goto error;
9180 		break;
9181 	case BPF_FUNC_redirect_map:
9182 		if (map->map_type != BPF_MAP_TYPE_DEVMAP &&
9183 		    map->map_type != BPF_MAP_TYPE_DEVMAP_HASH &&
9184 		    map->map_type != BPF_MAP_TYPE_CPUMAP &&
9185 		    map->map_type != BPF_MAP_TYPE_XSKMAP)
9186 			goto error;
9187 		break;
9188 	case BPF_FUNC_sk_redirect_map:
9189 	case BPF_FUNC_msg_redirect_map:
9190 	case BPF_FUNC_sock_map_update:
9191 		if (map->map_type != BPF_MAP_TYPE_SOCKMAP)
9192 			goto error;
9193 		break;
9194 	case BPF_FUNC_sk_redirect_hash:
9195 	case BPF_FUNC_msg_redirect_hash:
9196 	case BPF_FUNC_sock_hash_update:
9197 		if (map->map_type != BPF_MAP_TYPE_SOCKHASH)
9198 			goto error;
9199 		break;
9200 	case BPF_FUNC_get_local_storage:
9201 		if (map->map_type != BPF_MAP_TYPE_CGROUP_STORAGE &&
9202 		    map->map_type != BPF_MAP_TYPE_PERCPU_CGROUP_STORAGE)
9203 			goto error;
9204 		break;
9205 	case BPF_FUNC_sk_select_reuseport:
9206 		if (map->map_type != BPF_MAP_TYPE_REUSEPORT_SOCKARRAY &&
9207 		    map->map_type != BPF_MAP_TYPE_SOCKMAP &&
9208 		    map->map_type != BPF_MAP_TYPE_SOCKHASH)
9209 			goto error;
9210 		break;
9211 	case BPF_FUNC_map_pop_elem:
9212 		if (map->map_type != BPF_MAP_TYPE_QUEUE &&
9213 		    map->map_type != BPF_MAP_TYPE_STACK)
9214 			goto error;
9215 		break;
9216 	case BPF_FUNC_map_peek_elem:
9217 	case BPF_FUNC_map_push_elem:
9218 		if (map->map_type != BPF_MAP_TYPE_QUEUE &&
9219 		    map->map_type != BPF_MAP_TYPE_STACK &&
9220 		    map->map_type != BPF_MAP_TYPE_BLOOM_FILTER)
9221 			goto error;
9222 		break;
9223 	case BPF_FUNC_map_lookup_percpu_elem:
9224 		if (map->map_type != BPF_MAP_TYPE_PERCPU_ARRAY &&
9225 		    map->map_type != BPF_MAP_TYPE_PERCPU_HASH &&
9226 		    map->map_type != BPF_MAP_TYPE_LRU_PERCPU_HASH)
9227 			goto error;
9228 		break;
9229 	case BPF_FUNC_sk_storage_get:
9230 	case BPF_FUNC_sk_storage_delete:
9231 		if (map->map_type != BPF_MAP_TYPE_SK_STORAGE)
9232 			goto error;
9233 		break;
9234 	case BPF_FUNC_inode_storage_get:
9235 	case BPF_FUNC_inode_storage_delete:
9236 		if (map->map_type != BPF_MAP_TYPE_INODE_STORAGE)
9237 			goto error;
9238 		break;
9239 	case BPF_FUNC_task_storage_get:
9240 	case BPF_FUNC_task_storage_delete:
9241 		if (map->map_type != BPF_MAP_TYPE_TASK_STORAGE)
9242 			goto error;
9243 		break;
9244 	case BPF_FUNC_cgrp_storage_get:
9245 	case BPF_FUNC_cgrp_storage_delete:
9246 		if (map->map_type != BPF_MAP_TYPE_CGRP_STORAGE)
9247 			goto error;
9248 		break;
9249 	default:
9250 		break;
9251 	}
9252 
9253 	return 0;
9254 error:
9255 	verbose(env, "cannot pass map_type %d into func %s#%d\n",
9256 		map->map_type, func_id_name(func_id), func_id);
9257 	return -EINVAL;
9258 }
9259 
9260 static bool check_raw_mode_ok(const struct bpf_func_proto *fn, struct bpf_call_arg_meta *meta)
9261 {
9262 	int i;
9263 
9264 	for (i = 0; i < ARRAY_SIZE(fn->arg_type); i++) {
9265 		if (fn->arg_type[i] == ARG_DONTCARE)
9266 			break;
9267 		if (!arg_type_is_raw_mem(fn->arg_type[i]))
9268 			continue;
9269 		if (meta->arg_raw_mem.regno)
9270 			return false;
9271 		meta->arg_raw_mem.regno = i + 1;
9272 	}
9273 
9274 	return true;
9275 }
9276 
9277 static bool check_args_pair_invalid(const struct bpf_func_proto *fn, int arg)
9278 {
9279 	bool is_fixed = fn->arg_type[arg] & MEM_FIXED_SIZE;
9280 	bool has_size = fn->arg_size[arg] != 0;
9281 	bool is_next_size = false;
9282 
9283 	if (arg + 1 < ARRAY_SIZE(fn->arg_type))
9284 		is_next_size = arg_type_is_mem_size(fn->arg_type[arg + 1]);
9285 
9286 	if (base_type(fn->arg_type[arg]) != ARG_PTR_TO_MEM)
9287 		return is_next_size;
9288 
9289 	return has_size == is_next_size || is_next_size == is_fixed;
9290 }
9291 
9292 static bool check_arg_pair_ok(const struct bpf_func_proto *fn)
9293 {
9294 	/* bpf_xxx(..., buf, len) call will access 'len'
9295 	 * bytes from memory 'buf'. Both arg types need
9296 	 * to be paired, so make sure there's no buggy
9297 	 * helper function specification.
9298 	 */
9299 	if (arg_type_is_mem_size(fn->arg1_type) ||
9300 	    check_args_pair_invalid(fn, 0) ||
9301 	    check_args_pair_invalid(fn, 1) ||
9302 	    check_args_pair_invalid(fn, 2) ||
9303 	    check_args_pair_invalid(fn, 3) ||
9304 	    check_args_pair_invalid(fn, 4))
9305 		return false;
9306 
9307 	return true;
9308 }
9309 
9310 static bool check_btf_id_ok(const struct bpf_func_proto *fn)
9311 {
9312 	int i;
9313 
9314 	for (i = 0; i < ARRAY_SIZE(fn->arg_type); i++) {
9315 		if (fn->arg_type[i] == ARG_DONTCARE)
9316 			break;
9317 		if (base_type(fn->arg_type[i]) == ARG_PTR_TO_BTF_ID)
9318 			return !!fn->arg_btf_id[i];
9319 		if (base_type(fn->arg_type[i]) == ARG_PTR_TO_SPIN_LOCK)
9320 			return fn->arg_btf_id[i] == BPF_PTR_POISON;
9321 		if (base_type(fn->arg_type[i]) != ARG_PTR_TO_BTF_ID && fn->arg_btf_id[i] &&
9322 		    /* arg_btf_id and arg_size are in a union. */
9323 		    (base_type(fn->arg_type[i]) != ARG_PTR_TO_MEM ||
9324 		     !(fn->arg_type[i] & MEM_FIXED_SIZE)))
9325 			return false;
9326 	}
9327 
9328 	return true;
9329 }
9330 
9331 static bool check_mem_arg_rw_flag_ok(const struct bpf_func_proto *fn)
9332 {
9333 	int i;
9334 
9335 	for (i = 0; i < ARRAY_SIZE(fn->arg_type); i++) {
9336 		enum bpf_arg_type arg_type = fn->arg_type[i];
9337 
9338 		if (arg_type == ARG_DONTCARE)
9339 			break;
9340 		if (base_type(arg_type) != ARG_PTR_TO_MEM)
9341 			continue;
9342 		if (!(arg_type & (MEM_WRITE | MEM_RDONLY)))
9343 			return false;
9344 	}
9345 
9346 	return true;
9347 }
9348 
9349 static bool check_proto_release_reg(const struct bpf_func_proto *fn, struct bpf_call_arg_meta *meta)
9350 {
9351 	int i;
9352 
9353 	for (i = 0; i < ARRAY_SIZE(fn->arg_type); i++) {
9354 		enum bpf_arg_type arg_type = fn->arg_type[i];
9355 
9356 		if (arg_type == ARG_DONTCARE)
9357 			break;
9358 		if (arg_type_is_release(arg_type)) {
9359 			if (meta->release_regno)
9360 				return false;
9361 			meta->release_regno = i + 1;
9362 		}
9363 	}
9364 
9365 	return true;
9366 }
9367 
9368 static int check_func_proto(const struct bpf_func_proto *fn, struct bpf_call_arg_meta *meta)
9369 {
9370 	return check_raw_mode_ok(fn, meta) &&
9371 	       check_arg_pair_ok(fn) &&
9372 	       check_mem_arg_rw_flag_ok(fn) &&
9373 	       check_proto_release_reg(fn, meta) &&
9374 	       check_btf_id_ok(fn) ? 0 : -EINVAL;
9375 }
9376 
9377 /* Packet data might have moved, any old PTR_TO_PACKET[_META,_END]
9378  * are now invalid, so turn them into unknown SCALAR_VALUE.
9379  *
9380  * This also applies to dynptr slices belonging to skb and xdp dynptrs,
9381  * since these slices point to packet data.
9382  */
9383 static void clear_all_pkt_pointers(struct bpf_verifier_env *env)
9384 {
9385 	struct bpf_func_state *state;
9386 	struct bpf_reg_state *reg;
9387 
9388 	bpf_for_each_reg_in_vstate(env->cur_state, state, reg, ({
9389 		if (reg_is_pkt_pointer_any(reg) || reg_is_dynptr_slice_pkt(reg)) {
9390 			bpf_diag_record_scrub(env, reg, BPF_DIAG_MOD_PKT_DATA_CHANGE);
9391 			mark_reg_invalid(env, reg);
9392 		}
9393 	}));
9394 }
9395 
9396 enum {
9397 	AT_PKT_END = -1,
9398 	BEYOND_PKT_END = -2,
9399 };
9400 
9401 static void mark_pkt_end(struct bpf_verifier_state *vstate, int regn, bool range_open)
9402 {
9403 	struct bpf_func_state *state = vstate->frame[vstate->curframe];
9404 	struct bpf_reg_state *reg = &state->regs[regn];
9405 
9406 	if (reg->type != PTR_TO_PACKET)
9407 		/* PTR_TO_PACKET_META is not supported yet */
9408 		return;
9409 
9410 	/* The 'reg' is pkt > pkt_end or pkt >= pkt_end.
9411 	 * How far beyond pkt_end it goes is unknown.
9412 	 * if (!range_open) it's the case of pkt >= pkt_end
9413 	 * if (range_open) it's the case of pkt > pkt_end
9414 	 * hence this pointer is at least 1 byte bigger than pkt_end
9415 	 */
9416 	if (range_open)
9417 		reg->range = BEYOND_PKT_END;
9418 	else
9419 		reg->range = AT_PKT_END;
9420 }
9421 
9422 static int __release_reference_nomark(struct bpf_verifier_state *state, int id)
9423 {
9424 	int i;
9425 
9426 	for (i = 0; i < state->acquired_refs; i++) {
9427 		if (state->refs[i].type != REF_TYPE_PTR)
9428 			continue;
9429 		if (state->refs[i].id == id) {
9430 			release_reference_state(state, i);
9431 			return 0;
9432 		}
9433 	}
9434 	return -EINVAL;
9435 }
9436 
9437 static int release_reference_nomark(struct bpf_verifier_env *env, int id)
9438 {
9439 	int err;
9440 
9441 	err = __release_reference_nomark(env->cur_state, id);
9442 	if (!err)
9443 		bpf_diag_record_ref_release(env, env->insn_idx, id);
9444 	return err;
9445 }
9446 
9447 static int idstack_push(struct bpf_idmap *idmap, u32 id)
9448 {
9449 	int i;
9450 
9451 	if (!id)
9452 		return 0;
9453 
9454 	for (i = 0; i < idmap->cnt; i++)
9455 		if (idmap->map[i].old == id)
9456 			return 0;
9457 
9458 	if (WARN_ON_ONCE(idmap->cnt >= BPF_ID_MAP_SIZE))
9459 		return -EFAULT;
9460 
9461 	idmap->map[idmap->cnt++].old = id;
9462 	return 0;
9463 }
9464 
9465 static int idstack_pop(struct bpf_idmap *idmap)
9466 {
9467 	if (!idmap->cnt)
9468 		return 0;
9469 
9470 	return idmap->map[--idmap->cnt].old;
9471 }
9472 
9473 /* Release id and objects derived from it iteratively in a DFS manner */
9474 static int release_reference(struct bpf_verifier_env *env, int id)
9475 {
9476 	u32 mask = (1 << STACK_SPILL) | (1 << STACK_DYNPTR);
9477 	struct bpf_verifier_state *vstate = env->cur_state;
9478 	struct bpf_idmap *idstack = &env->idmap_scratch;
9479 	struct bpf_stack_state *stack;
9480 	struct bpf_func_state *state;
9481 	struct bpf_reg_state *reg;
9482 	int i, err;
9483 
9484 	idstack->cnt = 0;
9485 	err = idstack_push(idstack, id);
9486 	if (err)
9487 		return err;
9488 
9489 	if (find_reference_state(vstate, id)) {
9490 		err = release_reference_nomark(env, id);
9491 		WARN_ON_ONCE(err);
9492 	}
9493 
9494 	while ((id = idstack_pop(idstack))) {
9495 		/*
9496 		 * Child references are inaccessible after parent is released,
9497 		 * any child references that exist at this point are a leak.
9498 		 */
9499 		for (i = 0; i < vstate->acquired_refs; i++) {
9500 			if (vstate->refs[i].type != REF_TYPE_PTR)
9501 				continue;
9502 			if (vstate->refs[i].parent_id != id)
9503 				continue;
9504 			verbose(env, "Leaking reference id=%d alloc_insn=%d. Release it first.\n",
9505 				vstate->refs[i].id, vstate->refs[i].insn_idx);
9506 			return -EINVAL;
9507 		}
9508 
9509 		bpf_for_each_reg_in_vstate_mask(vstate, state, reg, stack, mask, ({
9510 			if (reg->id != id && reg->parent_id != id)
9511 				continue;
9512 
9513 			/* Free objects derived from the current object */
9514 			if (reg->parent_id == id) {
9515 				err = idstack_push(idstack, reg->id);
9516 				if (err)
9517 					return err;
9518 			}
9519 
9520 			/*
9521 			 * A dynptr occupies two stack slots that invalidate_dynptr()
9522 			 * clears together. Record both scrubs before invalidating it.
9523 			 */
9524 			if (stack && stack->slot_type[BPF_REG_SIZE - 1] == STACK_DYNPTR) {
9525 				struct bpf_stack_state *dyn_stack = stack;
9526 
9527 				if (reg->dynptr.first_slot)
9528 					dyn_stack--;
9529 				bpf_diag_record_scrub(env, &dyn_stack[0].spilled_ptr,
9530 						      BPF_DIAG_MOD_REF_RELEASE);
9531 				bpf_diag_record_scrub(env, &dyn_stack[1].spilled_ptr,
9532 						      BPF_DIAG_MOD_REF_RELEASE);
9533 				invalidate_dynptr(env, dyn_stack);
9534 				continue;
9535 			}
9536 			bpf_diag_record_scrub(env, reg, BPF_DIAG_MOD_REF_RELEASE);
9537 			if (!stack || stack->slot_type[BPF_REG_SIZE - 1] == STACK_SPILL)
9538 				mark_reg_invalid(env, reg);
9539 		}));
9540 	}
9541 
9542 	return 0;
9543 }
9544 
9545 static void invalidate_non_owning_refs(struct bpf_verifier_env *env)
9546 {
9547 	struct bpf_func_state *unused;
9548 	struct bpf_reg_state *reg;
9549 
9550 	bpf_for_each_reg_in_vstate(env->cur_state, unused, reg, ({
9551 		if (type_is_non_owning_ref(reg->type)) {
9552 			bpf_diag_record_scrub(env, reg, BPF_DIAG_MOD_NON_OWN_REF);
9553 			mark_reg_invalid(env, reg);
9554 		}
9555 	}));
9556 }
9557 
9558 static void invalidate_rcu_protected_refs(struct bpf_verifier_env *env)
9559 {
9560 	struct bpf_stack_state *stack;
9561 	struct bpf_func_state *state;
9562 	struct bpf_reg_state *reg;
9563 	u32 clear_mask = (1 << STACK_SPILL) | (1 << STACK_ITER);
9564 
9565 	bpf_for_each_reg_in_vstate_mask(env->cur_state, state, reg, stack, clear_mask, ({
9566 		if (reg->type & MEM_RCU) {
9567 			bpf_diag_mod_begin(env, reg, NULL, BPF_DIAG_MOD_WRITE);
9568 			reg->type &= ~(MEM_RCU | PTR_MAYBE_NULL | NON_OWN_REF);
9569 			reg->type |= PTR_UNTRUSTED;
9570 			bpf_diag_mod_end(env);
9571 		}
9572 	}));
9573 }
9574 
9575 static int ref_convert_alloc_rcu_protected(struct bpf_verifier_env *env, u32 id)
9576 {
9577 	struct bpf_func_state *state;
9578 	struct bpf_reg_state *reg;
9579 	int err;
9580 
9581 	err = release_reference_nomark(env, id);
9582 	if (err)
9583 		return err;
9584 
9585 	bpf_for_each_reg_in_vstate(env->cur_state, state, reg, ({
9586 		if (reg->id != id)
9587 			continue;
9588 		if ((reg->type & MEM_ALLOC) && (reg->type & MEM_PERCPU)) {
9589 			bpf_diag_mod_begin(env, reg, NULL, BPF_DIAG_MOD_WRITE);
9590 			reg->id = 0;
9591 			reg->type &= ~MEM_ALLOC;
9592 			reg->type |= MEM_RCU;
9593 			bpf_diag_mod_end(env);
9594 		}
9595 	}));
9596 
9597 	return err;
9598 }
9599 
9600 static void clear_caller_saved_regs(struct bpf_verifier_env *env,
9601 				    struct bpf_reg_state *regs)
9602 {
9603 	int i;
9604 
9605 	bpf_diag_record_caller_saved(env, regs);
9606 
9607 	/* after the call registers r0 - r5 were scratched */
9608 	for (i = 0; i < CALLER_SAVED_REGS; i++) {
9609 		bpf_mark_reg_not_init(env, &regs[caller_saved[i]]);
9610 		__check_reg_arg(env, regs, caller_saved[i], DST_OP_NO_MARK);
9611 	}
9612 }
9613 
9614 static void invalidate_outgoing_stack_args(struct bpf_verifier_env *env,
9615 					   struct bpf_func_state *state)
9616 {
9617 	int i, nslots = state->out_stack_arg_cnt;
9618 
9619 	for (i = 0; i < nslots; i++) {
9620 		bpf_diag_record_scrub(env, &state->stack_arg_regs[i], BPF_DIAG_MOD_CALLER_SAVED);
9621 		bpf_mark_reg_not_init(env, &state->stack_arg_regs[i]);
9622 	}
9623 }
9624 
9625 typedef int (*set_callee_state_fn)(struct bpf_verifier_env *env,
9626 				   struct bpf_func_state *caller,
9627 				   struct bpf_func_state *callee,
9628 				   int insn_idx);
9629 
9630 static int set_callee_state(struct bpf_verifier_env *env,
9631 			    struct bpf_func_state *caller,
9632 			    struct bpf_func_state *callee, int insn_idx);
9633 
9634 static int setup_func_entry(struct bpf_verifier_env *env, int subprog, int callsite,
9635 			    set_callee_state_fn set_callee_state_cb,
9636 			    struct bpf_verifier_state *state)
9637 {
9638 	struct bpf_func_state *caller, *callee;
9639 	int err;
9640 
9641 	if (state->curframe + 1 >= MAX_CALL_FRAMES) {
9642 		verbose(env, "the call stack of %d frames is too deep\n",
9643 			state->curframe + 2);
9644 		return -E2BIG;
9645 	}
9646 
9647 	if (state->frame[state->curframe + 1]) {
9648 		verifier_bug(env, "Frame %d already allocated", state->curframe + 1);
9649 		return -EFAULT;
9650 	}
9651 
9652 	caller = state->frame[state->curframe];
9653 	callee = kzalloc_obj(*callee, GFP_KERNEL_ACCOUNT);
9654 	if (!callee)
9655 		return -ENOMEM;
9656 	state->frame[state->curframe + 1] = callee;
9657 
9658 	/* callee cannot access r0, r6 - r9 for reading and has to write
9659 	 * into its own stack before reading from it.
9660 	 * callee can read/write into caller's stack
9661 	 */
9662 	init_func_state(env, callee,
9663 			/* remember the callsite, it will be used by bpf_exit */
9664 			callsite,
9665 			state->curframe + 1 /* frameno within this callchain */,
9666 			subprog /* subprog number within this prog */);
9667 	err = set_callee_state_cb(env, caller, callee, callsite);
9668 	if (err)
9669 		goto err_out;
9670 
9671 	/* only increment it after check_reg_arg() finished */
9672 	state->curframe++;
9673 
9674 	return 0;
9675 
9676 err_out:
9677 	free_func_state(callee);
9678 	state->frame[state->curframe + 1] = NULL;
9679 	return err;
9680 }
9681 
9682 static int btf_check_func_arg_match(struct bpf_verifier_env *env, int subprog,
9683 				    const struct btf *btf,
9684 				    struct bpf_reg_state *regs)
9685 {
9686 	struct bpf_subprog_info *sub = subprog_info(env, subprog);
9687 	struct bpf_func_state *caller = cur_func(env);
9688 	struct bpf_verifier_log *log = &env->log;
9689 	struct ref_obj_desc ref_obj = {};
9690 	const struct btf_param *args;
9691 	const struct btf_type *func, *func_proto;
9692 	u32 i;
9693 	int ret, err;
9694 
9695 	ret = btf_prepare_func_args(env, subprog);
9696 	if (ret) {
9697 		if (bpf_in_stack_arg_cnt(sub) > 0) {
9698 			err = check_outgoing_stack_args(env, caller, sub->arg_cnt,
9699 							bpf_subprog_name(env, subprog),
9700 							NULL, NULL);
9701 			if (err)
9702 				return err;
9703 		}
9704 		return ret;
9705 	}
9706 
9707 	func = btf_type_by_id(btf, env->prog->aux->func_info[subprog].type_id);
9708 	func_proto = btf_type_by_id(btf, func->type);
9709 	args = btf_params(func_proto);
9710 	ret = check_outgoing_stack_args(env, caller, sub->arg_cnt,
9711 					bpf_subprog_name(env, subprog), btf, args);
9712 	if (ret)
9713 		return ret;
9714 
9715 	/* check that BTF function arguments match actual types that the
9716 	 * verifier sees.
9717 	 */
9718 	for (i = 0; i < sub->arg_cnt; i++) {
9719 		argno_t argno = argno_from_arg(i + 1);
9720 		struct bpf_reg_state *reg = get_func_arg_reg(caller, regs, i);
9721 		struct bpf_subprog_arg_info *arg = &sub->args[i];
9722 
9723 		if (arg->arg_type == ARG_ANYTHING) {
9724 			if (reg->type != SCALAR_VALUE) {
9725 				bpf_log(log, "%s is not a scalar\n", reg_arg_name(env, argno));
9726 				return -EINVAL;
9727 			}
9728 		} else if (arg->arg_type & PTR_UNTRUSTED) {
9729 			/*
9730 			 * Anything is allowed for untrusted arguments, as these are
9731 			 * read-only and probe read instructions would protect against
9732 			 * invalid memory access.
9733 			 */
9734 		} else if (arg->arg_type == ARG_PTR_TO_CTX) {
9735 			ret = check_func_arg_reg_off(env, reg, argno, ARG_PTR_TO_CTX);
9736 			if (ret < 0)
9737 				return ret;
9738 			/* If function expects ctx type in BTF check that caller
9739 			 * is passing PTR_TO_CTX.
9740 			 */
9741 			if (reg->type != PTR_TO_CTX) {
9742 				bpf_log(log, "%s expects pointer to ctx\n",
9743 					reg_arg_name(env, argno));
9744 				return -EINVAL;
9745 			}
9746 		} else if (base_type(arg->arg_type) == ARG_PTR_TO_MEM) {
9747 			ret = check_func_arg_reg_off(env, reg, argno, ARG_DONTCARE);
9748 			if (ret < 0)
9749 				return ret;
9750 			if (check_mem_reg(env, reg, argno, arg->mem_size, BPF_READ | BPF_WRITE, NULL,
9751 					  NULL))
9752 				return -EINVAL;
9753 			/*
9754 			 * PTR_TO_PACKET get passed as PTR_TO_MEM, preventing
9755 			 * us from adjusting bounds tracking info.
9756 			 */
9757 			if ((reg_is_pkt_pointer_any(reg) || reg_is_dynptr_slice_pkt(reg)) &&
9758 			    sub->changes_pkt_data) {
9759 				bpf_log(log, "%s is a packet pointer, but func#%d may change packet data\n",
9760 						reg_arg_name(env, argno), subprog);
9761 				return -EINVAL;
9762 			}
9763 			if (!(arg->arg_type & PTR_MAYBE_NULL) &&
9764 			    (type_may_be_null(reg->type) || bpf_register_is_null(reg))) {
9765 				bpf_log(log, "%s is expected to be non-NULL\n",
9766 					reg_arg_name(env, argno));
9767 				return -EINVAL;
9768 			}
9769 		} else if (base_type(arg->arg_type) == ARG_PTR_TO_ARENA) {
9770 			/*
9771 			 * Can pass any value and the kernel won't crash, but
9772 			 * only PTR_TO_ARENA or SCALAR make sense. Everything
9773 			 * else is a bug in the bpf program. Point it out to
9774 			 * the user at the verification time instead of
9775 			 * run-time debug nightmare.
9776 			 */
9777 			if (reg->type != PTR_TO_ARENA && reg->type != SCALAR_VALUE) {
9778 				bpf_log(log, "%s is not a pointer to arena or scalar.\n",
9779 					reg_arg_name(env, argno));
9780 				return -EINVAL;
9781 			}
9782 		} else if (arg->arg_type == ARG_PTR_TO_DYNPTR) {
9783 			ret = check_func_arg_reg_off(env, reg, argno, ARG_PTR_TO_DYNPTR);
9784 			if (ret)
9785 				return ret;
9786 
9787 			ret = process_dynptr_func(env, reg, argno, env->insn_idx,
9788 						  bpf_subprog_name(env, subprog), arg->arg_type,
9789 						  &ref_obj, NULL);
9790 			if (ret)
9791 				return ret;
9792 		} else if (base_type(arg->arg_type) == ARG_PTR_TO_BTF_ID) {
9793 			struct bpf_call_arg_meta meta;
9794 			int err;
9795 
9796 			if (bpf_register_is_null(reg) && type_may_be_null(arg->arg_type)) {
9797 				err = mark_arg_precision(env, argno);
9798 				if (err)
9799 					return err;
9800 				continue;
9801 			}
9802 
9803 			memset(&meta, 0, sizeof(meta)); /* leave func_id as zero */
9804 			err = check_reg_type(env, reg, argno, arg->arg_type, &arg->btf_id, &meta,
9805 					     bpf_subprog_name(env, subprog));
9806 			err = err ?: check_func_arg_reg_off(env, reg, argno, arg->arg_type);
9807 			if (err)
9808 				return err;
9809 		} else {
9810 			verifier_bug(env, "unrecognized %s type %d",
9811 				     reg_arg_name(env, argno), arg->arg_type);
9812 			return -EFAULT;
9813 		}
9814 	}
9815 
9816 	return 0;
9817 }
9818 
9819 /* Compare BTF of a function call with given bpf_reg_state.
9820  * Returns:
9821  * EFAULT - there is a verifier bug. Abort verification.
9822  * EINVAL - there is a type mismatch or BTF is not available.
9823  * 0 - BTF matches with what bpf_reg_state expects.
9824  * Only PTR_TO_CTX and SCALAR_VALUE states are recognized.
9825  */
9826 static int btf_check_subprog_call(struct bpf_verifier_env *env, int subprog,
9827 				  struct bpf_reg_state *regs)
9828 {
9829 	struct bpf_prog *prog = env->prog;
9830 	struct btf *btf = prog->aux->btf;
9831 	u32 btf_id;
9832 	int err;
9833 
9834 	if (!prog->aux->func_info)
9835 		return -EINVAL;
9836 
9837 	btf_id = prog->aux->func_info[subprog].type_id;
9838 	if (!btf_id)
9839 		return -EFAULT;
9840 
9841 	if (prog->aux->func_info_aux[subprog].unreliable)
9842 		return -EINVAL;
9843 
9844 	err = btf_check_func_arg_match(env, subprog, btf, regs);
9845 	/* Compiler optimizations can remove arguments from static functions
9846 	 * or mismatched type can be passed into a global function.
9847 	 * In such cases mark the function as unreliable from BTF point of view.
9848 	 */
9849 	if (err)
9850 		prog->aux->func_info_aux[subprog].unreliable = true;
9851 	return err;
9852 }
9853 
9854 static int push_callback_call(struct bpf_verifier_env *env, struct bpf_insn *insn,
9855 			      int insn_idx, int subprog,
9856 			      set_callee_state_fn set_callee_state_cb)
9857 {
9858 	struct bpf_verifier_state *state = env->cur_state, *callback_state;
9859 	struct bpf_func_state *caller, *callee;
9860 	int err;
9861 
9862 	caller = state->frame[state->curframe];
9863 	err = btf_check_subprog_call(env, subprog, caller->regs);
9864 	if (err == -EFAULT)
9865 		return err;
9866 
9867 	/* set_callee_state is used for direct subprog calls, but we are
9868 	 * interested in validating only BPF helpers that can call subprogs as
9869 	 * callbacks
9870 	 */
9871 	env->subprog_info[subprog].is_cb = true;
9872 	if (bpf_pseudo_kfunc_call(insn) &&
9873 	    !is_callback_calling_kfunc(insn->imm)) {
9874 		verifier_bug(env, "kfunc %s#%d not marked as callback-calling",
9875 			     func_id_name(insn->imm), insn->imm);
9876 		return -EFAULT;
9877 	} else if (!bpf_pseudo_kfunc_call(insn) &&
9878 		   !is_callback_calling_function(insn->imm)) { /* helper */
9879 		verifier_bug(env, "helper %s#%d not marked as callback-calling",
9880 			     func_id_name(insn->imm), insn->imm);
9881 		return -EFAULT;
9882 	}
9883 
9884 	if (bpf_is_async_callback_calling_insn(insn)) {
9885 		struct bpf_verifier_state *async_cb;
9886 
9887 		/* there is no real recursion here. timer and workqueue callbacks are async */
9888 		env->subprog_info[subprog].is_async_cb = true;
9889 		async_cb = push_async_cb(env, env->subprog_info[subprog].start,
9890 					 insn_idx, subprog,
9891 					 is_async_cb_sleepable(env, insn));
9892 		if (IS_ERR(async_cb))
9893 			return PTR_ERR(async_cb);
9894 		callee = async_cb->frame[0];
9895 		callee->async_entry_cnt = caller->async_entry_cnt + 1;
9896 
9897 		/* Convert bpf_timer_set_callback() args into timer callback args */
9898 		err = set_callee_state_cb(env, caller, callee, insn_idx);
9899 		if (err)
9900 			return err;
9901 
9902 		return 0;
9903 	}
9904 
9905 	/* for callback functions enqueue entry to callback and
9906 	 * proceed with next instruction within current frame.
9907 	 */
9908 	callback_state = push_stack(env, env->subprog_info[subprog].start, insn_idx, false);
9909 	if (IS_ERR(callback_state))
9910 		return PTR_ERR(callback_state);
9911 
9912 	err = setup_func_entry(env, subprog, insn_idx, set_callee_state_cb,
9913 			       callback_state);
9914 	if (err)
9915 		return err;
9916 
9917 	callback_state->callback_unroll_depth++;
9918 	callback_state->frame[callback_state->curframe - 1]->callback_depth++;
9919 	caller->callback_depth = 0;
9920 	return 0;
9921 }
9922 
9923 static int process_bpf_exit_full(struct bpf_verifier_env *env,
9924 				 bool *do_print_state, bool exception_exit);
9925 
9926 static int check_func_call(struct bpf_verifier_env *env, struct bpf_insn *insn,
9927 			   int *insn_idx)
9928 {
9929 	struct bpf_verifier_state *state = env->cur_state;
9930 	struct bpf_subprog_info *caller_info;
9931 	u16 callee_incoming, stack_arg_cnt;
9932 	struct bpf_func_state *caller;
9933 	int err, subprog, target_insn;
9934 
9935 	target_insn = *insn_idx + insn->imm + 1;
9936 	subprog = bpf_find_subprog(env, target_insn);
9937 	if (verifier_bug_if(subprog < 0, env, "target of func call at insn %d is not a program",
9938 			    target_insn))
9939 		return -EFAULT;
9940 
9941 	caller = state->frame[state->curframe];
9942 	err = btf_check_subprog_call(env, subprog, caller->regs);
9943 	if (err == -EFAULT)
9944 		return err;
9945 	if (bpf_subprog_is_global(env, subprog)) {
9946 		struct bpf_func_info_aux *sub_aux = subprog_aux(env, subprog);
9947 		const char *sub_name = bpf_subprog_name(env, subprog);
9948 		const char *operation;
9949 		bool returns_void;
9950 
9951 		if (env->cur_state->active_locks) {
9952 			verbose(env, "global function calls are not allowed while holding a lock,\n"
9953 				     "use static function instead\n");
9954 			operation = bpf_diag_fmt(env, "global function %s()", sub_name);
9955 			bpf_diag_ctx_active(env, *insn_idx, operation, BPF_DIAG_CONTEXT_LOCK,
9956 					    "Release the lock before calling the global function, or use a static function instead.");
9957 			return -EINVAL;
9958 		}
9959 
9960 		if (env->subprog_info[subprog].might_sleep && !in_sleepable_context(env)) {
9961 			verbose(env, "sleepable global function %s() called in %s\n",
9962 				sub_name, non_sleepable_context_description(env));
9963 			operation = bpf_diag_fmt(env, "sleepable global function %s()", sub_name);
9964 			bpf_diag_ctx_forbidden(env, *insn_idx, operation,
9965 				"Move the call outside the critical section, or use a non-sleepable function.");
9966 			return -EINVAL;
9967 		}
9968 
9969 		if (err) {
9970 			verbose(env, "Caller passes invalid args into func#%d ('%s')\n",
9971 				subprog, sub_name);
9972 			return err;
9973 		}
9974 
9975 		if (env->log.level & BPF_LOG_LEVEL)
9976 			verbose(env, "Func#%d ('%s') is global and assumed valid.\n",
9977 				subprog, sub_name);
9978 		sub_aux->called[in_sleepable_context(env)] = true;
9979 		returns_void = subprog_returns_void(env, subprog);
9980 		if (env->subprog_info[subprog].changes_pkt_data)
9981 			clear_all_pkt_pointers(env);
9982 		if (returns_void)
9983 			bpf_diag_record_scrub(env, &caller->regs[BPF_REG_0], BPF_DIAG_MOD_CALLER_SAVED);
9984 		else
9985 			bpf_diag_mod_begin(env, &caller->regs[BPF_REG_0], NULL, BPF_DIAG_MOD_WRITE);
9986 		clear_caller_saved_regs(env, caller->regs);
9987 		invalidate_outgoing_stack_args(env, cur_func(env));
9988 
9989 		/* All non-void global functions return a 64-bit SCALAR_VALUE. */
9990 		if (!returns_void) {
9991 			mark_reg_unknown(env, caller->regs, BPF_REG_0);
9992 			bpf_diag_mod_end(env);
9993 		}
9994 
9995 		if (env->subprog_info[subprog].might_throw) {
9996 			struct bpf_verifier_state *branch;
9997 
9998 			branch = push_stack(env, *insn_idx + 1, *insn_idx, false);
9999 			if (IS_ERR(branch)) {
10000 				verbose(env, "failed to push state for global subprog exception path\n");
10001 				return PTR_ERR(branch);
10002 			}
10003 			return process_bpf_exit_full(env, NULL, true);
10004 		}
10005 
10006 		/* continue with next insn after call */
10007 		return 0;
10008 	}
10009 
10010 	/*
10011 	 * Track caller's total stack arg count (incoming + max outgoing).
10012 	 * This is needed so the JIT knows how much stack arg space to allocate.
10013 	 */
10014 	caller_info = &env->subprog_info[caller->subprogno];
10015 	callee_incoming = bpf_in_stack_arg_cnt(&env->subprog_info[subprog]);
10016 	stack_arg_cnt = bpf_in_stack_arg_cnt(caller_info) + callee_incoming;
10017 	if (stack_arg_cnt > caller_info->stack_arg_cnt)
10018 		caller_info->stack_arg_cnt = stack_arg_cnt;
10019 
10020 	/* for regular function entry setup new frame and continue
10021 	 * from that frame.
10022 	 */
10023 	err = setup_func_entry(env, subprog, *insn_idx, set_callee_state, state);
10024 	if (err)
10025 		return err;
10026 
10027 	bpf_diag_record_scrub(env, &caller->regs[BPF_REG_0], BPF_DIAG_MOD_CALLER_SAVED);
10028 	clear_caller_saved_regs(env, caller->regs);
10029 
10030 	/* and go analyze first insn of the callee */
10031 	*insn_idx = env->subprog_info[subprog].start - 1;
10032 
10033 	if (env->log.level & BPF_LOG_LEVEL) {
10034 		verbose(env, "caller:\n");
10035 		print_verifier_state(env, state, caller->frameno, true);
10036 		verbose(env, "callee:\n");
10037 		print_verifier_state(env, state, state->curframe, true);
10038 	}
10039 
10040 	return 0;
10041 }
10042 
10043 int map_set_for_each_callback_args(struct bpf_verifier_env *env,
10044 				   struct bpf_func_state *caller,
10045 				   struct bpf_func_state *callee)
10046 {
10047 	/* bpf_for_each_map_elem(struct bpf_map *map, void *callback_fn,
10048 	 *      void *callback_ctx, u64 flags);
10049 	 * callback_fn(struct bpf_map *map, void *key, void *value,
10050 	 *      void *callback_ctx);
10051 	 */
10052 	callee->regs[BPF_REG_1] = caller->regs[BPF_REG_1];
10053 
10054 	callee->regs[BPF_REG_2].type = PTR_TO_MAP_KEY;
10055 	__mark_reg_known_zero(&callee->regs[BPF_REG_2]);
10056 	callee->regs[BPF_REG_2].map_ptr = caller->regs[BPF_REG_1].map_ptr;
10057 	callee->regs[BPF_REG_2].map_uid = caller->regs[BPF_REG_1].map_uid;
10058 
10059 	callee->regs[BPF_REG_3].type = PTR_TO_MAP_VALUE;
10060 	__mark_reg_known_zero(&callee->regs[BPF_REG_3]);
10061 	callee->regs[BPF_REG_3].map_ptr = caller->regs[BPF_REG_1].map_ptr;
10062 	callee->regs[BPF_REG_3].map_uid = caller->regs[BPF_REG_1].map_uid;
10063 	callee->regs[BPF_REG_3].id = ++env->id_gen;
10064 
10065 	/* pointer to stack or null */
10066 	callee->regs[BPF_REG_4] = caller->regs[BPF_REG_3];
10067 
10068 	/* unused */
10069 	bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_5]);
10070 	return 0;
10071 }
10072 
10073 static int set_callee_state(struct bpf_verifier_env *env,
10074 			    struct bpf_func_state *caller,
10075 			    struct bpf_func_state *callee, int insn_idx)
10076 {
10077 	int i;
10078 
10079 	/* copy r1 - r5 args that callee can access.  The copy includes parent
10080 	 * pointers, which connects us up to the liveness chain
10081 	 */
10082 	for (i = BPF_REG_1; i <= BPF_REG_5; i++)
10083 		callee->regs[i] = caller->regs[i];
10084 	return 0;
10085 }
10086 
10087 static int set_map_elem_callback_state(struct bpf_verifier_env *env,
10088 				       struct bpf_func_state *caller,
10089 				       struct bpf_func_state *callee,
10090 				       int insn_idx)
10091 {
10092 	struct bpf_insn_aux_data *insn_aux = &env->insn_aux_data[insn_idx];
10093 	struct bpf_map *map;
10094 	int err;
10095 
10096 	/* valid map_ptr and poison value does not matter */
10097 	map = insn_aux->map_ptr_state.map_ptr;
10098 	if (!map->ops->map_set_for_each_callback_args ||
10099 	    !map->ops->map_for_each_callback) {
10100 		verbose(env, "callback function not allowed for map\n");
10101 		return -ENOTSUPP;
10102 	}
10103 
10104 	err = map->ops->map_set_for_each_callback_args(env, caller, callee);
10105 	if (err)
10106 		return err;
10107 
10108 	callee->in_callback_fn = true;
10109 	callee->callback_ret_range = retval_range(0, 1);
10110 	return 0;
10111 }
10112 
10113 static int set_loop_callback_state(struct bpf_verifier_env *env,
10114 				   struct bpf_func_state *caller,
10115 				   struct bpf_func_state *callee,
10116 				   int insn_idx)
10117 {
10118 	/* bpf_loop(u32 nr_loops, void *callback_fn, void *callback_ctx,
10119 	 *	    u64 flags);
10120 	 * callback_fn(u64 index, void *callback_ctx);
10121 	 */
10122 	callee->regs[BPF_REG_1].type = SCALAR_VALUE;
10123 	callee->regs[BPF_REG_2] = caller->regs[BPF_REG_3];
10124 
10125 	/* unused */
10126 	bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_3]);
10127 	bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_4]);
10128 	bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_5]);
10129 
10130 	callee->in_callback_fn = true;
10131 	callee->callback_ret_range = retval_range(0, 1);
10132 	return 0;
10133 }
10134 
10135 static int set_timer_callback_state(struct bpf_verifier_env *env,
10136 				    struct bpf_func_state *caller,
10137 				    struct bpf_func_state *callee,
10138 				    int insn_idx)
10139 {
10140 	struct bpf_map *map_ptr = caller->regs[BPF_REG_1].map_ptr;
10141 	u32 map_uid = caller->regs[BPF_REG_1].map_uid;
10142 
10143 	/* bpf_timer_set_callback(struct bpf_timer *timer, void *callback_fn);
10144 	 * callback_fn(struct bpf_map *map, void *key, void *value);
10145 	 */
10146 	callee->regs[BPF_REG_1].type = CONST_PTR_TO_MAP;
10147 	__mark_reg_known_zero(&callee->regs[BPF_REG_1]);
10148 	callee->regs[BPF_REG_1].map_ptr = map_ptr;
10149 	callee->regs[BPF_REG_1].map_uid = map_uid;
10150 
10151 	callee->regs[BPF_REG_2].type = PTR_TO_MAP_KEY;
10152 	__mark_reg_known_zero(&callee->regs[BPF_REG_2]);
10153 	callee->regs[BPF_REG_2].map_ptr = map_ptr;
10154 	callee->regs[BPF_REG_2].map_uid = map_uid;
10155 
10156 	callee->regs[BPF_REG_3].type = PTR_TO_MAP_VALUE;
10157 	__mark_reg_known_zero(&callee->regs[BPF_REG_3]);
10158 	callee->regs[BPF_REG_3].map_ptr = map_ptr;
10159 	callee->regs[BPF_REG_3].map_uid = map_uid;
10160 	callee->regs[BPF_REG_3].id = ++env->id_gen;
10161 
10162 	/* unused */
10163 	bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_4]);
10164 	bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_5]);
10165 	callee->in_async_callback_fn = true;
10166 	callee->callback_ret_range = retval_range(0, 0);
10167 	return 0;
10168 }
10169 
10170 static int set_find_vma_callback_state(struct bpf_verifier_env *env,
10171 				       struct bpf_func_state *caller,
10172 				       struct bpf_func_state *callee,
10173 				       int insn_idx)
10174 {
10175 	/* bpf_find_vma(struct task_struct *task, u64 addr,
10176 	 *               void *callback_fn, void *callback_ctx, u64 flags)
10177 	 * (callback_fn)(struct task_struct *task,
10178 	 *               struct vm_area_struct *vma, void *callback_ctx);
10179 	 */
10180 	callee->regs[BPF_REG_1] = caller->regs[BPF_REG_1];
10181 
10182 	callee->regs[BPF_REG_2].type = PTR_TO_BTF_ID;
10183 	__mark_reg_known_zero(&callee->regs[BPF_REG_2]);
10184 	callee->regs[BPF_REG_2].btf =  btf_vmlinux;
10185 	callee->regs[BPF_REG_2].btf_id = btf_tracing_ids[BTF_TRACING_TYPE_VMA];
10186 
10187 	/* pointer to stack or null */
10188 	callee->regs[BPF_REG_3] = caller->regs[BPF_REG_4];
10189 
10190 	/* unused */
10191 	bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_4]);
10192 	bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_5]);
10193 	callee->in_callback_fn = true;
10194 	callee->callback_ret_range = retval_range(0, 1);
10195 	return 0;
10196 }
10197 
10198 static int set_user_ringbuf_callback_state(struct bpf_verifier_env *env,
10199 					   struct bpf_func_state *caller,
10200 					   struct bpf_func_state *callee,
10201 					   int insn_idx)
10202 {
10203 	/* bpf_user_ringbuf_drain(struct bpf_map *map, void *callback_fn, void
10204 	 *			  callback_ctx, u64 flags);
10205 	 * callback_fn(const struct bpf_dynptr_t* dynptr, void *callback_ctx);
10206 	 */
10207 	bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_0]);
10208 	mark_dynptr_cb_reg(env, &callee->regs[BPF_REG_1], BPF_DYNPTR_TYPE_LOCAL);
10209 	callee->regs[BPF_REG_2] = caller->regs[BPF_REG_3];
10210 
10211 	/* unused */
10212 	bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_3]);
10213 	bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_4]);
10214 	bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_5]);
10215 
10216 	callee->in_callback_fn = true;
10217 	callee->callback_ret_range = retval_range(0, 1);
10218 	return 0;
10219 }
10220 
10221 static int set_rbtree_add_callback_state(struct bpf_verifier_env *env,
10222 					 struct bpf_func_state *caller,
10223 					 struct bpf_func_state *callee,
10224 					 int insn_idx)
10225 {
10226 	/* void bpf_rbtree_add_impl(struct bpf_rb_root *root, struct bpf_rb_node *node,
10227 	 *                     bool (less)(struct bpf_rb_node *a, const struct bpf_rb_node *b));
10228 	 *
10229 	 * 'struct bpf_rb_node *node' arg to bpf_rbtree_add_impl is the same PTR_TO_BTF_ID w/ offset
10230 	 * that 'less' callback args will be receiving. However, 'node' arg was release_reference'd
10231 	 * by this point, so look at 'root'
10232 	 */
10233 	struct btf_field *field;
10234 
10235 	field = reg_find_field_offset(&caller->regs[BPF_REG_1],
10236 				      caller->regs[BPF_REG_1].var_off.value,
10237 				      BPF_RB_ROOT);
10238 	if (!field || !field->graph_root.value_btf_id)
10239 		return -EFAULT;
10240 
10241 	mark_reg_graph_node(callee->regs, BPF_REG_1, &field->graph_root);
10242 	ref_set_non_owning(env, &callee->regs[BPF_REG_1]);
10243 	mark_reg_graph_node(callee->regs, BPF_REG_2, &field->graph_root);
10244 	ref_set_non_owning(env, &callee->regs[BPF_REG_2]);
10245 
10246 	bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_3]);
10247 	bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_4]);
10248 	bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_5]);
10249 	callee->in_callback_fn = true;
10250 	callee->callback_ret_range = retval_range(0, 1);
10251 	return 0;
10252 }
10253 
10254 static int set_task_work_schedule_callback_state(struct bpf_verifier_env *env,
10255 						 struct bpf_func_state *caller,
10256 						 struct bpf_func_state *callee,
10257 						 int insn_idx)
10258 {
10259 	struct bpf_map *map_ptr = caller->regs[BPF_REG_3].map_ptr;
10260 	u32 map_uid = caller->regs[BPF_REG_3].map_uid;
10261 
10262 	/*
10263 	 * callback_fn(struct bpf_map *map, void *key, void *value);
10264 	 */
10265 	callee->regs[BPF_REG_1].type = CONST_PTR_TO_MAP;
10266 	__mark_reg_known_zero(&callee->regs[BPF_REG_1]);
10267 	callee->regs[BPF_REG_1].map_ptr = map_ptr;
10268 	callee->regs[BPF_REG_1].map_uid = map_uid;
10269 
10270 	callee->regs[BPF_REG_2].type = PTR_TO_MAP_KEY;
10271 	__mark_reg_known_zero(&callee->regs[BPF_REG_2]);
10272 	callee->regs[BPF_REG_2].map_ptr = map_ptr;
10273 	callee->regs[BPF_REG_2].map_uid = map_uid;
10274 
10275 	callee->regs[BPF_REG_3].type = PTR_TO_MAP_VALUE;
10276 	__mark_reg_known_zero(&callee->regs[BPF_REG_3]);
10277 	callee->regs[BPF_REG_3].map_ptr = map_ptr;
10278 	callee->regs[BPF_REG_3].map_uid = map_uid;
10279 	callee->regs[BPF_REG_3].id = ++env->id_gen;
10280 
10281 	/* unused */
10282 	bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_4]);
10283 	bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_5]);
10284 	callee->in_async_callback_fn = true;
10285 	callee->callback_ret_range = retval_range(S32_MIN, S32_MAX);
10286 	return 0;
10287 }
10288 
10289 static bool is_rbtree_lock_required_kfunc(u32 btf_id);
10290 
10291 static void account_processed_insn(struct bpf_verifier_env *env)
10292 {
10293 	struct bpf_func_state *frame = cur_func(env);
10294 
10295 	env->insn_processed++;
10296 	frame->insns_subtotal++;
10297 	env->subprog_info[frame->subprogno].insns_self++;
10298 }
10299 
10300 static void account_processed_insns(struct bpf_verifier_env *env,
10301 				    struct bpf_func_state *callee,
10302 				    struct bpf_func_state *caller)
10303 {
10304 	u32 insns;
10305 
10306 	if (!callee)
10307 		return;
10308 
10309 	insns = callee->insns_subtotal;
10310 
10311 	env->subprog_info[callee->subprogno].insns_total += insns;
10312 	if (caller)
10313 		caller->insns_subtotal += insns;
10314 	callee->insns_subtotal = 0;
10315 }
10316 
10317 static void account_current_path(struct bpf_verifier_env *env)
10318 {
10319 	struct bpf_verifier_state *state = env->cur_state;
10320 	int frame;
10321 
10322 	for (frame = state->curframe; frame >= 0; frame--)
10323 		account_processed_insns(env, state->frame[frame],
10324 					frame ? state->frame[frame - 1] : NULL);
10325 }
10326 
10327 /*
10328  * Are we currently verifying the callback for an rbtree kfunc that must
10329  * be called with a lock held, or one of that callback's subprogs? If so,
10330  * no need to complain about an unreleased lock.
10331  */
10332 static bool in_rbtree_lock_required_cb(struct bpf_verifier_env *env)
10333 {
10334 	struct bpf_verifier_state *state = env->cur_state;
10335 	struct bpf_insn *insn = env->prog->insnsi;
10336 	struct bpf_func_state *callee;
10337 	int kfunc_btf_id;
10338 	u32 frame;
10339 
10340 	for (frame = state->curframe; frame; frame--) {
10341 		callee = state->frame[frame];
10342 		if (!callee->in_callback_fn)
10343 			continue;
10344 
10345 		kfunc_btf_id = insn[callee->callsite].imm;
10346 		if (is_rbtree_lock_required_kfunc(kfunc_btf_id))
10347 			return true;
10348 	}
10349 
10350 	return false;
10351 }
10352 
10353 static bool retval_range_within(struct bpf_retval_range range, const struct bpf_reg_state *reg)
10354 {
10355 	if (range.return_32bit)
10356 		return range.minval <= reg_s32_min(reg) && reg_s32_max(reg) <= range.maxval;
10357 	else
10358 		return range.minval <= reg_smin(reg) && reg_smax(reg) <= range.maxval;
10359 }
10360 
10361 static int prepare_func_exit(struct bpf_verifier_env *env, int *insn_idx)
10362 {
10363 	struct bpf_verifier_state *state = env->cur_state, *prev_st;
10364 	struct bpf_func_state *caller, *callee;
10365 	struct bpf_reg_state *r0;
10366 	bool in_callback_fn;
10367 	int err;
10368 
10369 	callee = state->frame[state->curframe];
10370 	r0 = &callee->regs[BPF_REG_0];
10371 	if (r0->type == PTR_TO_STACK) {
10372 		/* technically it's ok to return caller's stack pointer
10373 		 * (or caller's caller's pointer) back to the caller,
10374 		 * since these pointers are valid. Only current stack
10375 		 * pointer will be invalid as soon as function exits,
10376 		 * but let's be conservative
10377 		 */
10378 		verbose(env, "cannot return stack pointer to the caller\n");
10379 		return -EINVAL;
10380 	}
10381 
10382 	caller = state->frame[state->curframe - 1];
10383 	if (callee->in_callback_fn) {
10384 		if (r0->type != SCALAR_VALUE) {
10385 			verbose(env, "R0 not a scalar value\n");
10386 			return -EACCES;
10387 		}
10388 
10389 		/* we are going to rely on register's precise value */
10390 		err = mark_chain_precision(env, BPF_REG_0);
10391 		if (err)
10392 			return err;
10393 
10394 		/* enforce R0 return value range, and bpf_callback_t returns 64bit */
10395 		if (!retval_range_within(callee->callback_ret_range, r0)) {
10396 			verbose_invalid_scalar(env, r0, callee->callback_ret_range,
10397 					       "At callback return", "R0");
10398 			return -EINVAL;
10399 		}
10400 		if (!bpf_calls_callback(env, callee->callsite)) {
10401 			verifier_bug(env, "in callback at %d, callsite %d !calls_callback",
10402 				     *insn_idx, callee->callsite);
10403 			return -EFAULT;
10404 		}
10405 	} else {
10406 		/* return to the caller whatever r0 had in the callee */
10407 		bpf_diag_mod_begin(env, &caller->regs[BPF_REG_0], r0, BPF_DIAG_MOD_WRITE);
10408 		caller->regs[BPF_REG_0] = *r0;
10409 		bpf_diag_mod_end(env);
10410 	}
10411 
10412 	/* for callbacks like bpf_loop or bpf_for_each_map_elem go back to callsite,
10413 	 * there function call logic would reschedule callback visit. If iteration
10414 	 * converges is_state_visited() would prune that visit eventually.
10415 	 */
10416 	in_callback_fn = callee->in_callback_fn;
10417 	if (in_callback_fn)
10418 		*insn_idx = callee->callsite;
10419 	else
10420 		*insn_idx = callee->callsite + 1;
10421 
10422 	if (env->log.level & BPF_LOG_LEVEL) {
10423 		verbose(env, "returning from callee:\n");
10424 		print_verifier_state(env, state, callee->frameno, true);
10425 		verbose(env, "to caller at %d:\n", *insn_idx);
10426 		print_verifier_state(env, state, caller->frameno, true);
10427 	}
10428 	account_processed_insns(env, callee, caller);
10429 	/* clear everything in the callee. In case of exceptional exits using
10430 	 * bpf_throw, this will be done by copy_verifier_state for extra frames. */
10431 	free_func_state(callee);
10432 	state->frame[state->curframe--] = NULL;
10433 	invalidate_outgoing_stack_args(env, caller);
10434 
10435 	/* for callbacks widen imprecise scalars to make programs like below verify:
10436 	 *
10437 	 *   struct ctx { int i; }
10438 	 *   void cb(int idx, struct ctx *ctx) { ctx->i++; ... }
10439 	 *   ...
10440 	 *   struct ctx = { .i = 0; }
10441 	 *   bpf_loop(100, cb, &ctx, 0);
10442 	 *
10443 	 * This is similar to what is done in process_iter_next_call() for open
10444 	 * coded iterators.
10445 	 */
10446 	prev_st = in_callback_fn ? find_prev_entry(env, state, *insn_idx) : NULL;
10447 	if (prev_st) {
10448 		err = widen_imprecise_scalars(env, prev_st, state);
10449 		if (err)
10450 			return err;
10451 	}
10452 	return 0;
10453 }
10454 
10455 static int do_refine_retval_range(struct bpf_verifier_env *env,
10456 				  struct bpf_reg_state *regs, int ret_type,
10457 				  int func_id,
10458 				  struct bpf_call_arg_meta *meta)
10459 {
10460 	struct bpf_retval_range range;
10461 	struct bpf_reg_state *ret_reg = &regs[BPF_REG_0];
10462 	enum bpf_prog_type prog_type = resolve_prog_type(env->prog);
10463 
10464 	if (ret_type != RET_INTEGER)
10465 		return 0;
10466 
10467 	switch (func_id) {
10468 	case BPF_FUNC_get_stack:
10469 	case BPF_FUNC_get_task_stack:
10470 	case BPF_FUNC_probe_read_str:
10471 	case BPF_FUNC_probe_read_kernel_str:
10472 	case BPF_FUNC_probe_read_user_str:
10473 		reg_set_srange64(ret_reg, -MAX_ERRNO, meta->msize_max_value);
10474 		reg_set_srange32(ret_reg, -MAX_ERRNO, meta->msize_max_value);
10475 		reg_bounds_sync(ret_reg);
10476 		break;
10477 	case BPF_FUNC_get_smp_processor_id:
10478 		reg_set_urange64(ret_reg, 0, nr_cpu_ids - 1);
10479 		reg_set_urange32(ret_reg, 0, nr_cpu_ids - 1);
10480 		reg_bounds_sync(ret_reg);
10481 		break;
10482 	case BPF_FUNC_get_retval:
10483 		/*
10484 		 * bpf_get_retval may see arbitrary value passed by bpf_prog_run_array_cg for
10485 		 * CGROUP_GETSOCKOPT type.
10486 		 */
10487 		if (prog_type == BPF_PROG_TYPE_CGROUP_SOCKOPT &&
10488 		    env->prog->expected_attach_type == BPF_CGROUP_GETSOCKOPT)
10489 			break;
10490 
10491 		if (prog_type == BPF_PROG_TYPE_LSM &&
10492 		    env->prog->expected_attach_type == BPF_LSM_CGROUP) {
10493 			if (!env->prog->aux->attach_func_proto->type)
10494 				break;
10495 			bpf_lsm_get_retval_range(env->prog, &range);
10496 		} else {
10497 			range.minval = -MAX_ERRNO;
10498 			range.maxval = 0;
10499 		}
10500 
10501 		reg_set_srange64(ret_reg, range.minval, range.maxval);
10502 		reg_set_srange32(ret_reg, range.minval, range.maxval);
10503 		reg_bounds_sync(ret_reg);
10504 		break;
10505 	}
10506 
10507 	return reg_bounds_sanity_check(env, ret_reg, "retval");
10508 }
10509 
10510 static int
10511 record_func_map(struct bpf_verifier_env *env, struct bpf_call_arg_meta *meta,
10512 		int func_id, int insn_idx)
10513 {
10514 	struct bpf_insn_aux_data *aux = &env->insn_aux_data[insn_idx];
10515 	struct bpf_map *map = meta->map.ptr;
10516 
10517 	if (func_id != BPF_FUNC_tail_call &&
10518 	    func_id != BPF_FUNC_map_lookup_elem &&
10519 	    func_id != BPF_FUNC_map_update_elem &&
10520 	    func_id != BPF_FUNC_map_delete_elem &&
10521 	    func_id != BPF_FUNC_map_push_elem &&
10522 	    func_id != BPF_FUNC_map_pop_elem &&
10523 	    func_id != BPF_FUNC_map_peek_elem &&
10524 	    func_id != BPF_FUNC_for_each_map_elem &&
10525 	    func_id != BPF_FUNC_redirect_map &&
10526 	    func_id != BPF_FUNC_map_lookup_percpu_elem)
10527 		return 0;
10528 
10529 	if (map == NULL) {
10530 		verifier_bug(env, "expected map for helper call");
10531 		return -EFAULT;
10532 	}
10533 
10534 	/* In case of read-only, some additional restrictions
10535 	 * need to be applied in order to prevent altering the
10536 	 * state of the map from program side.
10537 	 */
10538 	if ((map->map_flags & BPF_F_RDONLY_PROG) &&
10539 	    (func_id == BPF_FUNC_map_delete_elem ||
10540 	     func_id == BPF_FUNC_map_update_elem ||
10541 	     func_id == BPF_FUNC_map_push_elem ||
10542 	     func_id == BPF_FUNC_map_pop_elem)) {
10543 		verbose(env, "write into map forbidden\n");
10544 		return -EACCES;
10545 	}
10546 
10547 	if (!aux->map_ptr_state.map_ptr)
10548 		bpf_map_ptr_store(aux, meta->map.ptr,
10549 				  !meta->map.ptr->bypass_spec_v1, false);
10550 	else if (aux->map_ptr_state.map_ptr != meta->map.ptr)
10551 		bpf_map_ptr_store(aux, meta->map.ptr,
10552 				  !meta->map.ptr->bypass_spec_v1, true);
10553 	return 0;
10554 }
10555 
10556 static int
10557 record_func_key(struct bpf_verifier_env *env, struct bpf_call_arg_meta *meta,
10558 		int func_id, int insn_idx)
10559 {
10560 	struct bpf_insn_aux_data *aux = &env->insn_aux_data[insn_idx];
10561 	struct bpf_reg_state *reg;
10562 	struct bpf_map *map = meta->map.ptr;
10563 	u64 val, max;
10564 	int err;
10565 
10566 	if (func_id != BPF_FUNC_tail_call)
10567 		return 0;
10568 	if (!map || map->map_type != BPF_MAP_TYPE_PROG_ARRAY) {
10569 		verbose(env, "expected prog array map for tail call");
10570 		return -EINVAL;
10571 	}
10572 
10573 	reg = reg_state(env, BPF_REG_3);
10574 	val = reg->var_off.value;
10575 	max = map->max_entries;
10576 
10577 	if (!(is_reg_const(reg, false) && val < max)) {
10578 		bpf_map_key_store(aux, BPF_MAP_KEY_POISON);
10579 		return 0;
10580 	}
10581 
10582 	err = mark_chain_precision(env, BPF_REG_3);
10583 	if (err)
10584 		return err;
10585 	if (bpf_map_key_unseen(aux))
10586 		bpf_map_key_store(aux, val);
10587 	else if (!bpf_map_key_poisoned(aux) &&
10588 		  bpf_map_key_immediate(aux) != val)
10589 		bpf_map_key_store(aux, BPF_MAP_KEY_POISON);
10590 	return 0;
10591 }
10592 
10593 static int check_reference_leak(struct bpf_verifier_env *env, bool exception_exit)
10594 {
10595 	struct bpf_verifier_state *state = env->cur_state;
10596 	enum bpf_prog_type type = resolve_prog_type(env->prog);
10597 	struct bpf_reg_state *reg = reg_state(env, BPF_REG_0);
10598 	bool refs_lingering = false;
10599 	int i;
10600 
10601 	if (!exception_exit && cur_func(env)->frameno)
10602 		return 0;
10603 
10604 	for (i = 0; i < state->acquired_refs; i++) {
10605 		if (state->refs[i].type != REF_TYPE_PTR)
10606 			continue;
10607 		/* Allow struct_ops programs to return a referenced kptr back to
10608 		 * kernel. Type checks are performed later in check_return_code.
10609 		 */
10610 		if (type == BPF_PROG_TYPE_STRUCT_OPS && !exception_exit &&
10611 		    reg->id == state->refs[i].id)
10612 			continue;
10613 		verbose(env, "Unreleased reference id=%d alloc_insn=%d\n",
10614 			state->refs[i].id, state->refs[i].insn_idx);
10615 		bpf_diag_leak(env, state->refs[i].id, state->refs[i].insn_idx, env->insn_idx);
10616 		refs_lingering = true;
10617 	}
10618 	return refs_lingering ? -EINVAL : 0;
10619 }
10620 
10621 static int check_resource_leak(struct bpf_verifier_env *env, bool exception_exit, bool check_lock, const char *prefix)
10622 {
10623 	int err;
10624 
10625 	if (check_lock && env->cur_state->active_locks) {
10626 		verbose(env, "%s cannot be used inside bpf_spin_lock-ed region\n", prefix);
10627 		bpf_diag_ctx_active(env, env->insn_idx, prefix, BPF_DIAG_CONTEXT_LOCK,
10628 				    "Release the BPF spin lock before this operation on every path.");
10629 		return -EINVAL;
10630 	}
10631 
10632 	err = check_reference_leak(env, exception_exit);
10633 	if (err) {
10634 		verbose(env, "%s would lead to reference leak\n", prefix);
10635 		return err;
10636 	}
10637 
10638 	if (check_lock && env->cur_state->active_irq_id) {
10639 		verbose(env, "%s cannot be used inside bpf_local_irq_save-ed region\n", prefix);
10640 		bpf_diag_ctx_active(env, env->insn_idx, prefix, BPF_DIAG_CONTEXT_IRQ,
10641 				    "Restore the saved IRQ state before this operation on every path.");
10642 		return -EINVAL;
10643 	}
10644 
10645 	if (check_lock && env->cur_state->active_rcu_locks) {
10646 		verbose(env, "%s cannot be used inside bpf_rcu_read_lock-ed region\n", prefix);
10647 		bpf_diag_ctx_active(env, env->insn_idx, prefix, BPF_DIAG_CONTEXT_RCU,
10648 				    "Call bpf_rcu_read_unlock() before this operation on every path.");
10649 		return -EINVAL;
10650 	}
10651 
10652 	if (check_lock && env->cur_state->active_preempt_locks) {
10653 		verbose(env, "%s cannot be used inside bpf_preempt_disable-ed region\n", prefix);
10654 		bpf_diag_ctx_active(
10655 			env, env->insn_idx, prefix, BPF_DIAG_CONTEXT_PREEMPT,
10656 			"Call bpf_preempt_enable() before this operation on every path.");
10657 		return -EINVAL;
10658 	}
10659 
10660 	return 0;
10661 }
10662 
10663 static int check_bpf_snprintf_call(struct bpf_verifier_env *env,
10664 				   struct bpf_reg_state *regs)
10665 {
10666 	struct bpf_reg_state *fmt_reg = &regs[BPF_REG_3];
10667 	struct bpf_reg_state *data_len_reg = &regs[BPF_REG_5];
10668 	struct bpf_map *fmt_map = fmt_reg->map_ptr;
10669 	struct bpf_bprintf_data data = {};
10670 	int err, fmt_map_off, num_args;
10671 	u64 fmt_addr;
10672 	char *fmt;
10673 
10674 	/* data must be an array of u64 */
10675 	if (data_len_reg->var_off.value % 8)
10676 		return -EINVAL;
10677 	num_args = data_len_reg->var_off.value / 8;
10678 
10679 	/* fmt being ARG_PTR_TO_CONST_STR guarantees that var_off is const
10680 	 * and map_direct_value_addr is set.
10681 	 */
10682 	fmt_map_off = fmt_reg->var_off.value;
10683 	err = fmt_map->ops->map_direct_value_addr(fmt_map, &fmt_addr,
10684 						  fmt_map_off);
10685 	if (err) {
10686 		verbose(env, "failed to retrieve map value address\n");
10687 		return -EFAULT;
10688 	}
10689 	fmt = (char *)(long)fmt_addr + fmt_map_off;
10690 
10691 	/* We are also guaranteed that fmt+fmt_map_off is NULL terminated, we
10692 	 * can focus on validating the format specifiers.
10693 	 */
10694 	err = bpf_bprintf_prepare(fmt, UINT_MAX, NULL, num_args, &data);
10695 	if (err < 0)
10696 		verbose(env, "Invalid format string\n");
10697 
10698 	return err;
10699 }
10700 
10701 static int check_get_func_ip(struct bpf_verifier_env *env)
10702 {
10703 	enum bpf_prog_type type = resolve_prog_type(env->prog);
10704 	int func_id = BPF_FUNC_get_func_ip;
10705 
10706 	if (type == BPF_PROG_TYPE_TRACING) {
10707 		if (!bpf_prog_has_trampoline(env->prog)) {
10708 			verbose(env, "func %s#%d supported only for fentry/fexit/fsession/fmod_ret programs\n",
10709 				func_id_name(func_id), func_id);
10710 			return -ENOTSUPP;
10711 		}
10712 		return 0;
10713 	} else if (type == BPF_PROG_TYPE_KPROBE) {
10714 		return 0;
10715 	}
10716 
10717 	verbose(env, "func %s#%d not supported for program type %d\n",
10718 		func_id_name(func_id), func_id, type);
10719 	return -ENOTSUPP;
10720 }
10721 
10722 static struct bpf_insn_aux_data *cur_aux(const struct bpf_verifier_env *env)
10723 {
10724 	return &env->insn_aux_data[env->insn_idx];
10725 }
10726 
10727 /* Returns 1 if R4 is a known zero, 0 if it is not, a negative errno on error. */
10728 static int loop_flag_is_zero(struct bpf_verifier_env *env)
10729 {
10730 	struct bpf_reg_state *reg = reg_state(env, BPF_REG_4);
10731 	int err;
10732 
10733 	if (!bpf_register_is_null(reg))
10734 		return 0;
10735 
10736 	err = mark_chain_precision(env, BPF_REG_4);
10737 	if (err)
10738 		return err;
10739 	return 1;
10740 }
10741 
10742 static int update_loop_inline_state(struct bpf_verifier_env *env, u32 subprogno)
10743 {
10744 	struct bpf_loop_inline_state *state = &cur_aux(env)->loop_inline_state;
10745 	int flag_is_zero;
10746 
10747 	if (!state->initialized) {
10748 		flag_is_zero = loop_flag_is_zero(env);
10749 		if (flag_is_zero < 0)
10750 			return flag_is_zero;
10751 		state->initialized = 1;
10752 		state->fit_for_inline = flag_is_zero;
10753 		state->callback_subprogno = subprogno;
10754 		return 0;
10755 	}
10756 
10757 	if (!state->fit_for_inline)
10758 		return 0;
10759 
10760 	flag_is_zero = loop_flag_is_zero(env);
10761 	if (flag_is_zero < 0)
10762 		return flag_is_zero;
10763 	state->fit_for_inline = (flag_is_zero &&
10764 				 state->callback_subprogno == subprogno);
10765 	return 0;
10766 }
10767 
10768 /* Returns whether or not the given map can potentially elide
10769  * lookup return value nullness check. This is possible if the key
10770  * is statically known.
10771  */
10772 static bool can_elide_value_nullness(const struct bpf_map *map)
10773 {
10774 	if (map->map_flags & BPF_F_INNER_MAP)
10775 		return false;
10776 
10777 	switch (map->map_type) {
10778 	case BPF_MAP_TYPE_ARRAY:
10779 	case BPF_MAP_TYPE_PERCPU_ARRAY:
10780 		return true;
10781 	default:
10782 		return false;
10783 	}
10784 }
10785 
10786 int bpf_get_helper_proto(struct bpf_verifier_env *env, int func_id,
10787 			 const struct bpf_func_proto **ptr)
10788 {
10789 	if (func_id < 0 || func_id >= __BPF_FUNC_MAX_ID)
10790 		return -ERANGE;
10791 
10792 	if (!env->ops->get_func_proto)
10793 		return -EINVAL;
10794 
10795 	*ptr = env->ops->get_func_proto(func_id, env->prog);
10796 	return *ptr && (*ptr)->func ? 0 : -EINVAL;
10797 }
10798 
10799 /* Check if we're in a sleepable context. */
10800 static inline bool in_sleepable_context(struct bpf_verifier_env *env)
10801 {
10802 	return !in_rcu_cs(env);
10803 }
10804 
10805 static const char *non_sleepable_context_description(struct bpf_verifier_env *env)
10806 {
10807 	if (env->cur_state->active_rcu_locks)
10808 		return "rcu_read_lock region";
10809 	if (env->cur_state->active_preempt_locks)
10810 		return "non-preemptible region";
10811 	if (env->cur_state->active_irq_id)
10812 		return "IRQ-disabled region";
10813 	if (env->cur_state->active_locks)
10814 		return "lock region";
10815 	return "non-sleepable prog";
10816 }
10817 
10818 static int release_reg(struct bpf_verifier_env *env, struct bpf_reg_state *reg,
10819 		       bool convert_rcu, bool release_dynptr)
10820 {
10821 	int err = -EINVAL;
10822 
10823 	if (bpf_register_is_null(reg))
10824 		return 0;
10825 
10826 	if (release_dynptr)
10827 		err = unmark_stack_slots_dynptr(env, reg);
10828 	else if (convert_rcu)
10829 		err = ref_convert_alloc_rcu_protected(env, reg->id);
10830 	else if (reg_is_referenced(env, reg))
10831 		err = release_reference(env, reg->id);
10832 
10833 	return err;
10834 }
10835 
10836 static int check_helper_call(struct bpf_verifier_env *env, struct bpf_insn *insn,
10837 			     int *insn_idx_p)
10838 {
10839 	enum bpf_prog_type prog_type = resolve_prog_type(env->prog);
10840 	bool returns_cpu_specific_alloc_ptr = false;
10841 	const struct bpf_func_proto *fn = NULL;
10842 	enum bpf_return_type ret_type;
10843 	enum bpf_type_flag ret_flag;
10844 	struct bpf_reg_state *regs;
10845 	struct bpf_call_arg_meta meta;
10846 	const char *operation;
10847 	int insn_idx = *insn_idx_p;
10848 	bool changes_data;
10849 	int i, err, func_id;
10850 
10851 	/* find function prototype */
10852 	func_id = insn->imm;
10853 	err = bpf_get_helper_proto(env, insn->imm, &fn);
10854 	if (err == -ERANGE) {
10855 		verbose(env, "invalid func %s#%d\n", func_id_name(func_id), func_id);
10856 		return -EINVAL;
10857 	}
10858 
10859 	if (err) {
10860 		verbose(env, "program of this type cannot use helper %s#%d\n",
10861 			func_id_name(func_id), func_id);
10862 		operation = bpf_diag_fmt(env, "helper %s#%d", func_id_name(func_id), func_id);
10863 		bpf_diag_policy(
10864 			env, insn_idx, operation, "this program type does not allow the helper",
10865 			"Use a helper allowed for this program type, or move the logic to a compatible program type.");
10866 		return err;
10867 	}
10868 
10869 	/* eBPF programs must be GPL compatible to use GPL-ed functions */
10870 	if (!env->prog->gpl_compatible && fn->gpl_only) {
10871 		verbose(env, "cannot call GPL-restricted function from non-GPL compatible program\n");
10872 		operation = bpf_diag_fmt(env, "helper %s#%d", func_id_name(func_id), func_id);
10873 		bpf_diag_policy(
10874 			env, insn_idx, operation,
10875 			"this helper is restricted to GPL-compatible programs",
10876 			"Use a GPL-compatible license, or replace the helper with one that is available to non-GPL programs.");
10877 		return -EINVAL;
10878 	}
10879 
10880 	if (fn->allowed && !fn->allowed(env->prog)) {
10881 		verbose(env, "helper call is not allowed in probe\n");
10882 		operation = bpf_diag_fmt(env, "helper %s#%d", func_id_name(func_id), func_id);
10883 		bpf_diag_policy(
10884 			env, insn_idx, operation,
10885 			"the helper-specific policy callback rejected this program",
10886 			"Use the helper only from an allowed attach point or program configuration.");
10887 		return -EINVAL;
10888 	}
10889 
10890 	/* With LD_ABS/IND some JITs save/restore skb from r1. */
10891 	changes_data = bpf_helper_changes_pkt_data(func_id);
10892 	if (changes_data && fn->arg1_type != ARG_PTR_TO_CTX) {
10893 		verifier_bug(env, "func %s#%d: r1 != ctx", func_id_name(func_id), func_id);
10894 		return -EFAULT;
10895 	}
10896 
10897 	memset(&meta, 0, sizeof(meta));
10898 
10899 	err = check_func_proto(fn, &meta);
10900 	if (err) {
10901 		verifier_bug(env, "incorrect func proto %s#%d", func_id_name(func_id), func_id);
10902 		return err;
10903 	}
10904 
10905 	if (fn->might_sleep && !in_sleepable_context(env)) {
10906 		verbose(env, "sleepable helper %s#%d in %s\n", func_id_name(func_id), func_id,
10907 			non_sleepable_context_description(env));
10908 		operation = bpf_diag_fmt(env, "sleepable helper %s#%d",
10909 					 func_id_name(func_id), func_id);
10910 		bpf_diag_ctx_forbidden(env, insn_idx, operation,
10911 			"Move the helper call outside the critical section, or use a non-sleepable helper.");
10912 		return -EINVAL;
10913 	}
10914 
10915 	/* Track non-sleepable context for helpers. */
10916 	if (!in_sleepable_context(env))
10917 		env->insn_aux_data[insn_idx].non_sleepable = true;
10918 
10919 	meta.func_id = func_id;
10920 	meta.fn = fn;
10921 	/* check args */
10922 	for (i = 0; i < MAX_BPF_FUNC_REG_ARGS; i++) {
10923 		err = check_func_arg(env, i, &meta, insn_idx);
10924 		if (err)
10925 			return err;
10926 	}
10927 
10928 	err = record_func_map(env, &meta, func_id, insn_idx);
10929 	if (err)
10930 		return err;
10931 
10932 	err = record_func_key(env, &meta, func_id, insn_idx);
10933 	if (err)
10934 		return err;
10935 
10936 	regs = cur_regs(env);
10937 
10938 	/* Mark slots with STACK_MISC in case of raw mode, stack offset
10939 	 * is inferred from register state.
10940 	 */
10941 	for (i = 0; i < meta.arg_raw_mem.size; i++) {
10942 		err = check_mem_access(env, insn_idx, regs + meta.arg_raw_mem.regno,
10943 				       argno_from_reg(meta.arg_raw_mem.regno), i, BPF_B,
10944 				       BPF_WRITE, -1, false, false);
10945 		if (err)
10946 			return err;
10947 	}
10948 
10949 	if (meta.release_regno) {
10950 		struct bpf_reg_state *reg = &regs[meta.release_regno];
10951 		bool convert_rcu = (func_id == BPF_FUNC_kptr_xchg) && in_rcu_cs(env) &&
10952 				   (reg->type & MEM_ALLOC) && (reg->type & MEM_PERCPU);
10953 
10954 		err = release_reg(env, reg, convert_rcu, !!meta.dynptr.id);
10955 		if (err)
10956 			return err;
10957 	}
10958 
10959 	switch (func_id) {
10960 	case BPF_FUNC_tail_call:
10961 		err = check_resource_leak(env, false, true, "tail_call");
10962 		if (err)
10963 			return err;
10964 		break;
10965 	case BPF_FUNC_get_local_storage:
10966 		/* check that flags argument in get_local_storage(map, flags) is 0,
10967 		 * this is required because get_local_storage() can't return an error.
10968 		 */
10969 		if (!bpf_register_is_null(&regs[BPF_REG_2])) {
10970 			verbose(env, "get_local_storage() doesn't support non-zero flags\n");
10971 			return -EINVAL;
10972 		}
10973 		err = mark_chain_precision(env, BPF_REG_2);
10974 		if (err)
10975 			return err;
10976 		break;
10977 	case BPF_FUNC_for_each_map_elem:
10978 		err = push_callback_call(env, insn, insn_idx, meta.subprogno,
10979 					 set_map_elem_callback_state);
10980 		break;
10981 	case BPF_FUNC_timer_set_callback:
10982 		err = push_callback_call(env, insn, insn_idx, meta.subprogno,
10983 					 set_timer_callback_state);
10984 		break;
10985 	case BPF_FUNC_find_vma:
10986 		err = push_callback_call(env, insn, insn_idx, meta.subprogno,
10987 					 set_find_vma_callback_state);
10988 		break;
10989 	case BPF_FUNC_snprintf:
10990 		err = check_bpf_snprintf_call(env, regs);
10991 		break;
10992 	case BPF_FUNC_loop:
10993 		err = update_loop_inline_state(env, meta.subprogno);
10994 		if (err)
10995 			return err;
10996 		/* Verifier relies on R1 value to determine if bpf_loop() iteration
10997 		 * is finished, thus mark it precise.
10998 		 */
10999 		err = mark_chain_precision(env, BPF_REG_1);
11000 		if (err)
11001 			return err;
11002 		if (cur_func(env)->callback_depth < reg_umax(&regs[BPF_REG_1])) {
11003 			err = push_callback_call(env, insn, insn_idx, meta.subprogno,
11004 						 set_loop_callback_state);
11005 		} else {
11006 			cur_func(env)->callback_depth = 0;
11007 			if (env->log.level & BPF_LOG_LEVEL2)
11008 				verbose(env, "frame%d bpf_loop iteration limit reached\n",
11009 					env->cur_state->curframe);
11010 		}
11011 		break;
11012 	case BPF_FUNC_dynptr_from_mem:
11013 		if (regs[BPF_REG_1].type != PTR_TO_MAP_VALUE) {
11014 			verbose(env, "Unsupported reg type %s for bpf_dynptr_from_mem data\n",
11015 				reg_type_str(env, regs[BPF_REG_1].type));
11016 			return -EACCES;
11017 		}
11018 		break;
11019 	case BPF_FUNC_set_retval:
11020 	{
11021 		struct bpf_retval_range range = {
11022 			.minval = -MAX_ERRNO,
11023 			.maxval = 0,
11024 			.return_32bit = true
11025 		};
11026 		struct bpf_reg_state *r1 = &regs[BPF_REG_1];
11027 
11028 		if (r1->type != SCALAR_VALUE) {
11029 			verbose(env, "R1 is not a scalar\n");
11030 			return -EINVAL;
11031 		}
11032 
11033 		/* CGROUP_GETSOCKOPT is allowed to return arbitrary value */
11034 		if (prog_type == BPF_PROG_TYPE_CGROUP_SOCKOPT &&
11035 		    env->prog->expected_attach_type == BPF_CGROUP_GETSOCKOPT)
11036 			break;
11037 
11038 		if (prog_type == BPF_PROG_TYPE_LSM &&
11039 		    env->prog->expected_attach_type == BPF_LSM_CGROUP) {
11040 			if (!env->prog->aux->attach_func_proto->type) {
11041 				/* Make sure programs that attach to void
11042 				 * hooks don't try to modify return value.
11043 				 */
11044 				verbose(env, "BPF_LSM_CGROUP that attach to void LSM hooks can't modify return value!\n");
11045 				return -EINVAL;
11046 			}
11047 			bpf_lsm_get_retval_range(env->prog, &range);
11048 		}
11049 
11050 		err = mark_chain_precision(env, BPF_REG_1);
11051 		if (err)
11052 			return err;
11053 
11054 		if (!retval_range_within(range, r1)) {
11055 			verbose_invalid_scalar(env, r1, range, "At bpf_set_retval", "R1");
11056 			return -EINVAL;
11057 		}
11058 
11059 		break;
11060 	}
11061 	case BPF_FUNC_dynptr_write:
11062 	{
11063 		enum bpf_dynptr_type dynptr_type = meta.dynptr.type;
11064 
11065 		if (dynptr_type == BPF_DYNPTR_TYPE_INVALID)
11066 			return -EFAULT;
11067 
11068 		if (dynptr_type == BPF_DYNPTR_TYPE_SKB ||
11069 		    dynptr_type == BPF_DYNPTR_TYPE_SKB_META)
11070 			/* this will trigger clear_all_pkt_pointers(), which will
11071 			 * invalidate all dynptr slices associated with the skb
11072 			 */
11073 			changes_data = true;
11074 
11075 		break;
11076 	}
11077 	case BPF_FUNC_per_cpu_ptr:
11078 	case BPF_FUNC_this_cpu_ptr:
11079 	{
11080 		struct bpf_reg_state *reg = &regs[BPF_REG_1];
11081 		const struct btf_type *type;
11082 
11083 		if (reg->type & MEM_RCU) {
11084 			type = btf_type_by_id(reg->btf, reg->btf_id);
11085 			if (!type || !btf_type_is_struct(type)) {
11086 				verbose(env, "Helper has invalid btf/btf_id in R1\n");
11087 				return -EFAULT;
11088 			}
11089 			returns_cpu_specific_alloc_ptr = true;
11090 			env->insn_aux_data[insn_idx].call_with_percpu_alloc_ptr = true;
11091 		}
11092 		break;
11093 	}
11094 	case BPF_FUNC_user_ringbuf_drain:
11095 		err = push_callback_call(env, insn, insn_idx, meta.subprogno,
11096 					 set_user_ringbuf_callback_state);
11097 		break;
11098 	}
11099 
11100 	if (err)
11101 		return err;
11102 
11103 	/* reset caller saved regs */
11104 	bpf_diag_record_caller_saved(env, regs);
11105 	bpf_diag_mod_begin(env, &regs[BPF_REG_0], NULL, BPF_DIAG_MOD_WRITE);
11106 	for (i = 0; i < CALLER_SAVED_REGS; i++) {
11107 		bpf_mark_reg_not_init(env, &regs[caller_saved[i]]);
11108 		check_reg_arg(env, caller_saved[i], DST_OP_NO_MARK);
11109 	}
11110 	invalidate_outgoing_stack_args(env, cur_func(env));
11111 
11112 	/* update return register (already marked as written above) */
11113 	ret_type = fn->ret_type;
11114 	ret_flag = type_flag(ret_type);
11115 
11116 	switch (base_type(ret_type)) {
11117 	case RET_INTEGER:
11118 		/* sets type to SCALAR_VALUE */
11119 		mark_reg_unknown(env, regs, BPF_REG_0);
11120 		break;
11121 	case RET_VOID:
11122 		regs[BPF_REG_0].type = NOT_INIT;
11123 		break;
11124 	case RET_PTR_TO_MAP_VALUE:
11125 		/* There is no offset yet applied, variable or fixed */
11126 		mark_reg_known_zero(env, regs, BPF_REG_0);
11127 		/* remember map_ptr, so that check_map_access()
11128 		 * can check 'value_size' boundary of memory access
11129 		 * to map element returned from bpf_map_lookup_elem()
11130 		 */
11131 		if (meta.map.ptr == NULL) {
11132 			verifier_bug(env, "unexpected null map_ptr");
11133 			return -EFAULT;
11134 		}
11135 
11136 		if (func_id == BPF_FUNC_map_lookup_elem &&
11137 		    can_elide_value_nullness(meta.map.ptr) &&
11138 		    meta.const_map_key >= 0 &&
11139 		    meta.const_map_key < meta.map.ptr->max_entries)
11140 			ret_flag &= ~PTR_MAYBE_NULL;
11141 
11142 		regs[BPF_REG_0].map_ptr = meta.map.ptr;
11143 		regs[BPF_REG_0].map_uid = meta.map.uid;
11144 		regs[BPF_REG_0].type = PTR_TO_MAP_VALUE | ret_flag;
11145 		if (type_may_be_null(ret_flag) ||
11146 		    btf_record_has_field(meta.map.ptr->record, BPF_SPIN_LOCK | BPF_RES_SPIN_LOCK)) {
11147 			regs[BPF_REG_0].id = ++env->id_gen;
11148 		}
11149 		/* requires regs[BPF_REG_0].id to be set because of the map-in-map case */
11150 		refine_map_lookup_value(&regs[BPF_REG_0]);
11151 		break;
11152 	case RET_PTR_TO_SOCKET:
11153 		mark_reg_known_zero(env, regs, BPF_REG_0);
11154 		regs[BPF_REG_0].type = PTR_TO_SOCKET | ret_flag;
11155 		break;
11156 	case RET_PTR_TO_SOCK_COMMON:
11157 		mark_reg_known_zero(env, regs, BPF_REG_0);
11158 		regs[BPF_REG_0].type = PTR_TO_SOCK_COMMON | ret_flag;
11159 		break;
11160 	case RET_PTR_TO_TCP_SOCK:
11161 		mark_reg_known_zero(env, regs, BPF_REG_0);
11162 		regs[BPF_REG_0].type = PTR_TO_TCP_SOCK | ret_flag;
11163 		break;
11164 	case RET_PTR_TO_MEM:
11165 		mark_reg_known_zero(env, regs, BPF_REG_0);
11166 		regs[BPF_REG_0].type = PTR_TO_MEM | ret_flag;
11167 		regs[BPF_REG_0].mem_size = meta.ret_mem.size;
11168 		break;
11169 	case RET_PTR_TO_MEM_OR_BTF_ID:
11170 	{
11171 		const struct btf_type *t;
11172 
11173 		mark_reg_known_zero(env, regs, BPF_REG_0);
11174 		t = btf_type_skip_modifiers(meta.ret_btf, meta.ret_btf_id, NULL);
11175 		if (!btf_type_is_struct(t)) {
11176 			u32 tsize;
11177 			const struct btf_type *ret;
11178 			const char *tname;
11179 
11180 			/* resolve the type size of ksym. */
11181 			ret = btf_resolve_size(meta.ret_btf, t, &tsize);
11182 			if (IS_ERR(ret)) {
11183 				tname = btf_name_by_offset(meta.ret_btf, t->name_off);
11184 				verbose(env, "unable to resolve the size of type '%s': %ld\n",
11185 					tname, PTR_ERR(ret));
11186 				return -EINVAL;
11187 			}
11188 			regs[BPF_REG_0].type = PTR_TO_MEM | ret_flag;
11189 			regs[BPF_REG_0].mem_size = tsize;
11190 		} else {
11191 			if (returns_cpu_specific_alloc_ptr) {
11192 				regs[BPF_REG_0].type = PTR_TO_BTF_ID | MEM_ALLOC | MEM_RCU;
11193 			} else {
11194 				/* MEM_RDONLY may be carried from ret_flag, but it
11195 				 * doesn't apply on PTR_TO_BTF_ID. Fold it, otherwise
11196 				 * it will confuse the check of PTR_TO_BTF_ID in
11197 				 * check_mem_access().
11198 				 */
11199 				ret_flag &= ~MEM_RDONLY;
11200 				regs[BPF_REG_0].type = PTR_TO_BTF_ID | ret_flag;
11201 			}
11202 
11203 			regs[BPF_REG_0].btf = meta.ret_btf;
11204 			regs[BPF_REG_0].btf_id = meta.ret_btf_id;
11205 		}
11206 		break;
11207 	}
11208 	case RET_PTR_TO_BTF_ID:
11209 	{
11210 		struct btf *ret_btf;
11211 		int ret_btf_id;
11212 
11213 		mark_reg_known_zero(env, regs, BPF_REG_0);
11214 		regs[BPF_REG_0].type = PTR_TO_BTF_ID | ret_flag;
11215 		if (func_id == BPF_FUNC_kptr_xchg) {
11216 			ret_btf = meta.kptr_field->kptr.btf;
11217 			ret_btf_id = meta.kptr_field->kptr.btf_id;
11218 			if (!btf_is_kernel(ret_btf)) {
11219 				regs[BPF_REG_0].type |= MEM_ALLOC;
11220 				if (meta.kptr_field->type == BPF_KPTR_PERCPU)
11221 					regs[BPF_REG_0].type |= MEM_PERCPU;
11222 			}
11223 		} else {
11224 			if (fn->ret_btf_id == BPF_PTR_POISON) {
11225 				verifier_bug(env, "func %s has non-overwritten BPF_PTR_POISON return type",
11226 					     func_id_name(func_id));
11227 				return -EFAULT;
11228 			}
11229 			ret_btf = btf_vmlinux;
11230 			ret_btf_id = *fn->ret_btf_id;
11231 		}
11232 		if (ret_btf_id == 0) {
11233 			verbose(env, "invalid return type %u of func %s#%d\n",
11234 				base_type(ret_type), func_id_name(func_id),
11235 				func_id);
11236 			return -EINVAL;
11237 		}
11238 		regs[BPF_REG_0].btf = ret_btf;
11239 		regs[BPF_REG_0].btf_id = ret_btf_id;
11240 		break;
11241 	}
11242 	default:
11243 		verbose(env, "unknown return type %u of func %s#%d\n",
11244 			base_type(ret_type), func_id_name(func_id), func_id);
11245 		return -EINVAL;
11246 	}
11247 
11248 	if (type_may_be_null(regs[BPF_REG_0].type) && !regs[BPF_REG_0].id)
11249 		regs[BPF_REG_0].id = ++env->id_gen;
11250 
11251 	if (is_ptr_cast_function(func_id) &&
11252 	    find_reference_state(env->cur_state, meta.ref_obj.id)) {
11253 		struct bpf_verifier_state *branch;
11254 		struct bpf_reg_state *r0;
11255 
11256 		err = validate_ref_obj(env, &meta.ref_obj);
11257 		if (err)
11258 			return err;
11259 
11260 		bpf_diag_mod_end(env);
11261 
11262 		/*
11263 		 * In order for a release of any of the original or cast pointers
11264 		 * to invalidate all other pointers, reuse the same reference id for
11265 		 * the cast result.
11266 		 * This reference id can't be used for nullness propagation,
11267 		 * as cast might return NULL for a non-NULL input.
11268 		 * Hence, explore the NULL case as a separate branch.
11269 		 */
11270 		branch = push_stack(env, env->insn_idx + 1, env->insn_idx, false);
11271 		if (IS_ERR(branch))
11272 			return PTR_ERR(branch);
11273 
11274 		r0 = &branch->frame[branch->curframe]->regs[BPF_REG_0];
11275 		__mark_reg_known_zero(r0);
11276 		r0->type = SCALAR_VALUE;
11277 
11278 		bpf_diag_mod_begin(env, &regs[BPF_REG_0], NULL, BPF_DIAG_MOD_WRITE);
11279 		regs[BPF_REG_0].type &= ~PTR_MAYBE_NULL;
11280 		regs[BPF_REG_0].id = meta.ref_obj.id;
11281 	} else if (is_acquire_function(func_id, meta.map.ptr)) {
11282 		int id = acquire_reference(env, insn_idx, 0);
11283 
11284 		if (id < 0)
11285 			return id;
11286 
11287 		regs[BPF_REG_0].id = id;
11288 	}
11289 
11290 	if (func_id == BPF_FUNC_dynptr_data)
11291 		regs[BPF_REG_0].parent_id = meta.dynptr.id;
11292 
11293 	err = do_refine_retval_range(env, regs, fn->ret_type, func_id, &meta);
11294 	if (err)
11295 		return err;
11296 
11297 	bpf_diag_mod_end(env);
11298 
11299 	err = check_map_func_compatibility(env, meta.map.ptr, func_id);
11300 	if (err)
11301 		return err;
11302 
11303 	if ((func_id == BPF_FUNC_get_stack ||
11304 	     func_id == BPF_FUNC_get_task_stack) &&
11305 	    !env->prog->has_callchain_buf) {
11306 		const char *err_str;
11307 
11308 #ifdef CONFIG_PERF_EVENTS
11309 		err = get_callchain_buffers(sysctl_perf_event_max_stack);
11310 		err_str = "cannot get callchain buffer for func %s#%d\n";
11311 #else
11312 		err = -ENOTSUPP;
11313 		err_str = "func %s#%d not supported without CONFIG_PERF_EVENTS\n";
11314 #endif
11315 		if (err) {
11316 			verbose(env, err_str, func_id_name(func_id), func_id);
11317 			return err;
11318 		}
11319 
11320 		env->prog->has_callchain_buf = true;
11321 	}
11322 
11323 	if (func_id == BPF_FUNC_get_stackid || func_id == BPF_FUNC_get_stack)
11324 		env->prog->call_get_stack = true;
11325 
11326 	if (func_id == BPF_FUNC_get_func_ip) {
11327 		if (check_get_func_ip(env))
11328 			return -ENOTSUPP;
11329 		env->prog->call_get_func_ip = true;
11330 	}
11331 
11332 	if (func_id == BPF_FUNC_tail_call) {
11333 		if (env->cur_state->curframe) {
11334 			struct bpf_verifier_state *branch;
11335 
11336 			/*
11337 			 * A taken tail call is modeled as a return from the current
11338 			 * frame. A callback frame cannot be left that way because
11339 			 * prepare_func_exit() would apply its return contract to the
11340 			 * unknown R0 synthesized below. Stack-depth validation rejects
11341 			 * this construct anyway.
11342 			 */
11343 			if (cur_func(env)->in_callback_fn) {
11344 				verbose(env, "cannot tail call within callback\n");
11345 				return -EINVAL;
11346 			}
11347 			mark_reg_scratched(env, BPF_REG_0);
11348 			branch = push_stack(env, env->insn_idx + 1, env->insn_idx, false);
11349 			if (IS_ERR(branch))
11350 				return PTR_ERR(branch);
11351 			clear_all_pkt_pointers(env);
11352 			mark_reg_unknown(env, regs, BPF_REG_0);
11353 			err = prepare_func_exit(env, &env->insn_idx);
11354 			if (err)
11355 				return err;
11356 			env->insn_idx--;
11357 		} else {
11358 			changes_data = false;
11359 		}
11360 	}
11361 
11362 	if (changes_data)
11363 		clear_all_pkt_pointers(env);
11364 	return 0;
11365 }
11366 
11367 static bool is_kfunc_acquire(struct bpf_call_arg_meta *meta)
11368 {
11369 	return meta->kfunc_flags & KF_ACQUIRE;
11370 }
11371 
11372 static bool is_kfunc_release(struct bpf_call_arg_meta *meta)
11373 {
11374 	return meta->kfunc_flags & KF_RELEASE;
11375 }
11376 
11377 static bool is_kfunc_destructive(struct bpf_call_arg_meta *meta)
11378 {
11379 	return meta->kfunc_flags & KF_DESTRUCTIVE;
11380 }
11381 
11382 static bool is_kfunc_perfmon(struct bpf_call_arg_meta *meta)
11383 {
11384 	return meta->kfunc_flags & KF_PERFMON;
11385 }
11386 
11387 static bool is_kfunc_rcu(struct bpf_call_arg_meta *meta)
11388 {
11389 	return meta->kfunc_flags & KF_RCU;
11390 }
11391 
11392 static bool is_kfunc_rcu_protected(struct bpf_call_arg_meta *meta)
11393 {
11394 	return meta->kfunc_flags & KF_RCU_PROTECTED;
11395 }
11396 
11397 static bool is_kfunc_arg_mem_size(const struct btf *btf,
11398 				  const struct btf_param *arg)
11399 {
11400 	const struct btf_type *t;
11401 
11402 	t = btf_type_skip_modifiers(btf, arg->type, NULL);
11403 	if (!btf_type_is_scalar(t))
11404 		return false;
11405 
11406 	return btf_param_match_suffix(btf, arg, "__sz");
11407 }
11408 
11409 static bool is_kfunc_arg_const_mem_size(const struct btf *btf,
11410 					const struct btf_param *arg)
11411 {
11412 	const struct btf_type *t;
11413 
11414 	t = btf_type_skip_modifiers(btf, arg->type, NULL);
11415 	if (!btf_type_is_scalar(t))
11416 		return false;
11417 
11418 	return btf_param_match_suffix(btf, arg, "__szk");
11419 }
11420 
11421 static bool is_kfunc_arg_constant(const struct btf *btf, const struct btf_param *arg)
11422 {
11423 	return btf_param_match_suffix(btf, arg, "__k");
11424 }
11425 
11426 static bool is_kfunc_arg_ignore(const struct btf *btf, const struct btf_param *arg)
11427 {
11428 	return btf_param_match_suffix(btf, arg, "__ign");
11429 }
11430 
11431 static bool is_kfunc_arg_map(const struct btf *btf, const struct btf_param *arg)
11432 {
11433 	return btf_param_match_suffix(btf, arg, "__map");
11434 }
11435 
11436 static bool is_kfunc_arg_const_map(const struct btf *btf, const struct btf_param *arg)
11437 {
11438 	return btf_param_match_suffix(btf, arg, "__const_map");
11439 }
11440 
11441 static bool is_kfunc_arg_alloc_obj(const struct btf *btf, const struct btf_param *arg)
11442 {
11443 	return btf_param_match_suffix(btf, arg, "__alloc");
11444 }
11445 
11446 static bool is_kfunc_arg_uninit(const struct btf *btf, const struct btf_param *arg)
11447 {
11448 	return btf_param_match_suffix(btf, arg, "__uninit");
11449 }
11450 
11451 static bool is_kfunc_arg_refcounted_kptr(const struct btf *btf, const struct btf_param *arg)
11452 {
11453 	return btf_param_match_suffix(btf, arg, "__refcounted_kptr");
11454 }
11455 
11456 static bool is_kfunc_arg_nullable(const struct btf *btf, const struct btf_param *arg)
11457 {
11458 	return btf_param_match_suffix(btf, arg, "__nullable") ||
11459 	       btf_param_match_suffix(btf, arg, "__arena");
11460 }
11461 
11462 static bool is_kfunc_arg_nonown_allowed(const struct btf *btf, const struct btf_param *arg)
11463 {
11464 	return btf_param_match_suffix(btf, arg, "__nonown_allowed");
11465 }
11466 
11467 static bool is_kfunc_arg_const_str(const struct btf *btf, const struct btf_param *arg)
11468 {
11469 	return btf_param_match_suffix(btf, arg, "__str");
11470 }
11471 
11472 static bool is_kfunc_arg_irq_flag(const struct btf *btf, const struct btf_param *arg)
11473 {
11474 	return btf_param_match_suffix(btf, arg, "__irq_flag");
11475 }
11476 
11477 static bool is_kfunc_arg_arena(const struct btf *btf, const struct btf_param *arg)
11478 {
11479 	return btf_param_match_suffix(btf, arg, "__arena__nullable") ||
11480 	       btf_param_match_suffix(btf, arg, "__arena");
11481 }
11482 
11483 static bool is_kfunc_arg_scalar_with_name(const struct btf *btf,
11484 					  const struct btf_param *arg,
11485 					  const char *name)
11486 {
11487 	int len, target_len = strlen(name);
11488 	const char *param_name;
11489 
11490 	param_name = btf_name_by_offset(btf, arg->name_off);
11491 	if (str_is_empty(param_name))
11492 		return false;
11493 	len = strlen(param_name);
11494 	if (len != target_len)
11495 		return false;
11496 	if (strcmp(param_name, name))
11497 		return false;
11498 
11499 	return true;
11500 }
11501 
11502 enum {
11503 	KF_ARG_DYNPTR_ID,
11504 	KF_ARG_LIST_HEAD_ID,
11505 	KF_ARG_LIST_NODE_ID,
11506 	KF_ARG_RB_ROOT_ID,
11507 	KF_ARG_RB_NODE_ID,
11508 	KF_ARG_WORKQUEUE_ID,
11509 	KF_ARG_RES_SPIN_LOCK_ID,
11510 	KF_ARG_TASK_WORK_ID,
11511 	KF_ARG_PROG_AUX_ID,
11512 	KF_ARG_TIMER_ID
11513 };
11514 
11515 BTF_ID_LIST(kf_arg_btf_ids)
11516 BTF_ID(struct, bpf_dynptr)
11517 BTF_ID(struct, bpf_list_head)
11518 BTF_ID(struct, bpf_list_node)
11519 BTF_ID(struct, bpf_rb_root)
11520 BTF_ID(struct, bpf_rb_node)
11521 BTF_ID(struct, bpf_wq)
11522 BTF_ID(struct, bpf_res_spin_lock)
11523 BTF_ID(struct, bpf_task_work)
11524 BTF_ID(struct, bpf_prog_aux)
11525 BTF_ID(struct, bpf_timer)
11526 
11527 static bool __is_kfunc_ptr_arg_type(const struct btf *btf,
11528 				    const struct btf_param *arg, int type)
11529 {
11530 	const struct btf_type *t;
11531 	u32 res_id;
11532 
11533 	t = btf_type_skip_modifiers(btf, arg->type, NULL);
11534 	if (!t)
11535 		return false;
11536 	if (!btf_type_is_ptr(t))
11537 		return false;
11538 	t = btf_type_skip_modifiers(btf, t->type, &res_id);
11539 	if (!t)
11540 		return false;
11541 	return btf_types_are_same(btf, res_id, btf_vmlinux, kf_arg_btf_ids[type]);
11542 }
11543 
11544 static bool is_kfunc_arg_dynptr(const struct btf *btf, const struct btf_param *arg)
11545 {
11546 	return __is_kfunc_ptr_arg_type(btf, arg, KF_ARG_DYNPTR_ID);
11547 }
11548 
11549 static bool is_kfunc_arg_list_head(const struct btf *btf, const struct btf_param *arg)
11550 {
11551 	return __is_kfunc_ptr_arg_type(btf, arg, KF_ARG_LIST_HEAD_ID);
11552 }
11553 
11554 static bool is_kfunc_arg_list_node(const struct btf *btf, const struct btf_param *arg)
11555 {
11556 	return __is_kfunc_ptr_arg_type(btf, arg, KF_ARG_LIST_NODE_ID);
11557 }
11558 
11559 static bool is_kfunc_arg_rbtree_root(const struct btf *btf, const struct btf_param *arg)
11560 {
11561 	return __is_kfunc_ptr_arg_type(btf, arg, KF_ARG_RB_ROOT_ID);
11562 }
11563 
11564 static bool is_kfunc_arg_rbtree_node(const struct btf *btf, const struct btf_param *arg)
11565 {
11566 	return __is_kfunc_ptr_arg_type(btf, arg, KF_ARG_RB_NODE_ID);
11567 }
11568 
11569 static bool is_kfunc_arg_timer(const struct btf *btf, const struct btf_param *arg)
11570 {
11571 	return __is_kfunc_ptr_arg_type(btf, arg, KF_ARG_TIMER_ID);
11572 }
11573 
11574 static bool is_kfunc_arg_wq(const struct btf *btf, const struct btf_param *arg)
11575 {
11576 	return __is_kfunc_ptr_arg_type(btf, arg, KF_ARG_WORKQUEUE_ID);
11577 }
11578 
11579 static bool is_kfunc_arg_task_work(const struct btf *btf, const struct btf_param *arg)
11580 {
11581 	return __is_kfunc_ptr_arg_type(btf, arg, KF_ARG_TASK_WORK_ID);
11582 }
11583 
11584 static bool is_kfunc_arg_res_spin_lock(const struct btf *btf, const struct btf_param *arg)
11585 {
11586 	return __is_kfunc_ptr_arg_type(btf, arg, KF_ARG_RES_SPIN_LOCK_ID);
11587 }
11588 
11589 static bool is_rbtree_node_type(const struct btf_type *t)
11590 {
11591 	return t == btf_type_by_id(btf_vmlinux, kf_arg_btf_ids[KF_ARG_RB_NODE_ID]);
11592 }
11593 
11594 static bool is_list_node_type(const struct btf_type *t)
11595 {
11596 	return t == btf_type_by_id(btf_vmlinux, kf_arg_btf_ids[KF_ARG_LIST_NODE_ID]);
11597 }
11598 
11599 static bool is_kfunc_arg_callback(struct bpf_verifier_env *env, const struct btf *btf,
11600 				  const struct btf_param *arg)
11601 {
11602 	const struct btf_type *t;
11603 
11604 	t = btf_type_resolve_func_ptr(btf, arg->type, NULL);
11605 	if (!t)
11606 		return false;
11607 
11608 	return true;
11609 }
11610 
11611 static bool is_kfunc_arg_prog_aux(const struct btf *btf, const struct btf_param *arg)
11612 {
11613 	return __is_kfunc_ptr_arg_type(btf, arg, KF_ARG_PROG_AUX_ID);
11614 }
11615 
11616 /*
11617  * A kfunc with KF_IMPLICIT_ARGS has two prototypes in BTF:
11618  *   - the _impl prototype with full arg list (meta->func_proto)
11619  *   - the BPF API prototype w/o implicit args (func->type in BTF)
11620  * To determine whether an argument is implicit, we compare its position
11621  * against the number of arguments in the prototype w/o implicit args.
11622  */
11623 static bool is_kfunc_arg_implicit(const struct bpf_call_arg_meta *meta, u32 arg_idx)
11624 {
11625 	const struct btf_type *func, *func_proto;
11626 	u32 argn;
11627 
11628 	if (!(meta->kfunc_flags & KF_IMPLICIT_ARGS))
11629 		return false;
11630 
11631 	func = btf_type_by_id(meta->btf, meta->func_id);
11632 	func_proto = btf_type_by_id(meta->btf, func->type);
11633 	argn = btf_type_vlen(func_proto);
11634 
11635 	return argn <= arg_idx;
11636 }
11637 
11638 /* Returns true if struct is composed of scalars, 4 levels of nesting allowed */
11639 static bool __btf_type_is_scalar_struct(struct bpf_verifier_env *env,
11640 					const struct btf *btf,
11641 					const struct btf_type *t, int rec)
11642 {
11643 	const struct btf_type *member_type;
11644 	const struct btf_member *member;
11645 	u32 i;
11646 
11647 	if (!btf_type_is_struct(t))
11648 		return false;
11649 
11650 	for_each_member(i, t, member) {
11651 		const struct btf_array *array;
11652 
11653 		member_type = btf_type_skip_modifiers(btf, member->type, NULL);
11654 		if (btf_type_is_struct(member_type)) {
11655 			if (rec >= 3) {
11656 				verbose(env, "max struct nesting depth exceeded\n");
11657 				return false;
11658 			}
11659 			if (!__btf_type_is_scalar_struct(env, btf, member_type, rec + 1))
11660 				return false;
11661 			continue;
11662 		}
11663 		if (btf_type_is_array(member_type)) {
11664 			array = btf_array(member_type);
11665 			if (!array->nelems)
11666 				return false;
11667 			member_type = btf_type_skip_modifiers(btf, array->type, NULL);
11668 			if (!btf_type_is_scalar(member_type))
11669 				return false;
11670 			continue;
11671 		}
11672 		if (!btf_type_is_scalar(member_type))
11673 			return false;
11674 	}
11675 	return true;
11676 }
11677 
11678 enum kfunc_ptr_arg_type {
11679 	KF_ARG_CONST_MEM_SIZE,
11680 	KF_ARG_MEM_SIZE,
11681 	KF_ARG_CONST,
11682 	KF_ARG_CONST_ALLOC_SIZE_OR_ZERO,
11683 	KF_ARG_ANYTHING,
11684 	KF_ARG_PTR_TO_CTX,
11685 	KF_ARG_PTR_TO_ALLOC_BTF_ID,    /* Allocated object */
11686 	KF_ARG_PTR_TO_REFCOUNTED_KPTR, /* Refcounted local kptr */
11687 	KF_ARG_PTR_TO_DYNPTR,
11688 	KF_ARG_PTR_TO_ITER,
11689 	KF_ARG_PTR_TO_LIST_HEAD,
11690 	KF_ARG_PTR_TO_LIST_NODE,
11691 	KF_ARG_PTR_TO_BTF_ID,	       /* Also covers reg2btf_ids conversions */
11692 	KF_ARG_PTR_TO_MEM,
11693 	KF_ARG_PTR_TO_CALLBACK,
11694 	KF_ARG_PTR_TO_RB_ROOT,
11695 	KF_ARG_PTR_TO_RB_NODE,
11696 	KF_ARG_PTR_TO_CONST_STR,
11697 	KF_ARG_CONST_MAP_PTR,
11698 	KF_ARG_PTR_TO_TIMER,
11699 	KF_ARG_PTR_TO_WORKQUEUE,
11700 	KF_ARG_PTR_TO_IRQ_FLAG,
11701 	KF_ARG_PTR_TO_RES_SPIN_LOCK,
11702 	KF_ARG_PTR_TO_TASK_WORK,
11703 	KF_ARG_PTR_TO_ARENA,
11704 };
11705 
11706 enum special_kfunc_type {
11707 	KF_bpf_obj_new_impl,
11708 	KF_bpf_obj_new,
11709 	KF_bpf_obj_drop_impl,
11710 	KF_bpf_obj_drop,
11711 	KF_bpf_refcount_acquire_impl,
11712 	KF_bpf_refcount_acquire,
11713 	KF_bpf_list_push_front_impl,
11714 	KF_bpf_list_push_front,
11715 	KF_bpf_list_push_back_impl,
11716 	KF_bpf_list_push_back,
11717 	KF_bpf_list_add,
11718 	KF_bpf_list_pop_front,
11719 	KF_bpf_list_pop_back,
11720 	KF_bpf_list_del,
11721 	KF_bpf_list_front,
11722 	KF_bpf_list_back,
11723 	KF_bpf_list_is_first,
11724 	KF_bpf_list_is_last,
11725 	KF_bpf_list_empty,
11726 	KF_bpf_cast_to_kern_ctx,
11727 	KF_bpf_rdonly_cast,
11728 	KF_bpf_rcu_read_lock,
11729 	KF_bpf_rcu_read_unlock,
11730 	KF_bpf_rbtree_remove,
11731 	KF_bpf_rbtree_add_impl,
11732 	KF_bpf_rbtree_add,
11733 	KF_bpf_rbtree_first,
11734 	KF_bpf_rbtree_root,
11735 	KF_bpf_rbtree_left,
11736 	KF_bpf_rbtree_right,
11737 	KF_bpf_dynptr_from_skb,
11738 	KF_bpf_dynptr_from_xdp,
11739 	KF_bpf_dynptr_from_skb_meta,
11740 	KF_bpf_xdp_pull_data,
11741 	KF_bpf_dynptr_slice,
11742 	KF_bpf_dynptr_slice_rdwr,
11743 	KF_bpf_dynptr_clone,
11744 	KF_bpf_percpu_obj_new_impl,
11745 	KF_bpf_percpu_obj_new,
11746 	KF_bpf_percpu_obj_drop_impl,
11747 	KF_bpf_percpu_obj_drop,
11748 	KF_bpf_throw,
11749 	KF_bpf_wq_set_callback,
11750 	KF_bpf_preempt_disable,
11751 	KF_bpf_preempt_enable,
11752 	KF_bpf_iter_css_task_new,
11753 	KF_bpf_session_cookie,
11754 	KF_bpf_get_kmem_cache,
11755 	KF_bpf_local_irq_save,
11756 	KF_bpf_local_irq_restore,
11757 	KF_bpf_iter_num_new,
11758 	KF_bpf_iter_num_next,
11759 	KF_bpf_iter_num_destroy,
11760 	KF_bpf_set_dentry_xattr,
11761 	KF_bpf_remove_dentry_xattr,
11762 	KF_bpf_res_spin_lock,
11763 	KF_bpf_res_spin_unlock,
11764 	KF_bpf_res_spin_lock_irqsave,
11765 	KF_bpf_res_spin_unlock_irqrestore,
11766 	KF_bpf_dynptr_from_file,
11767 	KF_bpf_dynptr_file_discard,
11768 	KF___bpf_trap,
11769 	KF_bpf_task_work_schedule_signal,
11770 	KF_bpf_task_work_schedule_resume,
11771 	KF_bpf_arena_alloc_pages,
11772 	KF_bpf_arena_free_pages,
11773 	KF_bpf_session_is_return,
11774 };
11775 
11776 BTF_ID_LIST(special_kfunc_list)
11777 BTF_ID(func, bpf_obj_new_impl)
11778 BTF_ID(func, bpf_obj_new)
11779 BTF_ID(func, bpf_obj_drop_impl)
11780 BTF_ID(func, bpf_obj_drop)
11781 BTF_ID(func, bpf_refcount_acquire_impl)
11782 BTF_ID(func, bpf_refcount_acquire)
11783 BTF_ID(func, bpf_list_push_front_impl)
11784 BTF_ID(func, bpf_list_push_front)
11785 BTF_ID(func, bpf_list_push_back_impl)
11786 BTF_ID(func, bpf_list_push_back)
11787 BTF_ID(func, bpf_list_add)
11788 BTF_ID(func, bpf_list_pop_front)
11789 BTF_ID(func, bpf_list_pop_back)
11790 BTF_ID(func, bpf_list_del)
11791 BTF_ID(func, bpf_list_front)
11792 BTF_ID(func, bpf_list_back)
11793 BTF_ID(func, bpf_list_is_first)
11794 BTF_ID(func, bpf_list_is_last)
11795 BTF_ID(func, bpf_list_empty)
11796 BTF_ID(func, bpf_cast_to_kern_ctx)
11797 BTF_ID(func, bpf_rdonly_cast)
11798 BTF_ID(func, bpf_rcu_read_lock)
11799 BTF_ID(func, bpf_rcu_read_unlock)
11800 BTF_ID(func, bpf_rbtree_remove)
11801 BTF_ID(func, bpf_rbtree_add_impl)
11802 BTF_ID(func, bpf_rbtree_add)
11803 BTF_ID(func, bpf_rbtree_first)
11804 BTF_ID(func, bpf_rbtree_root)
11805 BTF_ID(func, bpf_rbtree_left)
11806 BTF_ID(func, bpf_rbtree_right)
11807 #ifdef CONFIG_NET
11808 BTF_ID(func, bpf_dynptr_from_skb)
11809 BTF_ID(func, bpf_dynptr_from_xdp)
11810 BTF_ID(func, bpf_dynptr_from_skb_meta)
11811 BTF_ID(func, bpf_xdp_pull_data)
11812 #else
11813 BTF_ID_UNUSED
11814 BTF_ID_UNUSED
11815 BTF_ID_UNUSED
11816 BTF_ID_UNUSED
11817 #endif
11818 BTF_ID(func, bpf_dynptr_slice)
11819 BTF_ID(func, bpf_dynptr_slice_rdwr)
11820 BTF_ID(func, bpf_dynptr_clone)
11821 BTF_ID(func, bpf_percpu_obj_new_impl)
11822 BTF_ID(func, bpf_percpu_obj_new)
11823 BTF_ID(func, bpf_percpu_obj_drop_impl)
11824 BTF_ID(func, bpf_percpu_obj_drop)
11825 BTF_ID(func, bpf_throw)
11826 BTF_ID(func, bpf_wq_set_callback)
11827 BTF_ID(func, bpf_preempt_disable)
11828 BTF_ID(func, bpf_preempt_enable)
11829 #ifdef CONFIG_CGROUPS
11830 BTF_ID(func, bpf_iter_css_task_new)
11831 #else
11832 BTF_ID_UNUSED
11833 #endif
11834 #ifdef CONFIG_BPF_EVENTS
11835 BTF_ID(func, bpf_session_cookie)
11836 #else
11837 BTF_ID_UNUSED
11838 #endif
11839 BTF_ID(func, bpf_get_kmem_cache)
11840 BTF_ID(func, bpf_local_irq_save)
11841 BTF_ID(func, bpf_local_irq_restore)
11842 BTF_ID(func, bpf_iter_num_new)
11843 BTF_ID(func, bpf_iter_num_next)
11844 BTF_ID(func, bpf_iter_num_destroy)
11845 #ifdef CONFIG_BPF_LSM
11846 BTF_ID(func, bpf_set_dentry_xattr)
11847 BTF_ID(func, bpf_remove_dentry_xattr)
11848 #else
11849 BTF_ID_UNUSED
11850 BTF_ID_UNUSED
11851 #endif
11852 BTF_ID(func, bpf_res_spin_lock)
11853 BTF_ID(func, bpf_res_spin_unlock)
11854 BTF_ID(func, bpf_res_spin_lock_irqsave)
11855 BTF_ID(func, bpf_res_spin_unlock_irqrestore)
11856 BTF_ID(func, bpf_dynptr_from_file)
11857 BTF_ID(func, bpf_dynptr_file_discard)
11858 BTF_ID(func, __bpf_trap)
11859 BTF_ID(func, bpf_task_work_schedule_signal)
11860 BTF_ID(func, bpf_task_work_schedule_resume)
11861 BTF_ID(func, bpf_arena_alloc_pages)
11862 BTF_ID(func, bpf_arena_free_pages)
11863 #ifdef CONFIG_BPF_EVENTS
11864 BTF_ID(func, bpf_session_is_return)
11865 #else
11866 BTF_ID_UNUSED
11867 #endif
11868 
11869 static bool is_bpf_obj_new_kfunc(u32 func_id)
11870 {
11871 	return func_id == special_kfunc_list[KF_bpf_obj_new] ||
11872 	       func_id == special_kfunc_list[KF_bpf_obj_new_impl];
11873 }
11874 
11875 static bool is_bpf_percpu_obj_new_kfunc(u32 func_id)
11876 {
11877 	return func_id == special_kfunc_list[KF_bpf_percpu_obj_new] ||
11878 	       func_id == special_kfunc_list[KF_bpf_percpu_obj_new_impl];
11879 }
11880 
11881 static bool is_bpf_obj_drop_kfunc(u32 func_id)
11882 {
11883 	return func_id == special_kfunc_list[KF_bpf_obj_drop] ||
11884 	       func_id == special_kfunc_list[KF_bpf_obj_drop_impl];
11885 }
11886 
11887 static bool is_bpf_percpu_obj_drop_kfunc(u32 func_id)
11888 {
11889 	return func_id == special_kfunc_list[KF_bpf_percpu_obj_drop] ||
11890 	       func_id == special_kfunc_list[KF_bpf_percpu_obj_drop_impl];
11891 }
11892 
11893 static bool is_bpf_refcount_acquire_kfunc(u32 func_id)
11894 {
11895 	return func_id == special_kfunc_list[KF_bpf_refcount_acquire] ||
11896 	       func_id == special_kfunc_list[KF_bpf_refcount_acquire_impl];
11897 }
11898 
11899 static bool is_bpf_list_push_kfunc(u32 func_id)
11900 {
11901 	return func_id == special_kfunc_list[KF_bpf_list_push_front] ||
11902 	       func_id == special_kfunc_list[KF_bpf_list_push_front_impl] ||
11903 	       func_id == special_kfunc_list[KF_bpf_list_push_back] ||
11904 	       func_id == special_kfunc_list[KF_bpf_list_push_back_impl] ||
11905 	       func_id == special_kfunc_list[KF_bpf_list_add];
11906 }
11907 
11908 static bool is_bpf_rbtree_add_kfunc(u32 func_id)
11909 {
11910 	return func_id == special_kfunc_list[KF_bpf_rbtree_add] ||
11911 	       func_id == special_kfunc_list[KF_bpf_rbtree_add_impl];
11912 }
11913 
11914 static bool is_task_work_add_kfunc(u32 func_id)
11915 {
11916 	return func_id == special_kfunc_list[KF_bpf_task_work_schedule_signal] ||
11917 	       func_id == special_kfunc_list[KF_bpf_task_work_schedule_resume];
11918 }
11919 
11920 static bool is_kfunc_ret_null(struct bpf_call_arg_meta *meta)
11921 {
11922 	if (is_bpf_refcount_acquire_kfunc(meta->func_id) && meta->arg_owning_ref)
11923 		return false;
11924 
11925 	return meta->kfunc_flags & KF_RET_NULL;
11926 }
11927 
11928 static bool is_kfunc_bpf_rcu_read_lock(struct bpf_call_arg_meta *meta)
11929 {
11930 	return meta->func_id == special_kfunc_list[KF_bpf_rcu_read_lock];
11931 }
11932 
11933 static bool is_kfunc_bpf_rcu_read_unlock(struct bpf_call_arg_meta *meta)
11934 {
11935 	return meta->func_id == special_kfunc_list[KF_bpf_rcu_read_unlock];
11936 }
11937 
11938 static bool is_kfunc_bpf_preempt_disable(struct bpf_call_arg_meta *meta)
11939 {
11940 	return meta->func_id == special_kfunc_list[KF_bpf_preempt_disable];
11941 }
11942 
11943 static bool is_kfunc_bpf_preempt_enable(struct bpf_call_arg_meta *meta)
11944 {
11945 	return meta->func_id == special_kfunc_list[KF_bpf_preempt_enable];
11946 }
11947 
11948 bool bpf_is_kfunc_pkt_changing(struct bpf_call_arg_meta *meta)
11949 {
11950 	return meta->func_id == special_kfunc_list[KF_bpf_xdp_pull_data];
11951 }
11952 
11953 static int
11954 get_kfunc_arg_type(struct bpf_verifier_env *env, struct bpf_call_arg_meta *meta,
11955 		   const struct btf_param *args, int arg, int nargs)
11956 {
11957 	const struct btf_type *t, *ref_t = NULL;
11958 	argno_t argno = argno_from_arg(arg + 1);
11959 	const char *ref_tname = NULL;
11960 	int arg_type;
11961 
11962 	t = btf_type_skip_modifiers(meta->btf, args[arg].type, NULL);
11963 
11964 	/* Scalar arguments are classified from their BTF suffix/name alone. */
11965 	if (btf_type_is_scalar(t)) {
11966 		if (is_kfunc_arg_constant(meta->btf, &args[arg]))
11967 			return KF_ARG_CONST;
11968 		if (is_kfunc_arg_const_mem_size(meta->btf, &args[arg]))
11969 			return KF_ARG_CONST_MEM_SIZE;
11970 		if (is_kfunc_arg_mem_size(meta->btf, &args[arg]))
11971 			return KF_ARG_MEM_SIZE;
11972 		if (is_kfunc_arg_scalar_with_name(meta->btf, &args[arg], "rdonly_buf_size") ||
11973 		    is_kfunc_arg_scalar_with_name(meta->btf, &args[arg], "rdwr_buf_size"))
11974 			return KF_ARG_CONST_ALLOC_SIZE_OR_ZERO;
11975 		return KF_ARG_ANYTHING;
11976 	}
11977 
11978 	if (!btf_type_is_ptr(t)) {
11979 		verbose(env, "Unrecognized %s type %s\n",
11980 			reg_arg_name(env, argno), btf_type_str(t));
11981 		return -EINVAL;
11982 	}
11983 	ref_t = btf_type_skip_modifiers(meta->btf, t->type, NULL);
11984 	ref_tname = btf_name_by_offset(meta->btf, ref_t->name_off);
11985 
11986 	/* In this function, we verify the kfunc's BTF as per the argument type,
11987 	 * leaving the rest of the verification with respect to the register
11988 	 * type to our caller. When a set of conditions hold in the BTF type of
11989 	 * arguments, we resolve it to a known kfunc_ptr_arg_type.
11990 	 */
11991 	if (meta->func_id == special_kfunc_list[KF_bpf_cast_to_kern_ctx] ||
11992 	    meta->func_id == special_kfunc_list[KF_bpf_session_is_return] ||
11993 	    meta->func_id == special_kfunc_list[KF_bpf_session_cookie])
11994 		arg_type = KF_ARG_PTR_TO_CTX;
11995 	else if (btf_is_prog_ctx_type(&env->log, meta->btf, t, resolve_prog_type(env->prog), arg))
11996 		arg_type = KF_ARG_PTR_TO_CTX;
11997 	else if (is_kfunc_arg_alloc_obj(meta->btf, &args[arg]))
11998 		arg_type = KF_ARG_PTR_TO_ALLOC_BTF_ID;
11999 	else if (is_kfunc_arg_refcounted_kptr(meta->btf, &args[arg]))
12000 		arg_type = KF_ARG_PTR_TO_REFCOUNTED_KPTR;
12001 	else if (is_kfunc_arg_dynptr(meta->btf, &args[arg]))
12002 		arg_type = KF_ARG_PTR_TO_DYNPTR;
12003 	else if (is_kfunc_arg_iter(meta, arg, &args[arg]))
12004 		arg_type = KF_ARG_PTR_TO_ITER;
12005 	else if (is_kfunc_arg_list_head(meta->btf, &args[arg]))
12006 		arg_type = KF_ARG_PTR_TO_LIST_HEAD;
12007 	else if (is_kfunc_arg_list_node(meta->btf, &args[arg]))
12008 		arg_type = KF_ARG_PTR_TO_LIST_NODE;
12009 	else if (is_kfunc_arg_rbtree_root(meta->btf, &args[arg]))
12010 		arg_type = KF_ARG_PTR_TO_RB_ROOT;
12011 	else if (is_kfunc_arg_rbtree_node(meta->btf, &args[arg]))
12012 		arg_type = KF_ARG_PTR_TO_RB_NODE;
12013 	else if (is_kfunc_arg_const_str(meta->btf, &args[arg]))
12014 		arg_type = KF_ARG_PTR_TO_CONST_STR;
12015 	else if (is_kfunc_arg_const_map(meta->btf, &args[arg]))
12016 		arg_type = KF_ARG_CONST_MAP_PTR;
12017 	else if (is_kfunc_arg_map(meta->btf, &args[arg]))
12018 		arg_type = KF_ARG_PTR_TO_BTF_ID;
12019 	else if (is_kfunc_arg_wq(meta->btf, &args[arg]))
12020 		arg_type = KF_ARG_PTR_TO_WORKQUEUE;
12021 	else if (is_kfunc_arg_timer(meta->btf, &args[arg]))
12022 		arg_type = KF_ARG_PTR_TO_TIMER;
12023 	else if (is_kfunc_arg_task_work(meta->btf, &args[arg]))
12024 		arg_type = KF_ARG_PTR_TO_TASK_WORK;
12025 	else if (is_kfunc_arg_irq_flag(meta->btf, &args[arg]))
12026 		arg_type = KF_ARG_PTR_TO_IRQ_FLAG;
12027 	else if (is_kfunc_arg_res_spin_lock(meta->btf, &args[arg]))
12028 		arg_type = KF_ARG_PTR_TO_RES_SPIN_LOCK;
12029 	else if (is_kfunc_arg_callback(env, meta->btf, &args[arg]))
12030 		arg_type = KF_ARG_PTR_TO_CALLBACK;
12031 	else if (is_kfunc_arg_arena(meta->btf, &args[arg])) {
12032 		if (!bpf_jit_supports_arena_args()) {
12033 			verbose(env, "JIT does not support kfunc %s() with arena pointer arguments\n",
12034 				meta->func_name);
12035 			return -ENOTSUPP;
12036 		}
12037 		if (!env->prog->aux->arena) {
12038 			verbose(env,
12039 				"%s arena pointer requires a program with an associated arena\n",
12040 				reg_arg_name(env, argno));
12041 			return -EINVAL;
12042 		}
12043 		if (reg_from_argno(argno) < 0) {
12044 			verbose(env, "%s arena pointer cannot be a stack argument\n",
12045 				reg_arg_name(env, argno));
12046 			return -EINVAL;
12047 		}
12048 		/*
12049 		 * Both suffixes accept a constant zero. The function model determines
12050 		 * whether the JIT rebases it to the arena base or preserves NULL.
12051 		 * The common nullable path below records that verifier property.
12052 		 */
12053 		arg_type = KF_ARG_PTR_TO_ARENA;
12054 	} else if (arg + 1 < nargs &&
12055 		 (is_kfunc_arg_mem_size(meta->btf, &args[arg + 1]) ||
12056 		  is_kfunc_arg_const_mem_size(meta->btf, &args[arg + 1]))) {
12057 		if (!btf_type_is_void(ref_t) && !btf_type_is_scalar(ref_t) &&
12058 		    !__btf_type_is_scalar_struct(env, meta->btf, ref_t, 0)) {
12059 			verbose(env, "%s pointer type %s %s must point to void, scalar, or struct with scalar\n",
12060 				reg_arg_name(env, argno), btf_type_str(ref_t), ref_tname);
12061 			return -EINVAL;
12062 		}
12063 		arg_type = KF_ARG_PTR_TO_MEM;
12064 	} else if (btf_type_is_struct(ref_t))
12065 		/* A pointer to a struct without a size argument is classified as KF_ARG_PTR_TO_BTF_ID */
12066 		arg_type = KF_ARG_PTR_TO_BTF_ID;
12067 	else {
12068 		/*
12069 		 * Otherwise this is a fixed-size memory buffer supported by
12070 		 * check_helper_mem_access(): a pointer to a scalar or a struct of
12071 		 * scalars. The access size is derived from the pointed-to BTF type.
12072 		 */
12073 		if (!btf_type_is_scalar(ref_t) &&
12074 		    !__btf_type_is_scalar_struct(env, meta->btf, ref_t, 0)) {
12075 			verbose(env, "%s pointer type %s %s must point to scalar, or struct with scalar\n",
12076 				reg_arg_name(env, argno), btf_type_str(ref_t), ref_tname);
12077 			return -EINVAL;
12078 		}
12079 		arg_type = KF_ARG_PTR_TO_MEM | MEM_FIXED_SIZE;
12080 	}
12081 
12082 	if (is_kfunc_arg_nullable(meta->btf, &args[arg]))
12083 		arg_type |= PTR_MAYBE_NULL;
12084 
12085 	return arg_type;
12086 }
12087 
12088 static int gen_kfunc_arg_proto(struct bpf_verifier_env *env, struct bpf_call_arg_meta *meta,
12089 			       struct bpf_func_proto *proto)
12090 {
12091 	const struct btf *btf = meta->btf;
12092 	const struct btf_param *args;
12093 	u32 i, nargs;
12094 	int arg_type;
12095 
12096 	args = (const struct btf_param *)(meta->func_proto + 1);
12097 	nargs = btf_type_vlen(meta->func_proto);
12098 	if (nargs > MAX_BPF_FUNC_ARGS) {
12099 		verbose(env, "Function %s has %d > %d args\n", meta->func_name,
12100 			nargs, MAX_BPF_FUNC_ARGS);
12101 		return -EINVAL;
12102 	}
12103 	if (nargs > MAX_BPF_FUNC_REG_ARGS && !bpf_jit_supports_stack_args()) {
12104 		verbose(env, "JIT does not support kfunc %s() with %d args\n",
12105 			meta->func_name, nargs);
12106 		return -ENOTSUPP;
12107 	}
12108 
12109 	for (i = 0; i < nargs; i++) {
12110 		if (is_kfunc_arg_prog_aux(btf, &args[i]) ||
12111 		    is_kfunc_arg_ignore(btf, &args[i]) ||
12112 		    is_kfunc_arg_implicit(meta, i))
12113 			continue;
12114 
12115 		arg_type = get_kfunc_arg_type(env, meta, args, i, nargs);
12116 		if (arg_type < 0)
12117 			return arg_type;
12118 
12119 		proto->arg_type[i] = arg_type;
12120 	}
12121 
12122 	return 0;
12123 }
12124 
12125 static int process_kf_arg_ptr_to_btf_id(struct bpf_verifier_env *env,
12126 					struct bpf_reg_state *reg,
12127 					const struct btf_type *ref_t,
12128 					const char *ref_tname, u32 ref_id,
12129 					struct bpf_call_arg_meta *meta,
12130 					int arg, argno_t argno)
12131 {
12132 	const struct btf_type *reg_ref_t;
12133 	bool strict_type_match = false;
12134 	const struct btf *reg_btf;
12135 	const char *reg_ref_tname;
12136 	bool taking_projection;
12137 	bool struct_same;
12138 	u32 reg_ref_id;
12139 
12140 	if (base_type(reg->type) == PTR_TO_BTF_ID) {
12141 		reg_btf = reg->btf;
12142 		reg_ref_id = reg->btf_id;
12143 	} else {
12144 		reg_btf = btf_vmlinux;
12145 		reg_ref_id = *reg2btf_ids[base_type(reg->type)];
12146 	}
12147 
12148 	/* Enforce strict type matching for calls to kfuncs that are acquiring
12149 	 * or releasing a reference, or are no-cast aliases. We do _not_
12150 	 * enforce strict matching for kfuncs by default,
12151 	 * as we want to enable BPF programs to pass types that are bitwise
12152 	 * equivalent without forcing them to explicitly cast with something
12153 	 * like bpf_cast_to_kern_ctx().
12154 	 *
12155 	 * For example, say we had a type like the following:
12156 	 *
12157 	 * struct bpf_cpumask {
12158 	 *	cpumask_t cpumask;
12159 	 *	refcount_t usage;
12160 	 * };
12161 	 *
12162 	 * Note that as specified in <linux/cpumask.h>, cpumask_t is typedef'ed
12163 	 * to a struct cpumask, so it would be safe to pass a struct
12164 	 * bpf_cpumask * to a kfunc expecting a struct cpumask *.
12165 	 *
12166 	 * The philosophy here is similar to how we allow scalars of different
12167 	 * types to be passed to kfuncs as long as the size is the same. The
12168 	 * only difference here is that we're simply allowing
12169 	 * btf_struct_ids_match() to walk the struct at the 0th offset, and
12170 	 * resolve types.
12171 	 */
12172 	if ((is_kfunc_release(meta) && reg_is_referenced(env, reg)) ||
12173 	    btf_type_ids_nocast_alias(&env->log, reg_btf, reg_ref_id, meta->btf, ref_id))
12174 		strict_type_match = true;
12175 
12176 	WARN_ON_ONCE(is_kfunc_release(meta) && !tnum_is_const(reg->var_off));
12177 
12178 	reg_ref_t = btf_type_skip_modifiers(reg_btf, reg_ref_id, &reg_ref_id);
12179 	reg_ref_tname = btf_name_by_offset(reg_btf, reg_ref_t->name_off);
12180 	struct_same = btf_struct_ids_match(&env->log, reg_btf, reg_ref_id, reg->var_off.value,
12181 					   meta->btf, ref_id, strict_type_match,
12182 					   !type_is_alloc(reg->type));
12183 	/* If kfunc is accepting a projection type (ie. __sk_buff), it cannot
12184 	 * actually use it -- it must cast to the underlying type. So we allow
12185 	 * caller to pass in the underlying type.
12186 	 */
12187 	taking_projection = btf_is_projection_of(ref_tname, reg_ref_tname);
12188 	if (!taking_projection && !struct_same) {
12189 		verbose(env, "kernel function %s %s expected pointer to %s %s but %s has a pointer to %s %s\n",
12190 			meta->func_name, reg_arg_name(env, argno),
12191 			btf_type_str(ref_t), ref_tname, reg_arg_name(env, argno),
12192 			btf_type_str(reg_ref_t), reg_ref_tname);
12193 		return -EINVAL;
12194 	}
12195 	return 0;
12196 }
12197 
12198 static int process_irq_flag(struct bpf_verifier_env *env, struct bpf_reg_state *reg, argno_t argno,
12199 			     struct bpf_call_arg_meta *meta)
12200 {
12201 	int err, spi, kfunc_class = IRQ_NATIVE_KFUNC;
12202 	bool irq_save;
12203 
12204 	if (meta->func_id == special_kfunc_list[KF_bpf_local_irq_save] ||
12205 	    meta->func_id == special_kfunc_list[KF_bpf_res_spin_lock_irqsave]) {
12206 		irq_save = true;
12207 		if (meta->func_id == special_kfunc_list[KF_bpf_res_spin_lock_irqsave])
12208 			kfunc_class = IRQ_LOCK_KFUNC;
12209 	} else if (meta->func_id == special_kfunc_list[KF_bpf_local_irq_restore] ||
12210 		   meta->func_id == special_kfunc_list[KF_bpf_res_spin_unlock_irqrestore]) {
12211 		irq_save = false;
12212 		if (meta->func_id == special_kfunc_list[KF_bpf_res_spin_unlock_irqrestore])
12213 			kfunc_class = IRQ_LOCK_KFUNC;
12214 	} else {
12215 		verifier_bug(env, "unknown irq flags kfunc");
12216 		return -EFAULT;
12217 	}
12218 
12219 	if (irq_save) {
12220 		if (!is_irq_flag_reg_valid_uninit(env, reg)) {
12221 			verbose(env, "expected uninitialized irq flag as %s\n",
12222 				reg_arg_name(env, argno));
12223 			bpf_diag_res(env, env->insn_idx, "IRQ flag is already initialized",
12224 				     "Saving IRQ state requires an uninitialized stack slot for "
12225 				     "the IRQ flag, but this slot already contains tracked IRQ "
12226 				     "flag state.",
12227 				     "Use a fresh stack slot for this save operation, or restore "
12228 				     "the existing IRQ flag before reusing the slot.");
12229 			return -EINVAL;
12230 		}
12231 
12232 		err = check_mem_access(env, env->insn_idx, reg, argno, 0, BPF_DW,
12233 				       BPF_WRITE, -1, false, false);
12234 		if (err)
12235 			return err;
12236 
12237 		err = mark_stack_slot_irq_flag(env, meta, reg, env->insn_idx, kfunc_class);
12238 		if (err)
12239 			return err;
12240 	} else {
12241 		err = is_irq_flag_reg_valid_init(env, reg);
12242 		if (err) {
12243 			verbose(env, "expected an initialized irq flag as %s\n",
12244 				reg_arg_name(env, argno));
12245 			bpf_diag_res(env, env->insn_idx, "uninitialized IRQ flag restore",
12246 				     "Restoring IRQ state requires a stack slot that was "
12247 				     "initialized by a matching IRQ save operation on this path.",
12248 				     "Pass the same stack slot that was previously initialized by "
12249 				     "the matching IRQ save kfunc.");
12250 			return err;
12251 		}
12252 
12253 		spi = irq_flag_get_spi(env, reg);
12254 		if (spi < 0)
12255 			return spi;
12256 
12257 		mark_stack_slots_scratched(env, spi, 1);
12258 
12259 		err = unmark_stack_slot_irq_flag(env, reg, kfunc_class);
12260 		if (err)
12261 			return err;
12262 
12263 		if (!in_rcu_cs(env))
12264 			invalidate_rcu_protected_refs(env);
12265 	}
12266 	return 0;
12267 }
12268 
12269 static int ref_set_non_owning(struct bpf_verifier_env *env, struct bpf_reg_state *reg)
12270 {
12271 	struct btf_record *rec = reg_btf_record(reg);
12272 
12273 	if (!env->cur_state->active_locks) {
12274 		verifier_bug(env, "%s w/o active lock", __func__);
12275 		return -EFAULT;
12276 	}
12277 
12278 	if (type_flag(reg->type) & NON_OWN_REF) {
12279 		verifier_bug(env, "NON_OWN_REF already set");
12280 		return -EFAULT;
12281 	}
12282 
12283 	reg->type |= NON_OWN_REF;
12284 	if (rec->refcount_off >= 0)
12285 		reg->type |= MEM_RCU;
12286 
12287 	return 0;
12288 }
12289 
12290 static void ref_convert_owning_non_owning(struct bpf_verifier_env *env, u32 id)
12291 {
12292 	struct bpf_func_state *unused;
12293 	struct bpf_reg_state *reg;
12294 	int err;
12295 
12296 	err = release_reference_nomark(env, id);
12297 	WARN_ON_ONCE(err);
12298 
12299 	bpf_for_each_reg_in_vstate(env->cur_state, unused, reg, ({
12300 		if (reg->id == id) {
12301 			reg->id = 0;
12302 			ref_set_non_owning(env, reg);
12303 		}
12304 	}));
12305 
12306 	return;
12307 }
12308 
12309 /* Implementation details:
12310  *
12311  * Each register points to some region of memory, which we define as an
12312  * allocation. Each allocation may embed a bpf_spin_lock which protects any
12313  * special BPF objects (bpf_list_head, bpf_rb_root, etc.) part of the same
12314  * allocation. The lock and the data it protects are colocated in the same
12315  * memory region.
12316  *
12317  * Hence, everytime a register holds a pointer value pointing to such
12318  * allocation, the verifier preserves a unique reg->id for it.
12319  *
12320  * The verifier remembers the lock 'ptr' and the lock 'id' whenever
12321  * bpf_spin_lock is called.
12322  *
12323  * To enable this, lock state in the verifier captures two values:
12324  *	active_lock.ptr = Register's type specific pointer
12325  *	active_lock.id  = A unique ID for each register pointer value
12326  *
12327  * Currently, PTR_TO_MAP_VALUE and PTR_TO_BTF_ID | MEM_ALLOC are the two
12328  * supported register types.
12329  *
12330  * The active_lock.ptr in case of map values is the reg->map_ptr, and in case of
12331  * allocated objects is the reg->btf pointer.
12332  *
12333  * The active_lock.id is non-unique for maps supporting direct_value_addr, as we
12334  * can establish the provenance of the map value statically for each distinct
12335  * lookup into such maps. They always contain a single map value hence unique
12336  * IDs for each pseudo load pessimizes the algorithm and rejects valid programs.
12337  *
12338  * So, in case of global variables, they use array maps with max_entries = 1,
12339  * hence their active_lock.ptr becomes map_ptr and id = 0 (since they all point
12340  * into the same map value as max_entries is 1, as described above).
12341  *
12342  * In case of inner map lookups, the inner map pointer has same map_ptr as the
12343  * outer map pointer (in verifier context), but each lookup into an inner map
12344  * assigns a fresh reg->id to the lookup, so while lookups into distinct inner
12345  * maps from the same outer map share the same map_ptr as active_lock.ptr, they
12346  * will get different reg->id assigned to each lookup, hence different
12347  * active_lock.id.
12348  *
12349  * In case of allocated objects, active_lock.ptr is the reg->btf, and the
12350  * reg->id is a unique ID preserved after the NULL pointer check on the pointer
12351  * returned from bpf_obj_new. Each allocation receives a new reg->id.
12352  */
12353 static int check_reg_allocation_locked(struct bpf_verifier_env *env, struct bpf_reg_state *reg)
12354 {
12355 	struct bpf_reference_state *s;
12356 	void *ptr;
12357 	u32 id;
12358 
12359 	switch ((int)reg->type) {
12360 	case PTR_TO_MAP_VALUE:
12361 		ptr = reg->map_ptr;
12362 		break;
12363 	case PTR_TO_BTF_ID | MEM_ALLOC:
12364 		ptr = reg->btf;
12365 		break;
12366 	default:
12367 		verifier_bug(env, "unknown reg type for lock check");
12368 		return -EFAULT;
12369 	}
12370 	id = reg->id;
12371 
12372 	if (!env->cur_state->active_locks)
12373 		return -EINVAL;
12374 	s = find_lock_state(env->cur_state, REF_TYPE_LOCK_MASK, id, ptr);
12375 	if (!s) {
12376 		verbose(env, "held lock and object are not in the same allocation\n");
12377 		return -EINVAL;
12378 	}
12379 	return 0;
12380 }
12381 
12382 static bool is_bpf_list_api_kfunc(u32 btf_id)
12383 {
12384 	return is_bpf_list_push_kfunc(btf_id) ||
12385 	       btf_id == special_kfunc_list[KF_bpf_list_pop_front] ||
12386 	       btf_id == special_kfunc_list[KF_bpf_list_pop_back] ||
12387 	       btf_id == special_kfunc_list[KF_bpf_list_del] ||
12388 	       btf_id == special_kfunc_list[KF_bpf_list_front] ||
12389 	       btf_id == special_kfunc_list[KF_bpf_list_back] ||
12390 	       btf_id == special_kfunc_list[KF_bpf_list_is_first] ||
12391 	       btf_id == special_kfunc_list[KF_bpf_list_is_last] ||
12392 	       btf_id == special_kfunc_list[KF_bpf_list_empty];
12393 }
12394 
12395 static bool is_bpf_rbtree_api_kfunc(u32 btf_id)
12396 {
12397 	return is_bpf_rbtree_add_kfunc(btf_id) ||
12398 	       btf_id == special_kfunc_list[KF_bpf_rbtree_remove] ||
12399 	       btf_id == special_kfunc_list[KF_bpf_rbtree_first] ||
12400 	       btf_id == special_kfunc_list[KF_bpf_rbtree_root] ||
12401 	       btf_id == special_kfunc_list[KF_bpf_rbtree_left] ||
12402 	       btf_id == special_kfunc_list[KF_bpf_rbtree_right];
12403 }
12404 
12405 static bool is_bpf_res_spin_lock_kfunc(u32 btf_id)
12406 {
12407 	return btf_id == special_kfunc_list[KF_bpf_res_spin_lock] ||
12408 	       btf_id == special_kfunc_list[KF_bpf_res_spin_unlock] ||
12409 	       btf_id == special_kfunc_list[KF_bpf_res_spin_lock_irqsave] ||
12410 	       btf_id == special_kfunc_list[KF_bpf_res_spin_unlock_irqrestore];
12411 }
12412 
12413 static bool kfunc_spin_allowed(struct bpf_verifier_env *env, s32 func_id, s16 offset)
12414 {
12415 	struct bpf_kfunc_meta kfunc;
12416 	int err;
12417 
12418 	err = fetch_kfunc_meta(env, func_id, offset, &kfunc);
12419 	if (err || !kfunc.flags)
12420 		return false;
12421 
12422 	return *kfunc.flags & KF_SPINLOCK_SAFE;
12423 }
12424 
12425 static bool is_sync_callback_calling_kfunc(u32 btf_id)
12426 {
12427 	return is_bpf_rbtree_add_kfunc(btf_id);
12428 }
12429 
12430 static bool is_async_callback_calling_kfunc(u32 btf_id)
12431 {
12432 	return is_bpf_wq_set_callback_kfunc(btf_id) ||
12433 	       is_task_work_add_kfunc(btf_id);
12434 }
12435 
12436 bool bpf_is_throw_kfunc(struct bpf_insn *insn)
12437 {
12438 	return bpf_pseudo_kfunc_call(insn) && insn->off == 0 &&
12439 	       insn->imm == special_kfunc_list[KF_bpf_throw];
12440 }
12441 
12442 static bool is_bpf_wq_set_callback_kfunc(u32 btf_id)
12443 {
12444 	return btf_id == special_kfunc_list[KF_bpf_wq_set_callback];
12445 }
12446 
12447 static bool is_callback_calling_kfunc(u32 btf_id)
12448 {
12449 	return is_sync_callback_calling_kfunc(btf_id) ||
12450 	       is_async_callback_calling_kfunc(btf_id);
12451 }
12452 
12453 static bool is_rbtree_lock_required_kfunc(u32 btf_id)
12454 {
12455 	return is_bpf_rbtree_api_kfunc(btf_id);
12456 }
12457 
12458 static bool check_kfunc_is_graph_root_api(struct bpf_verifier_env *env,
12459 					  enum btf_field_type head_field_type,
12460 					  u32 kfunc_btf_id)
12461 {
12462 	bool ret;
12463 
12464 	switch (head_field_type) {
12465 	case BPF_LIST_HEAD:
12466 		ret = is_bpf_list_api_kfunc(kfunc_btf_id);
12467 		break;
12468 	case BPF_RB_ROOT:
12469 		ret = is_bpf_rbtree_api_kfunc(kfunc_btf_id);
12470 		break;
12471 	default:
12472 		verbose(env, "verifier internal error: unexpected graph root argument type %s\n",
12473 			btf_field_type_name(head_field_type));
12474 		return false;
12475 	}
12476 
12477 	if (!ret)
12478 		verbose(env, "verifier internal error: %s head arg for unknown kfunc\n",
12479 			btf_field_type_name(head_field_type));
12480 	return ret;
12481 }
12482 
12483 static bool check_kfunc_is_graph_node_api(struct bpf_verifier_env *env,
12484 					  enum btf_field_type node_field_type,
12485 					  u32 kfunc_btf_id)
12486 {
12487 	bool ret;
12488 
12489 	switch (node_field_type) {
12490 	case BPF_LIST_NODE:
12491 		ret = is_bpf_list_push_kfunc(kfunc_btf_id) ||
12492 		      kfunc_btf_id == special_kfunc_list[KF_bpf_list_del] ||
12493 		      kfunc_btf_id == special_kfunc_list[KF_bpf_list_is_first] ||
12494 		      kfunc_btf_id == special_kfunc_list[KF_bpf_list_is_last];
12495 		break;
12496 	case BPF_RB_NODE:
12497 		ret = (is_bpf_rbtree_add_kfunc(kfunc_btf_id) ||
12498 		       kfunc_btf_id == special_kfunc_list[KF_bpf_rbtree_remove] ||
12499 		       kfunc_btf_id == special_kfunc_list[KF_bpf_rbtree_left] ||
12500 		       kfunc_btf_id == special_kfunc_list[KF_bpf_rbtree_right]);
12501 		break;
12502 	default:
12503 		verbose(env, "verifier internal error: unexpected graph node argument type %s\n",
12504 			btf_field_type_name(node_field_type));
12505 		return false;
12506 	}
12507 
12508 	if (!ret)
12509 		verbose(env, "verifier internal error: %s node arg for unknown kfunc\n",
12510 			btf_field_type_name(node_field_type));
12511 	return ret;
12512 }
12513 
12514 static int
12515 __process_kf_arg_ptr_to_graph_root(struct bpf_verifier_env *env,
12516 				   struct bpf_reg_state *reg, argno_t argno,
12517 				   struct bpf_call_arg_meta *meta,
12518 				   enum btf_field_type head_field_type,
12519 				   struct btf_field **head_field)
12520 {
12521 	const char *head_type_name;
12522 	struct btf_field *field;
12523 	struct btf_record *rec;
12524 	u32 head_off;
12525 
12526 	if (meta->btf != btf_vmlinux) {
12527 		verifier_bug(env, "unexpected btf mismatch in kfunc call");
12528 		return -EFAULT;
12529 	}
12530 
12531 	if (!check_kfunc_is_graph_root_api(env, head_field_type, meta->func_id))
12532 		return -EFAULT;
12533 
12534 	head_type_name = btf_field_type_name(head_field_type);
12535 	if (!tnum_is_const(reg->var_off)) {
12536 		verbose(env,
12537 			"%s doesn't have constant offset. %s has to be at the constant offset\n",
12538 			reg_arg_name(env, argno), head_type_name);
12539 		return -EINVAL;
12540 	}
12541 
12542 	rec = reg_btf_record(reg);
12543 	head_off = reg->var_off.value;
12544 	field = btf_record_find(rec, head_off, head_field_type);
12545 	if (!field) {
12546 		verbose(env, "%s not found at offset=%u\n", head_type_name, head_off);
12547 		return -EINVAL;
12548 	}
12549 
12550 	/* All functions require bpf_list_head to be protected using a bpf_spin_lock */
12551 	if (check_reg_allocation_locked(env, reg)) {
12552 		verbose(env, "bpf_spin_lock at off=%d must be held for %s\n",
12553 			rec->spin_lock_off, head_type_name);
12554 		return -EINVAL;
12555 	}
12556 
12557 	if (*head_field) {
12558 		verifier_bug(env, "repeating %s arg", head_type_name);
12559 		return -EFAULT;
12560 	}
12561 	*head_field = field;
12562 	return 0;
12563 }
12564 
12565 static int process_kf_arg_ptr_to_list_head(struct bpf_verifier_env *env,
12566 					   struct bpf_reg_state *reg, argno_t argno,
12567 					   struct bpf_call_arg_meta *meta)
12568 {
12569 	return __process_kf_arg_ptr_to_graph_root(env, reg, argno, meta, BPF_LIST_HEAD,
12570 							  &meta->arg_list_head.field);
12571 }
12572 
12573 static int process_kf_arg_ptr_to_rbtree_root(struct bpf_verifier_env *env,
12574 					     struct bpf_reg_state *reg, argno_t argno,
12575 					     struct bpf_call_arg_meta *meta)
12576 {
12577 	return __process_kf_arg_ptr_to_graph_root(env, reg, argno, meta, BPF_RB_ROOT,
12578 							  &meta->arg_rbtree_root.field);
12579 }
12580 
12581 static int
12582 __process_kf_arg_ptr_to_graph_node(struct bpf_verifier_env *env,
12583 				   struct bpf_reg_state *reg, argno_t argno,
12584 				   struct bpf_call_arg_meta *meta,
12585 				   enum btf_field_type head_field_type,
12586 				   enum btf_field_type node_field_type,
12587 				   struct btf_field **node_field)
12588 {
12589 	const char *node_type_name;
12590 	const struct btf_type *et, *t;
12591 	struct btf_field *field;
12592 	u32 node_off;
12593 
12594 	if (meta->btf != btf_vmlinux) {
12595 		verifier_bug(env, "unexpected btf mismatch in kfunc call");
12596 		return -EFAULT;
12597 	}
12598 
12599 	if (!check_kfunc_is_graph_node_api(env, node_field_type, meta->func_id))
12600 		return -EFAULT;
12601 
12602 	node_type_name = btf_field_type_name(node_field_type);
12603 	if (!tnum_is_const(reg->var_off)) {
12604 		verbose(env,
12605 			"%s doesn't have constant offset. %s has to be at the constant offset\n",
12606 			reg_arg_name(env, argno), node_type_name);
12607 		return -EINVAL;
12608 	}
12609 
12610 	node_off = reg->var_off.value;
12611 	field = reg_find_field_offset(reg, node_off, node_field_type);
12612 	if (!field) {
12613 		verbose(env, "%s not found at offset=%u\n", node_type_name, node_off);
12614 		return -EINVAL;
12615 	}
12616 
12617 	field = *node_field;
12618 
12619 	et = btf_type_by_id(field->graph_root.btf, field->graph_root.value_btf_id);
12620 	t = btf_type_by_id(reg->btf, reg->btf_id);
12621 	if (!btf_struct_ids_match(&env->log, reg->btf, reg->btf_id, 0, field->graph_root.btf,
12622 				  field->graph_root.value_btf_id, true,
12623 				  !type_is_alloc(reg->type))) {
12624 		verbose(env, "operation on %s expects arg#1 %s at offset=%d "
12625 			"in struct %s, but arg is at offset=%d in struct %s\n",
12626 			btf_field_type_name(head_field_type),
12627 			btf_field_type_name(node_field_type),
12628 			field->graph_root.node_offset,
12629 			btf_name_by_offset(field->graph_root.btf, et->name_off),
12630 			node_off, btf_name_by_offset(reg->btf, t->name_off));
12631 		return -EINVAL;
12632 	}
12633 	meta->arg_btf = reg->btf;
12634 	meta->arg_btf_id = reg->btf_id;
12635 
12636 	if (node_off != field->graph_root.node_offset) {
12637 		verbose(env, "arg#1 offset=%d, but expected %s at offset=%d in struct %s\n",
12638 			node_off, btf_field_type_name(node_field_type),
12639 			field->graph_root.node_offset,
12640 			btf_name_by_offset(field->graph_root.btf, et->name_off));
12641 		return -EINVAL;
12642 	}
12643 
12644 	return 0;
12645 }
12646 
12647 static int process_kf_arg_ptr_to_list_node(struct bpf_verifier_env *env,
12648 					   struct bpf_reg_state *reg, argno_t argno,
12649 					   struct bpf_call_arg_meta *meta)
12650 {
12651 	return __process_kf_arg_ptr_to_graph_node(env, reg, argno, meta,
12652 						  BPF_LIST_HEAD, BPF_LIST_NODE,
12653 						  &meta->arg_list_head.field);
12654 }
12655 
12656 static int process_kf_arg_ptr_to_rbtree_node(struct bpf_verifier_env *env,
12657 					     struct bpf_reg_state *reg, argno_t argno,
12658 					     struct bpf_call_arg_meta *meta)
12659 {
12660 	return __process_kf_arg_ptr_to_graph_node(env, reg, argno, meta,
12661 						  BPF_RB_ROOT, BPF_RB_NODE,
12662 						  &meta->arg_rbtree_root.field);
12663 }
12664 
12665 /*
12666  * css_task iter allowlist is needed to avoid dead locking on css_set_lock.
12667  * LSM hooks and iters (both sleepable and non-sleepable) are safe.
12668  * Any sleepable progs are also safe since bpf_check_attach_target() enforce
12669  * them can only be attached to some specific hook points.
12670  */
12671 static bool check_css_task_iter_allowlist(struct bpf_verifier_env *env)
12672 {
12673 	enum bpf_prog_type prog_type = resolve_prog_type(env->prog);
12674 
12675 	switch (prog_type) {
12676 	case BPF_PROG_TYPE_LSM:
12677 		return true;
12678 	case BPF_PROG_TYPE_TRACING:
12679 		if (env->prog->expected_attach_type == BPF_TRACE_ITER)
12680 			return true;
12681 		fallthrough;
12682 	default:
12683 		return in_sleepable(env);
12684 	}
12685 }
12686 
12687 static int check_kfunc_args(struct bpf_verifier_env *env, struct bpf_call_arg_meta *meta,
12688 			    int insn_idx)
12689 {
12690 	const char *func_name = meta->func_name, *ref_tname;
12691 	struct bpf_func_state *caller = cur_func(env);
12692 	struct bpf_reg_state *regs = cur_regs(env);
12693 	const struct btf *btf = meta->btf;
12694 	const struct btf_param *args;
12695 	struct btf_record *rec;
12696 	u32 i, nargs;
12697 	int ret;
12698 
12699 	args = (const struct btf_param *)(meta->func_proto + 1);
12700 	nargs = btf_type_vlen(meta->func_proto);
12701 
12702 	ret = check_outgoing_stack_args(env, caller, nargs, func_name, btf, args);
12703 	if (ret)
12704 		return ret;
12705 
12706 	/* Check that BTF function arguments match actual types that the
12707 	 * verifier sees.
12708 	 */
12709 	for (i = 0; i < nargs; i++) {
12710 		struct bpf_reg_state *reg = get_func_arg_reg(caller, regs, i);
12711 		const struct btf_type *t, *ref_t, *resolve_ret;
12712 		enum bpf_arg_type arg_type = ARG_DONTCARE;
12713 		argno_t argno = argno_from_arg(i + 1);
12714 		int regno = reg_from_argno(argno);
12715 		bool btf_id_fixed_off_ok = true;
12716 		u32 ref_id = args[i].type, type_size;
12717 		int kf_arg_type = meta->fn->arg_type[i];
12718 
12719 		if (is_kfunc_arg_prog_aux(btf, &args[i])) {
12720 			/* Reject repeated use bpf_prog_aux */
12721 			if (meta->arg_prog) {
12722 				verifier_bug(env, "Only 1 prog->aux argument supported per-kfunc");
12723 				return -EFAULT;
12724 			}
12725 			if (regno < 0) {
12726 				verbose(env, "%s prog->aux cannot be a stack argument\n",
12727 					reg_arg_name(env, argno));
12728 				return -EINVAL;
12729 			}
12730 			meta->arg_prog = true;
12731 			cur_aux(env)->arg_prog = regno;
12732 			continue;
12733 		}
12734 
12735 		if (is_kfunc_arg_ignore(btf, &args[i]) || is_kfunc_arg_implicit(meta, i))
12736 			continue;
12737 
12738 		t = btf_type_skip_modifiers(btf, args[i].type, NULL);
12739 
12740 		if (btf_type_is_ptr(t)) {
12741 			ref_t = btf_type_skip_modifiers(btf, t->type, &ref_id);
12742 			ref_tname = btf_name_by_offset(btf, ref_t->name_off);
12743 		}
12744 
12745 		if (btf_type_is_ptr(t) &&
12746 		    (bpf_register_is_null(reg) || type_may_be_null(reg->type)) &&
12747 		    !type_may_be_null(kf_arg_type)) {
12748 			const char *expected_type;
12749 
12750 			expected_type = bpf_diag_fmt_btf_type(env, btf, args[i].type);
12751 			verbose(env, "Possibly NULL pointer passed to trusted %s\n",
12752 				reg_arg_name(env, argno));
12753 			bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name,
12754 					      "Add a NULL check and call the kfunc only on the non-NULL path.",
12755 					      "the pointer may be NULL, but this kfunc requires a non-NULL value of type %s",
12756 					      expected_type);
12757 			return -EACCES;
12758 		}
12759 
12760 		if (regno == meta->release_regno && !is_kfunc_arg_dynptr(meta->btf, &args[i]) &&
12761 		    !reg_is_referenced(env, reg) && !bpf_register_is_null(reg)) {
12762 			const char *expected_type;
12763 
12764 			expected_type = bpf_diag_fmt_btf_type(env, btf, ref_id);
12765 			verbose(env, "release kfunc %s expects referenced PTR_TO_BTF_ID passed to %s\n",
12766 				func_name, reg_arg_name(env, argno));
12767 			bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name,
12768 					      "Pass the resource-owning pointer returned by the matching acquire kfunc, and avoid calling the release kfunc after ownership has already been transferred or released.",
12769 					      "release kfuncs require a resource-owning value of type %s returned by a matching acquire kfunc",
12770 					      expected_type);
12771 			return -EINVAL;
12772 		}
12773 
12774 		if (reg_is_referenced(env, reg))
12775 			update_ref_obj(&meta->ref_obj, reg);
12776 
12777 		if (bpf_register_is_null(reg) && type_may_be_null(kf_arg_type)) {
12778 			ret = mark_arg_precision(env, argno);
12779 			if (ret)
12780 				return ret;
12781 			continue;
12782 		}
12783 
12784 		if (is_kfunc_arg_map(btf, &args[i])) {
12785 			ref_id = *reg2btf_ids[CONST_PTR_TO_MAP];
12786 			ref_t = btf_type_by_id(btf_vmlinux, ref_id);
12787 			ref_tname = btf_name_by_offset(btf, ref_t->name_off);
12788 		}
12789 
12790 		switch (base_type(kf_arg_type)) {
12791 		case KF_ARG_CONST:
12792 		case KF_ARG_CONST_MEM_SIZE:
12793 		case KF_ARG_MEM_SIZE:
12794 		case KF_ARG_ANYTHING:
12795 		case KF_ARG_CONST_ALLOC_SIZE_OR_ZERO:
12796 		case KF_ARG_PTR_TO_ALLOC_BTF_ID:
12797 		case KF_ARG_PTR_TO_BTF_ID:
12798 		case KF_ARG_CONST_MAP_PTR:
12799 		case KF_ARG_PTR_TO_ITER:
12800 		case KF_ARG_PTR_TO_LIST_HEAD:
12801 		case KF_ARG_PTR_TO_LIST_NODE:
12802 		case KF_ARG_PTR_TO_RB_ROOT:
12803 		case KF_ARG_PTR_TO_RB_NODE:
12804 		case KF_ARG_PTR_TO_MEM:
12805 		case KF_ARG_PTR_TO_CALLBACK:
12806 		case KF_ARG_PTR_TO_CONST_STR:
12807 		case KF_ARG_PTR_TO_WORKQUEUE:
12808 		case KF_ARG_PTR_TO_TIMER:
12809 		case KF_ARG_PTR_TO_TASK_WORK:
12810 		case KF_ARG_PTR_TO_IRQ_FLAG:
12811 		case KF_ARG_PTR_TO_RES_SPIN_LOCK:
12812 		case KF_ARG_PTR_TO_ARENA:
12813 			break;
12814 		case KF_ARG_PTR_TO_DYNPTR:
12815 			arg_type = ARG_PTR_TO_DYNPTR;
12816 			break;
12817 		case KF_ARG_PTR_TO_CTX:
12818 			arg_type = ARG_PTR_TO_CTX;
12819 			break;
12820 		case KF_ARG_PTR_TO_REFCOUNTED_KPTR:
12821 			arg_type = ARG_PTR_TO_BTF_ID;
12822 			btf_id_fixed_off_ok = false;
12823 			break;
12824 		default:
12825 			verifier_bug(env, "unknown kfunc arg type %d", kf_arg_type);
12826 			return -EFAULT;
12827 		}
12828 
12829 		if (regno == meta->release_regno)
12830 			arg_type |= OBJ_RELEASE;
12831 		ret = __check_func_arg_reg_off(env, reg, argno, arg_type,
12832 					       btf_id_fixed_off_ok);
12833 		if (ret < 0)
12834 			return ret;
12835 
12836 		switch (base_type(kf_arg_type)) {
12837 		case KF_ARG_CONST:
12838 			if (reg->type != SCALAR_VALUE) {
12839 				verbose(env, "%s is not a scalar\n", reg_arg_name(env, argno));
12840 				bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name,
12841 						      "Pass an integer scalar value for this argument, not a pointer or resource object.",
12842 						      "the kfunc expects an integer scalar, but %s is %s",
12843 						      reg_arg_name(env, argno),
12844 						      bpf_diag_reg_type_plain(env, reg->type));
12845 				return -EINVAL;
12846 			}
12847 
12848 			ret = process_const_arg(env, reg, argno, meta);
12849 			if (ret < 0) {
12850 				if (ret == -EINVAL)
12851 					bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name,
12852 							      "Pass a compile-time constant or a value the verifier can prove is constant at this call.",
12853 							      "the kfunc requires this scalar argument to be a verifier-known constant, but %s is variable on this path",
12854 							      reg_arg_name(env, argno));
12855 				return ret;
12856 			}
12857 			break;
12858 		case KF_ARG_ANYTHING:
12859 			if (reg->type != SCALAR_VALUE) {
12860 				verbose(env, "%s is not a scalar\n", reg_arg_name(env, argno));
12861 				bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name,
12862 						      "Pass an integer scalar value for this argument, not a pointer or resource object.",
12863 						      "the kfunc expects an integer scalar, but %s is %s",
12864 						      reg_arg_name(env, argno),
12865 						      bpf_diag_reg_type_plain(env, reg->type));
12866 				return -EINVAL;
12867 			}
12868 			break;
12869 		case KF_ARG_CONST_ALLOC_SIZE_OR_ZERO:
12870 			if (reg->type != SCALAR_VALUE) {
12871 				verbose(env, "%s is not a scalar\n", reg_arg_name(env, argno));
12872 				bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name,
12873 						      "Pass an integer scalar value for this argument, not a pointer or resource object.",
12874 						      "the kfunc expects an integer scalar, but %s is %s",
12875 						      reg_arg_name(env, argno),
12876 						      bpf_diag_reg_type_plain(env, reg->type));
12877 				return -EINVAL;
12878 			}
12879 
12880 			if (is_kfunc_arg_scalar_with_name(btf, &args[i], "rdonly_buf_size"))
12881 				meta->r0_rdonly = true;
12882 			ret = process_const_alloc_mem_size(env, reg, argno, &meta->ret_mem);
12883 			if (ret < 0) {
12884 				if (ret == -EINVAL)
12885 					bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name,
12886 							      "Pass a verifier-known constant size for this kfunc buffer argument.",
12887 							      "the kfunc uses this argument as a return-buffer size, but %s is invalid or variable on this path",
12888 							      reg_arg_name(env, argno));
12889 				return ret;
12890 			}
12891 			break;
12892 		case KF_ARG_PTR_TO_CTX:
12893 			if (reg->type != PTR_TO_CTX) {
12894 				verbose(env, "%s expected pointer to ctx, but got %s\n",
12895 					reg_arg_name(env, argno), reg_type_str(env, reg->type));
12896 				bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name,
12897 						      "Pass the original program context pointer or preserve it before modifying registers.",
12898 						      "the kfunc expects a context pointer, but %s is %s",
12899 						      reg_arg_name(env, argno),
12900 						      bpf_diag_reg_type_plain(env, reg->type));
12901 				return -EINVAL;
12902 			}
12903 
12904 			if (meta->func_id == special_kfunc_list[KF_bpf_cast_to_kern_ctx]) {
12905 				ret = get_kern_ctx_btf_id(&env->log, resolve_prog_type(env->prog));
12906 				if (ret < 0)
12907 					return -EINVAL;
12908 				meta->ret_btf_id  = ret;
12909 			}
12910 			break;
12911 		case KF_ARG_PTR_TO_ARENA:
12912 			if (reg->type != PTR_TO_ARENA && reg->type != SCALAR_VALUE) {
12913 				verbose(env, "%s is not a pointer to arena or scalar\n",
12914 					reg_arg_name(env, argno));
12915 				return -EINVAL;
12916 			}
12917 			break;
12918 		case KF_ARG_PTR_TO_ALLOC_BTF_ID:
12919 			if (reg->type == (PTR_TO_BTF_ID | MEM_ALLOC)) {
12920 				if (!is_bpf_obj_drop_kfunc(meta->func_id)) {
12921 					verbose(env, "%s expected for bpf_obj_drop()\n",
12922 						reg_arg_name(env, argno));
12923 					return -EINVAL;
12924 				}
12925 			} else if (reg->type == (PTR_TO_BTF_ID | MEM_ALLOC | MEM_PERCPU)) {
12926 				if (!is_bpf_percpu_obj_drop_kfunc(meta->func_id)) {
12927 					verbose(env, "%s expected for bpf_percpu_obj_drop()\n",
12928 						reg_arg_name(env, argno));
12929 					return -EINVAL;
12930 				}
12931 			} else {
12932 				verbose(env, "%s expected pointer to allocated object\n",
12933 					reg_arg_name(env, argno));
12934 				bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name,
12935 						      "Pass a pointer returned by the matching BPF object allocation path.",
12936 						      "the kfunc expects an allocated object pointer, but %s is %s",
12937 						      reg_arg_name(env, argno),
12938 						      bpf_diag_reg_type_plain(env, reg->type));
12939 				return -EINVAL;
12940 			}
12941 			if (!reg_is_referenced(env, reg)) {
12942 				verbose(env, "allocated object must be referenced\n");
12943 				bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name,
12944 						      "Pass the owned object pointer before it is released or transferred.",
12945 						      "the allocated object pointer in %s must still carry verifier-tracked ownership, but this pointer no longer owns a live resource",
12946 						      reg_arg_name(env, argno));
12947 				return -EINVAL;
12948 			}
12949 			if (meta->btf == btf_vmlinux) {
12950 				meta->arg_btf = reg->btf;
12951 				meta->arg_btf_id = reg->btf_id;
12952 			}
12953 			break;
12954 		case KF_ARG_PTR_TO_DYNPTR:
12955 		{
12956 			enum bpf_arg_type dynptr_arg_type = ARG_PTR_TO_DYNPTR;
12957 
12958 			if (is_kfunc_arg_uninit(btf, &args[i]))
12959 				dynptr_arg_type |= MEM_UNINIT;
12960 
12961 			if (meta->func_id == special_kfunc_list[KF_bpf_dynptr_from_skb]) {
12962 				dynptr_arg_type |= DYNPTR_TYPE_SKB;
12963 			} else if (meta->func_id == special_kfunc_list[KF_bpf_dynptr_from_xdp]) {
12964 				dynptr_arg_type |= DYNPTR_TYPE_XDP;
12965 			} else if (meta->func_id == special_kfunc_list[KF_bpf_dynptr_from_skb_meta]) {
12966 				dynptr_arg_type |= DYNPTR_TYPE_SKB_META;
12967 			} else if (meta->func_id == special_kfunc_list[KF_bpf_dynptr_from_file]) {
12968 				dynptr_arg_type |= DYNPTR_TYPE_FILE;
12969 			} else if (meta->func_id == special_kfunc_list[KF_bpf_dynptr_file_discard]) {
12970 				dynptr_arg_type |= DYNPTR_TYPE_FILE | OBJ_RELEASE;
12971 			} else if (meta->func_id == special_kfunc_list[KF_bpf_dynptr_clone] &&
12972 				   (dynptr_arg_type & MEM_UNINIT)) {
12973 				enum bpf_dynptr_type parent_type = meta->dynptr.type;
12974 
12975 				if (parent_type == BPF_DYNPTR_TYPE_INVALID) {
12976 					verifier_bug(env, "no dynptr type for parent of clone");
12977 					return -EFAULT;
12978 				}
12979 
12980 				dynptr_arg_type |= (unsigned int)get_dynptr_type_flag(parent_type);
12981 			}
12982 
12983 			ret = process_dynptr_func(env, reg, argno, insn_idx, func_name,
12984 						  dynptr_arg_type, &meta->ref_obj, &meta->dynptr);
12985 			if (ret < 0)
12986 				return ret;
12987 			break;
12988 		}
12989 		case KF_ARG_PTR_TO_ITER:
12990 			if (meta->func_id == special_kfunc_list[KF_bpf_iter_css_task_new]) {
12991 				if (!check_css_task_iter_allowlist(env)) {
12992 					verbose(env, "css_task_iter is only allowed in bpf_lsm, bpf_iter and sleepable progs\n");
12993 					return -EINVAL;
12994 				}
12995 			}
12996 			ret = process_iter_arg(env, reg, argno, insn_idx, meta);
12997 			if (ret < 0)
12998 				return ret;
12999 			break;
13000 		case KF_ARG_PTR_TO_LIST_HEAD:
13001 			if (reg->type != PTR_TO_MAP_VALUE &&
13002 			    reg->type != (PTR_TO_BTF_ID | MEM_ALLOC)) {
13003 				verbose(env, "%s expected pointer to map value or allocated object\n",
13004 					reg_arg_name(env, argno));
13005 				return -EINVAL;
13006 			}
13007 			if (reg->type == (PTR_TO_BTF_ID | MEM_ALLOC) &&
13008 			    !reg_is_referenced(env, reg)) {
13009 				verbose(env, "allocated object must be referenced\n");
13010 				return -EINVAL;
13011 			}
13012 			ret = process_kf_arg_ptr_to_list_head(env, reg, argno, meta);
13013 			if (ret < 0)
13014 				return ret;
13015 			break;
13016 		case KF_ARG_PTR_TO_RB_ROOT:
13017 			if (reg->type != PTR_TO_MAP_VALUE &&
13018 			    reg->type != (PTR_TO_BTF_ID | MEM_ALLOC)) {
13019 				verbose(env, "%s expected pointer to map value or allocated object\n",
13020 					reg_arg_name(env, argno));
13021 				return -EINVAL;
13022 			}
13023 			if (reg->type == (PTR_TO_BTF_ID | MEM_ALLOC) &&
13024 			    !reg_is_referenced(env, reg)) {
13025 				verbose(env, "allocated object must be referenced\n");
13026 				return -EINVAL;
13027 			}
13028 			ret = process_kf_arg_ptr_to_rbtree_root(env, reg, argno, meta);
13029 			if (ret < 0)
13030 				return ret;
13031 			break;
13032 		case KF_ARG_PTR_TO_LIST_NODE:
13033 			if (is_kfunc_arg_nonown_allowed(btf, &args[i]) &&
13034 			    type_is_non_owning_ref(reg->type) && !reg_is_referenced(env, reg)) {
13035 				/* Allow bpf_list_front/back return value for
13036 				 * __nonown_allowed list-node arguments.
13037 				 */
13038 				goto check_ok;
13039 			}
13040 			if (reg->type != (PTR_TO_BTF_ID | MEM_ALLOC)) {
13041 				verbose(env, "%s expected pointer to allocated object\n",
13042 					reg_arg_name(env, argno));
13043 				return -EINVAL;
13044 			}
13045 			if (!reg_is_referenced(env, reg)) {
13046 				verbose(env, "allocated object must be referenced\n");
13047 				return -EINVAL;
13048 			}
13049 check_ok:
13050 			ret = process_kf_arg_ptr_to_list_node(env, reg, argno, meta);
13051 			if (ret < 0)
13052 				return ret;
13053 			break;
13054 		case KF_ARG_PTR_TO_RB_NODE:
13055 			if (is_bpf_rbtree_add_kfunc(meta->func_id)) {
13056 				if (reg->type != (PTR_TO_BTF_ID | MEM_ALLOC)) {
13057 					verbose(env, "%s expected pointer to allocated object\n",
13058 						reg_arg_name(env, argno));
13059 					return -EINVAL;
13060 				}
13061 				if (!reg_is_referenced(env, reg)) {
13062 					verbose(env, "allocated object must be referenced\n");
13063 					return -EINVAL;
13064 				}
13065 			} else {
13066 				if (!type_is_non_owning_ref(reg->type) &&
13067 				    !reg_is_referenced(env, reg)) {
13068 					verbose(env, "%s can only take non-owning or refcounted bpf_rb_node pointer\n", func_name);
13069 					return -EINVAL;
13070 				}
13071 				if (in_rbtree_lock_required_cb(env)) {
13072 					verbose(env, "%s not allowed in rbtree cb\n", func_name);
13073 					return -EINVAL;
13074 				}
13075 			}
13076 
13077 			ret = process_kf_arg_ptr_to_rbtree_node(env, reg, argno, meta);
13078 			if (ret < 0)
13079 				return ret;
13080 			break;
13081 		case KF_ARG_CONST_MAP_PTR:
13082 			if (base_type(reg->type) != CONST_PTR_TO_MAP ||
13083 			    type_may_be_null(reg->type)) {
13084 				verbose(env, "pointer in %s isn't map pointer\n",
13085 					reg_arg_name(env, argno));
13086 				return -EINVAL;
13087 			}
13088 			ret = process_map_ptr_arg(env, reg, argno, meta);
13089 			if (ret < 0)
13090 				return ret;
13091 			break;
13092 		case KF_ARG_PTR_TO_BTF_ID:
13093 			/* Only base_type is checked, further checks are done here */
13094 			if (base_type(reg->type) == PTR_TO_BTF_ID ||
13095 			    reg2btf_ids[base_type(reg->type)]) {
13096 				if (!is_trusted_reg(env, reg) ||
13097 				    bpf_type_has_unsafe_modifiers(reg->type)) {
13098 					if (!is_kfunc_rcu(meta)) {
13099 						const char *expected_type;
13100 
13101 						expected_type = bpf_diag_fmt_btf_type(env, btf, ref_id);
13102 						verbose(env, "%s must be referenced or trusted\n",
13103 							reg_arg_name(env, argno));
13104 						bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name,
13105 								      "Pass a pointer acquired from a verifier-tracked source, or call this kfunc only inside the required protection if it accepts RCU pointers.",
13106 								      "the kfunc requires a trusted or resource-owning pointer to %s, but %s is %s",
13107 								      expected_type,
13108 								      reg_arg_name(env, argno),
13109 								      bpf_diag_reg_type_plain(env, reg->type));
13110 						return -EINVAL;
13111 					}
13112 					if (!is_rcu_reg(reg)) {
13113 						const char *expected_type;
13114 
13115 						expected_type = bpf_diag_fmt_btf_type(env, btf, ref_id);
13116 						verbose(env, "%s must be a rcu pointer\n",
13117 							reg_arg_name(env, argno));
13118 						bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name,
13119 								      "Use this kfunc with a pointer that is valid in an RCU read lock region.",
13120 								      "the kfunc requires an RCU-protected pointer to %s, but %s is %s",
13121 								      expected_type,
13122 								      reg_arg_name(env, argno),
13123 								      bpf_diag_reg_type_plain(env, reg->type));
13124 						return -EINVAL;
13125 					}
13126 				}
13127 
13128 				ret = process_kf_arg_ptr_to_btf_id(env, reg, ref_t, ref_tname, ref_id, meta, i, argno);
13129 				if (ret < 0)
13130 					return ret;
13131 				break;
13132 			}
13133 
13134 			if (!__btf_type_is_scalar_struct(env, meta->btf, ref_t, 0)) {
13135 				enum bpf_reg_type reg2btf_type = lookup_reg2btf_ids(ref_id);
13136 				const char *expected_type;
13137 
13138 				verbose(env, "%s is %s expected %s %s",
13139 					reg_arg_name(env, argno), reg_type_str(env, reg->type),
13140 					btf_type_str(ref_t), ref_tname);
13141 				if (reg2btf_type != NOT_INIT)
13142 					verbose(env, " or %s", reg_type_str(env, reg2btf_type));
13143 				verbose(env, "\n");
13144 				expected_type = bpf_diag_fmt_btf_type(env, btf, ref_id);
13145 				bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name,
13146 						      "Pass a verifier-tracked pointer to the expected kernel object type, not a pointer to stack storage or another memory buffer.",
13147 						      "the kfunc expects a pointer to %s, but this argument is %s and cannot be used as that kernel object pointer",
13148 						      expected_type,
13149 						      bpf_diag_reg_type_plain(env, reg->type));
13150 				return -EINVAL;
13151 			}
13152 
13153 			/*
13154 			 * If the register does not contain btf id but the argument type is a pointer to
13155 			 * scalar-only struct, allow verifying it as a fixed size memory.
13156 			 */
13157 			kf_arg_type = KF_ARG_PTR_TO_MEM | MEM_FIXED_SIZE;
13158 			fallthrough;
13159 		case KF_ARG_PTR_TO_MEM:
13160 			if (kf_arg_type & MEM_FIXED_SIZE) {
13161 				bool known_memory;
13162 
13163 				resolve_ret = btf_resolve_size(btf, ref_t, &type_size);
13164 				if (IS_ERR(resolve_ret)) {
13165 					verbose(env, "%s reference type('%s %s') size cannot be determined: %ld\n",
13166 						reg_arg_name(env, argno), btf_type_str(ref_t),
13167 						ref_tname, PTR_ERR(resolve_ret));
13168 					return -EINVAL;
13169 				}
13170 				ret = check_mem_reg(env, reg, argno, type_size, BPF_READ | BPF_WRITE,
13171 						    meta, &known_memory);
13172 				if (ret < 0) {
13173 					const char *expected_type;
13174 
13175 					expected_type = bpf_diag_fmt_btf_type(env, btf, ref_id);
13176 					if (known_memory)
13177 						bpf_diag_call_arg_fmt(
13178 							env, insn_idx, argno, func_name,
13179 							"Pass memory with at least the required number of accessible bytes and suitable read and write access.",
13180 							"the kfunc expects %u bytes of memory for %s, but the verifier cannot prove that %s provides a readable and writable range of that size",
13181 							type_size, expected_type,
13182 							bpf_diag_reg_type_plain(env, reg->type));
13183 					else
13184 						bpf_diag_call_arg_fmt(
13185 							env, insn_idx, argno, func_name,
13186 							"Pass stack, map, context, or other verifier-known memory of the expected type and size, not an integer cast to a pointer.",
13187 							"the kfunc expects %u bytes of memory for %s, but it is %s and not verifier-known memory",
13188 							type_size, expected_type,
13189 							bpf_diag_reg_type_plain(env, reg->type));
13190 					return ret;
13191 				}
13192 			}
13193 			break;
13194 		case KF_ARG_CONST_MEM_SIZE:
13195 			ret = process_const_arg(env, reg, argno, meta);
13196 			if (ret < 0) {
13197 				if (ret == -EINVAL)
13198 					bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name,
13199 							      "Pass a compile-time constant or a value the verifier can prove is constant at this call.",
13200 							      "the kfunc requires this memory size to be a verifier-known constant, but %s is variable on this path",
13201 							      reg_arg_name(env, argno));
13202 				return ret;
13203 			}
13204 			fallthrough;
13205 		case KF_ARG_MEM_SIZE:
13206 		{
13207 			struct bpf_reg_state *buff_reg = get_func_arg_reg(caller, regs, i - 1);
13208 			struct bpf_reg_state *size_reg = reg;
13209 			argno_t buff_argno = argno_from_arg(i);
13210 			enum bpf_mem_size_failure failure;
13211 
13212 			if (reg->type != SCALAR_VALUE) {
13213 				verbose(env, "%s is not a scalar\n", reg_arg_name(env, argno));
13214 				bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name,
13215 						      "Pass an integer scalar length for this memory argument.",
13216 						      "the kfunc expects a scalar memory size, but %s is %s",
13217 						      reg_arg_name(env, argno),
13218 						      bpf_diag_reg_type_plain(env, reg->type));
13219 				return -EINVAL;
13220 			}
13221 
13222 			if (bpf_register_is_null(buff_reg))
13223 				break;
13224 
13225 			ret = check_mem_size_reg(env, buff_reg, size_reg, buff_argno, argno,
13226 						 BPF_READ | BPF_WRITE, true, meta, &failure);
13227 			if (ret < 0) {
13228 				const char *buff_arg, *size_arg;
13229 
13230 				buff_arg = bpf_diag_arg_name(env, buff_argno);
13231 				size_arg = bpf_diag_arg_name(env, argno);
13232 				verbose(env, "%s and ", reg_arg_name(env, buff_argno));
13233 				verbose(env, "%s memory, len pair leads to invalid memory access\n",
13234 					reg_arg_name(env, argno));
13235 				if (failure == BPF_MEM_SIZE_FAIL_MEMORY) {
13236 					bpf_diag_call_arg_fmt(env, insn_idx, buff_argno, func_name,
13237 							      "Pass a stack, map, context, or other verifier-known memory pointer, and keep the paired length within that object.",
13238 							      "it is the memory pointer in a memory/length pair with %s, but %s does not describe verifier-readable memory for the requested length",
13239 							      size_arg, buff_arg);
13240 				} else if (failure == BPF_MEM_SIZE_FAIL_SIZE) {
13241 					if (reg_smin(size_reg) < 0)
13242 						bpf_diag_call_arg_fmt(
13243 							env, insn_idx, argno, func_name,
13244 							"Constrain the memory size to a non-negative value smaller than BPF_MAX_VAR_SIZ before this call.",
13245 							"the memory size in %s may be negative because its signed minimum is %lld",
13246 							size_arg, reg_smin(size_reg));
13247 					else
13248 						bpf_diag_call_arg_fmt(
13249 							env, insn_idx, argno, func_name,
13250 							"Constrain the memory size to a non-negative value smaller than BPF_MAX_VAR_SIZ before this call.",
13251 							"the memory size in %s may reach %llu bytes, but variable memory accesses must stay below %u bytes",
13252 							size_arg, reg_umax(size_reg), BPF_MAX_VAR_SIZ);
13253 				}
13254 				return ret;
13255 			}
13256 			break;
13257 		}
13258 		case KF_ARG_PTR_TO_CALLBACK:
13259 			if (reg->type != PTR_TO_FUNC) {
13260 				verbose(env, "%s expected pointer to func\n", reg_arg_name(env, argno));
13261 				return -EINVAL;
13262 			}
13263 			meta->subprogno = reg->subprogno;
13264 			break;
13265 		case KF_ARG_PTR_TO_REFCOUNTED_KPTR:
13266 			if (!type_is_ptr_alloc_obj(reg->type)) {
13267 				verbose(env, "%s is neither owning or non-owning ref\n",
13268 					reg_arg_name(env, argno));
13269 				bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name,
13270 						      "Pass an owning or non-owning pointer to a BPF-managed object containing a bpf_refcount field.",
13271 						      "the kfunc expects a pointer to a BPF-managed refcounted object, but %s is %s",
13272 						      reg_arg_name(env, argno),
13273 						      bpf_diag_reg_type_plain(env, reg->type));
13274 				return -EINVAL;
13275 			}
13276 			if (!type_is_non_owning_ref(reg->type) && reg_is_referenced(env, reg))
13277 				meta->arg_owning_ref = true;
13278 
13279 			rec = reg_btf_record(reg);
13280 			if (!rec) {
13281 				verifier_bug(env, "Couldn't find btf_record");
13282 				return -EFAULT;
13283 			}
13284 
13285 			if (rec->refcount_off < 0) {
13286 				verbose(env, "%s doesn't point to a type with bpf_refcount field\n",
13287 					reg_arg_name(env, argno));
13288 				return -EINVAL;
13289 			}
13290 
13291 			meta->arg_btf = reg->btf;
13292 			meta->arg_btf_id = reg->btf_id;
13293 			break;
13294 		case KF_ARG_PTR_TO_CONST_STR:
13295 			if (reg->type != PTR_TO_MAP_VALUE) {
13296 				verbose(env, "%s doesn't point to a const string\n",
13297 					reg_arg_name(env, argno));
13298 				bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name,
13299 						      "Pass a constant string pointer that the verifier recognizes, such as a string stored in a read-only map value.",
13300 						      "the kfunc expects a pointer to a constant string stored in verifier-known memory, but %s is %s",
13301 						      reg_arg_name(env, argno),
13302 						      bpf_diag_reg_type_plain(env, reg->type));
13303 				return -EINVAL;
13304 			}
13305 			ret = check_arg_const_str(env, reg, argno);
13306 			if (ret)
13307 				return ret;
13308 			break;
13309 		case KF_ARG_PTR_TO_WORKQUEUE:
13310 			if (reg->type != PTR_TO_MAP_VALUE) {
13311 				verbose(env, "%s doesn't point to a map value\n",
13312 					reg_arg_name(env, argno));
13313 				return -EINVAL;
13314 			}
13315 			ret = check_map_field_pointer(env, reg, argno, BPF_WORKQUEUE, &meta->map);
13316 			if (ret < 0)
13317 				return ret;
13318 			break;
13319 		case KF_ARG_PTR_TO_TIMER:
13320 			if (reg->type != PTR_TO_MAP_VALUE) {
13321 				verbose(env, "%s doesn't point to a map value\n",
13322 					reg_arg_name(env, argno));
13323 				return -EINVAL;
13324 			}
13325 			ret = process_timer_func(env, reg, argno, &meta->map);
13326 			if (ret < 0)
13327 				return ret;
13328 			break;
13329 		case KF_ARG_PTR_TO_TASK_WORK:
13330 			if (reg->type != PTR_TO_MAP_VALUE) {
13331 				verbose(env, "%s doesn't point to a map value\n",
13332 					reg_arg_name(env, argno));
13333 				return -EINVAL;
13334 			}
13335 			ret = check_map_field_pointer(env, reg, argno, BPF_TASK_WORK, &meta->map);
13336 			if (ret < 0)
13337 				return ret;
13338 			break;
13339 		case KF_ARG_PTR_TO_IRQ_FLAG:
13340 			if (reg->type != PTR_TO_STACK) {
13341 				verbose(env, "%s doesn't point to an irq flag on stack\n",
13342 					reg_arg_name(env, argno));
13343 				bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name,
13344 						      "Pass the same stack slot used by bpf_local_irq_save() or bpf_res_spin_lock_irqsave().",
13345 						      "the kfunc expects a stack pointer to an IRQ flag slot, but %s is %s",
13346 						      reg_arg_name(env, argno),
13347 						      bpf_diag_reg_type_plain(env, reg->type));
13348 				return -EINVAL;
13349 			}
13350 			ret = process_irq_flag(env, reg, argno, meta);
13351 			if (ret < 0)
13352 				return ret;
13353 			break;
13354 		case KF_ARG_PTR_TO_RES_SPIN_LOCK:
13355 		{
13356 			int flags = PROCESS_RES_LOCK;
13357 
13358 			if (in_rbtree_lock_required_cb(env)) {
13359 				verbose(env, "can't res_spin_{lock,unlock} in rbtree cb\n");
13360 				return -EACCES;
13361 			}
13362 
13363 			if (reg->type != PTR_TO_MAP_VALUE && reg->type != (PTR_TO_BTF_ID | MEM_ALLOC)) {
13364 				verbose(env, "%s doesn't point to map value or allocated object\n",
13365 					reg_arg_name(env, argno));
13366 				return -EINVAL;
13367 			}
13368 
13369 			if (!is_bpf_res_spin_lock_kfunc(meta->func_id))
13370 				return -EFAULT;
13371 			if (meta->func_id == special_kfunc_list[KF_bpf_res_spin_lock] ||
13372 			    meta->func_id == special_kfunc_list[KF_bpf_res_spin_lock_irqsave])
13373 				flags |= PROCESS_SPIN_LOCK;
13374 			if (meta->func_id == special_kfunc_list[KF_bpf_res_spin_lock_irqsave] ||
13375 			    meta->func_id == special_kfunc_list[KF_bpf_res_spin_unlock_irqrestore])
13376 				flags |= PROCESS_LOCK_IRQ;
13377 			ret = process_spin_lock(env, reg, argno, flags);
13378 			if (ret < 0)
13379 				return ret;
13380 			break;
13381 		}
13382 		}
13383 	}
13384 
13385 	return 0;
13386 }
13387 
13388 int bpf_fetch_kfunc_arg_meta(struct bpf_verifier_env *env,
13389 			     s32 func_id,
13390 			     s16 offset,
13391 			     struct bpf_call_arg_meta *meta)
13392 {
13393 	struct bpf_kfunc_meta kfunc;
13394 	int err;
13395 
13396 	memset(meta, 0, sizeof(*meta));
13397 
13398 	err = fetch_kfunc_meta(env, func_id, offset, &kfunc);
13399 	if (err)
13400 		return err;
13401 
13402 	meta->btf = kfunc.btf;
13403 	meta->func_id = kfunc.id;
13404 	meta->func_proto = kfunc.proto;
13405 	meta->func_name = kfunc.name;
13406 
13407 	if (!kfunc.flags || !btf_kfunc_is_allowed(kfunc.btf, kfunc.id, env->prog))
13408 		return -EACCES;
13409 
13410 	meta->kfunc_flags = *kfunc.flags;
13411 
13412 	/* Only support release referenced argument passed by register */
13413 	if (is_kfunc_release(meta))
13414 		meta->release_regno = BPF_REG_1;
13415 
13416 	return 0;
13417 }
13418 
13419 /*
13420  * Determine how many bytes a helper accesses through a stack pointer at
13421  * argument position @arg (0-based, corresponding to R1-R5).
13422  *
13423  * Returns:
13424  *   > 0   known read access size in bytes
13425  *     0   doesn't read anything directly
13426  * S64_MIN unknown
13427  *   < 0   known write access of (-return) bytes
13428  */
13429 s64 bpf_helper_stack_access_bytes(struct bpf_verifier_env *env, struct bpf_insn *insn,
13430 				  int arg, int insn_idx)
13431 {
13432 	struct bpf_insn_aux_data *aux = &env->insn_aux_data[insn_idx];
13433 	const struct bpf_func_proto *fn;
13434 	enum bpf_arg_type at;
13435 	s64 size;
13436 
13437 	if (bpf_get_helper_proto(env, insn->imm, &fn) < 0)
13438 		return S64_MIN;
13439 
13440 	at = fn->arg_type[arg];
13441 
13442 	switch (base_type(at)) {
13443 	case ARG_PTR_TO_MAP_KEY:
13444 	case ARG_PTR_TO_MAP_VALUE: {
13445 		bool is_key = base_type(at) == ARG_PTR_TO_MAP_KEY;
13446 		u64 val;
13447 		int i, map_reg;
13448 
13449 		for (i = 0; i < arg; i++) {
13450 			if (base_type(fn->arg_type[i]) == ARG_CONST_MAP_PTR)
13451 				break;
13452 		}
13453 		if (i >= arg)
13454 			goto scan_all_maps;
13455 
13456 		map_reg = BPF_REG_1 + i;
13457 
13458 		if (!(aux->const_reg_map_mask & BIT(map_reg)))
13459 			goto scan_all_maps;
13460 
13461 		i = aux->const_reg_vals[map_reg];
13462 		if (i < env->used_map_cnt) {
13463 			size = is_key ? env->used_maps[i]->key_size
13464 				      : env->used_maps[i]->value_size;
13465 			goto out;
13466 		}
13467 scan_all_maps:
13468 		/*
13469 		 * Map pointer is not known at this call site (e.g. different
13470 		 * maps on merged paths).  Conservatively return the largest
13471 		 * key_size or value_size across all maps used by the program.
13472 		 */
13473 		val = 0;
13474 		for (i = 0; i < env->used_map_cnt; i++) {
13475 			struct bpf_map *map = env->used_maps[i];
13476 			u32 sz = is_key ? map->key_size : map->value_size;
13477 
13478 			if (sz > val)
13479 				val = sz;
13480 			if (map->inner_map_meta) {
13481 				sz = is_key ? map->inner_map_meta->key_size
13482 					    : map->inner_map_meta->value_size;
13483 				if (sz > val)
13484 					val = sz;
13485 			}
13486 		}
13487 		if (!val)
13488 			return S64_MIN;
13489 		size = val;
13490 		goto out;
13491 	}
13492 	case ARG_PTR_TO_MEM:
13493 		if (at & MEM_FIXED_SIZE) {
13494 			size = fn->arg_size[arg];
13495 			goto out;
13496 		}
13497 		if (arg + 1 < ARRAY_SIZE(fn->arg_type) &&
13498 		    arg_type_is_mem_size(fn->arg_type[arg + 1])) {
13499 			int size_reg = BPF_REG_1 + arg + 1;
13500 
13501 			if (aux->const_reg_mask & BIT(size_reg)) {
13502 				size = (s64)aux->const_reg_vals[size_reg];
13503 				goto out;
13504 			}
13505 			/*
13506 			 * Size arg is const on each path but differs across merged
13507 			 * paths. MAX_BPF_STACK is a safe upper bound for reads.
13508 			 */
13509 			if (at & MEM_UNINIT)
13510 				return 0;
13511 			return MAX_BPF_STACK;
13512 		}
13513 		return S64_MIN;
13514 	case ARG_PTR_TO_DYNPTR:
13515 		size = BPF_DYNPTR_SIZE;
13516 		break;
13517 	case ARG_PTR_TO_STACK:
13518 		/*
13519 		 * Only used by bpf_calls_callback() helpers. The helper itself
13520 		 * doesn't access stack. The callback subprog does and it's
13521 		 * analyzed separately.
13522 		 */
13523 		return 0;
13524 	default:
13525 		return S64_MIN;
13526 	}
13527 out:
13528 	/*
13529 	 * MEM_UNINIT args are write-only: the helper initializes the
13530 	 * buffer without reading it.
13531 	 */
13532 	if (at & MEM_UNINIT)
13533 		return -size;
13534 	return size;
13535 }
13536 
13537 /*
13538  * Determine how many bytes a kfunc accesses through a stack pointer at
13539  * argument position @arg (0-based, corresponding to R1-R5).
13540  *
13541  * Returns:
13542  *   > 0      known read access size in bytes
13543  *     0      doesn't access memory through that argument (ex: not a pointer)
13544  *   S64_MIN  unknown
13545  *   < 0      known write access of (-return) bytes
13546  */
13547 s64 bpf_kfunc_stack_access_bytes(struct bpf_verifier_env *env, struct bpf_insn *insn,
13548 				 int arg, int insn_idx)
13549 {
13550 	struct bpf_insn_aux_data *aux = &env->insn_aux_data[insn_idx];
13551 	struct bpf_call_arg_meta meta;
13552 	const struct btf_param *args;
13553 	const struct btf_type *t, *ref_t;
13554 	const struct btf *btf;
13555 	u32 nargs, type_size;
13556 	s64 size;
13557 
13558 	if (bpf_fetch_kfunc_arg_meta(env, insn->imm, insn->off, &meta) < 0)
13559 		return S64_MIN;
13560 
13561 	btf = meta.btf;
13562 	args = btf_params(meta.func_proto);
13563 	nargs = btf_type_vlen(meta.func_proto);
13564 	if (arg >= nargs)
13565 		return 0;
13566 
13567 	t = btf_type_skip_modifiers(btf, args[arg].type, NULL);
13568 	if (!btf_type_is_ptr(t))
13569 		return 0;
13570 
13571 	/* dynptr: fixed 16-byte on-stack representation */
13572 	if (is_kfunc_arg_dynptr(btf, &args[arg])) {
13573 		size = BPF_DYNPTR_SIZE;
13574 		goto out;
13575 	}
13576 
13577 	/* ptr + __sz/__szk pair: size is in the next register */
13578 	if (arg + 1 < nargs &&
13579 	    (btf_param_match_suffix(btf, &args[arg + 1], "__sz") ||
13580 	     btf_param_match_suffix(btf, &args[arg + 1], "__szk"))) {
13581 		int size_reg = BPF_REG_1 + arg + 1;
13582 
13583 		if (aux->const_reg_mask & BIT(size_reg)) {
13584 			size = (s64)aux->const_reg_vals[size_reg];
13585 			goto out;
13586 		}
13587 		return MAX_BPF_STACK;
13588 	}
13589 
13590 	/* fixed-size pointed-to type: resolve via BTF */
13591 	ref_t = btf_type_skip_modifiers(btf, t->type, NULL);
13592 	if (!IS_ERR(btf_resolve_size(btf, ref_t, &type_size))) {
13593 		size = type_size;
13594 		goto out;
13595 	}
13596 
13597 	return S64_MIN;
13598 out:
13599 	/* KF_ITER_NEW kfuncs initialize the iterator state at arg 0 */
13600 	if (arg == 0 && meta.kfunc_flags & KF_ITER_NEW)
13601 		return -size;
13602 	if (is_kfunc_arg_uninit(btf, &args[arg]))
13603 		return -size;
13604 	return size;
13605 }
13606 
13607 /* check special kfuncs and return:
13608  *  1  - not fall-through to 'else' branch, continue verification
13609  *  0  - fall-through to 'else' branch
13610  * < 0 - not fall-through to 'else' branch, return error
13611  */
13612 static int check_special_kfunc(struct bpf_verifier_env *env, struct bpf_call_arg_meta *meta,
13613 			       struct bpf_reg_state *regs, struct bpf_insn_aux_data *insn_aux,
13614 			       const struct btf_type *ptr_type, struct btf *desc_btf)
13615 {
13616 	const struct btf_type *ret_t;
13617 	int err = 0;
13618 
13619 	if (meta->btf != btf_vmlinux)
13620 		return 0;
13621 
13622 	if (is_bpf_obj_new_kfunc(meta->func_id) || is_bpf_percpu_obj_new_kfunc(meta->func_id)) {
13623 		struct btf_struct_meta *struct_meta;
13624 		struct btf *ret_btf;
13625 		u32 ret_btf_id;
13626 
13627 		if (is_bpf_obj_new_kfunc(meta->func_id) && !bpf_global_ma_set)
13628 			return -ENOMEM;
13629 
13630 		if (((u64)(u32)meta->arg_constant.value) != meta->arg_constant.value) {
13631 			verbose(env, "local type ID argument must be in range [0, U32_MAX]\n");
13632 			return -EINVAL;
13633 		}
13634 
13635 		ret_btf = env->prog->aux->btf;
13636 		ret_btf_id = meta->arg_constant.value;
13637 
13638 		/* This may be NULL due to user not supplying a BTF */
13639 		if (!ret_btf) {
13640 			verbose(env, "bpf_obj_new/bpf_percpu_obj_new requires prog BTF\n");
13641 			return -EINVAL;
13642 		}
13643 
13644 		ret_t = btf_type_by_id(ret_btf, ret_btf_id);
13645 		if (!ret_t || !__btf_type_is_struct(ret_t)) {
13646 			verbose(env, "bpf_obj_new/bpf_percpu_obj_new type ID argument must be of a struct\n");
13647 			return -EINVAL;
13648 		}
13649 
13650 		if (is_bpf_percpu_obj_new_kfunc(meta->func_id)) {
13651 			if (ret_t->size > BPF_GLOBAL_PERCPU_MA_MAX_SIZE) {
13652 				verbose(env, "bpf_percpu_obj_new type size (%d) is greater than %d\n",
13653 					ret_t->size, BPF_GLOBAL_PERCPU_MA_MAX_SIZE);
13654 				return -EINVAL;
13655 			}
13656 
13657 			if (!bpf_global_percpu_ma_set) {
13658 				mutex_lock(&bpf_percpu_ma_lock);
13659 				if (!bpf_global_percpu_ma_set) {
13660 					/* Charge memory allocated with bpf_global_percpu_ma to
13661 					 * root memcg. The obj_cgroup for root memcg is NULL.
13662 					 */
13663 					err = bpf_mem_alloc_percpu_init(&bpf_global_percpu_ma, NULL);
13664 					if (!err)
13665 						bpf_global_percpu_ma_set = true;
13666 				}
13667 				mutex_unlock(&bpf_percpu_ma_lock);
13668 				if (err)
13669 					return err;
13670 			}
13671 
13672 			mutex_lock(&bpf_percpu_ma_lock);
13673 			err = bpf_mem_alloc_percpu_unit_init(&bpf_global_percpu_ma, ret_t->size);
13674 			mutex_unlock(&bpf_percpu_ma_lock);
13675 			if (err)
13676 				return err;
13677 		}
13678 
13679 		struct_meta = btf_find_struct_meta(ret_btf, ret_btf_id);
13680 		if (is_bpf_percpu_obj_new_kfunc(meta->func_id)) {
13681 			if (!__btf_type_is_scalar_struct(env, ret_btf, ret_t, 0)) {
13682 				verbose(env, "bpf_percpu_obj_new type ID argument must be of a struct of scalars\n");
13683 				return -EINVAL;
13684 			}
13685 
13686 			if (struct_meta) {
13687 				verbose(env, "bpf_percpu_obj_new type ID argument must not contain special fields\n");
13688 				return -EINVAL;
13689 			}
13690 		}
13691 
13692 		mark_reg_known_zero(env, regs, BPF_REG_0);
13693 		regs[BPF_REG_0].type = PTR_TO_BTF_ID | MEM_ALLOC;
13694 		regs[BPF_REG_0].btf = ret_btf;
13695 		regs[BPF_REG_0].btf_id = ret_btf_id;
13696 		if (is_bpf_percpu_obj_new_kfunc(meta->func_id))
13697 			regs[BPF_REG_0].type |= MEM_PERCPU;
13698 
13699 		insn_aux->obj_new_size = ret_t->size;
13700 		insn_aux->kptr_struct_meta = struct_meta;
13701 	} else if (is_bpf_refcount_acquire_kfunc(meta->func_id)) {
13702 		mark_reg_known_zero(env, regs, BPF_REG_0);
13703 		regs[BPF_REG_0].type = PTR_TO_BTF_ID | MEM_ALLOC;
13704 		regs[BPF_REG_0].btf = meta->arg_btf;
13705 		regs[BPF_REG_0].btf_id = meta->arg_btf_id;
13706 
13707 		insn_aux->kptr_struct_meta =
13708 			btf_find_struct_meta(meta->arg_btf,
13709 					     meta->arg_btf_id);
13710 	} else if (is_list_node_type(ptr_type)) {
13711 		struct btf_field *field = meta->arg_list_head.field;
13712 
13713 		mark_reg_graph_node(regs, BPF_REG_0, &field->graph_root);
13714 	} else if (is_rbtree_node_type(ptr_type)) {
13715 		struct btf_field *field = meta->arg_rbtree_root.field;
13716 
13717 		mark_reg_graph_node(regs, BPF_REG_0, &field->graph_root);
13718 	} else if (meta->func_id == special_kfunc_list[KF_bpf_cast_to_kern_ctx]) {
13719 		mark_reg_known_zero(env, regs, BPF_REG_0);
13720 		regs[BPF_REG_0].type = PTR_TO_BTF_ID | PTR_TRUSTED;
13721 		regs[BPF_REG_0].btf = desc_btf;
13722 		regs[BPF_REG_0].btf_id = meta->ret_btf_id;
13723 	} else if (meta->func_id == special_kfunc_list[KF_bpf_rdonly_cast]) {
13724 		ret_t = btf_type_by_id(desc_btf, meta->arg_constant.value);
13725 		if (!ret_t) {
13726 			verbose(env, "Unknown type ID %lld passed to kfunc bpf_rdonly_cast\n",
13727 				meta->arg_constant.value);
13728 			return -EINVAL;
13729 		} else if (btf_type_is_struct(ret_t)) {
13730 			mark_reg_known_zero(env, regs, BPF_REG_0);
13731 			regs[BPF_REG_0].type = PTR_TO_BTF_ID | PTR_UNTRUSTED;
13732 			regs[BPF_REG_0].btf = desc_btf;
13733 			regs[BPF_REG_0].btf_id = meta->arg_constant.value;
13734 		} else if (btf_type_is_void(ret_t)) {
13735 			mark_reg_known_zero(env, regs, BPF_REG_0);
13736 			regs[BPF_REG_0].type = PTR_TO_MEM | MEM_RDONLY | PTR_UNTRUSTED;
13737 			regs[BPF_REG_0].mem_size = 0;
13738 		} else {
13739 			verbose(env,
13740 				"kfunc bpf_rdonly_cast type ID argument must be of a struct or void\n");
13741 			return -EINVAL;
13742 		}
13743 	} else if (meta->func_id == special_kfunc_list[KF_bpf_dynptr_slice] ||
13744 		   meta->func_id == special_kfunc_list[KF_bpf_dynptr_slice_rdwr]) {
13745 		enum bpf_type_flag type_flag = get_dynptr_type_flag(meta->dynptr.type);
13746 
13747 		mark_reg_known_zero(env, regs, BPF_REG_0);
13748 
13749 		if (!meta->arg_constant.found) {
13750 			verifier_bug(env, "bpf_dynptr_slice(_rdwr) no constant size");
13751 			return -EFAULT;
13752 		}
13753 
13754 		regs[BPF_REG_0].mem_size = meta->arg_constant.value;
13755 
13756 		/* PTR_MAYBE_NULL will be added when is_kfunc_ret_null is checked */
13757 		regs[BPF_REG_0].type = PTR_TO_MEM | type_flag;
13758 
13759 		if (meta->func_id == special_kfunc_list[KF_bpf_dynptr_slice]) {
13760 			regs[BPF_REG_0].type |= MEM_RDONLY;
13761 		} else {
13762 			/* this will set env->seen_direct_write to true */
13763 			if (!may_access_direct_pkt_data(env, NULL, BPF_WRITE)) {
13764 				verbose(env, "the prog does not allow writes to packet data\n");
13765 				return -EINVAL;
13766 			}
13767 		}
13768 
13769 		if (!meta->dynptr.id) {
13770 			verifier_bug(env, "no dynptr id");
13771 			return -EFAULT;
13772 		}
13773 		regs[BPF_REG_0].parent_id = meta->dynptr.id;
13774 	} else {
13775 		return 0;
13776 	}
13777 
13778 	return 1;
13779 }
13780 
13781 static int check_return_code(struct bpf_verifier_env *env, int regno, const char *reg_name);
13782 
13783 static int check_kfunc_call(struct bpf_verifier_env *env, struct bpf_insn *insn,
13784 			    int *insn_idx_p)
13785 {
13786 	bool sleepable, rcu_lock, rcu_unlock, preempt_disable, preempt_enable;
13787 	enum bpf_prog_type prog_type = resolve_prog_type(env->prog);
13788 	struct bpf_reg_state *regs = cur_regs(env);
13789 	const char *func_name, *ptr_type_name;
13790 	const struct btf_type *t, *ptr_type;
13791 	struct bpf_call_arg_meta meta;
13792 	struct bpf_insn_aux_data *insn_aux;
13793 	const char *operation;
13794 	int err, insn_idx = *insn_idx_p;
13795 	u32 i, nargs, ptr_type_id;
13796 	struct bpf_kfunc_desc *desc;
13797 	struct btf *desc_btf;
13798 	int id;
13799 
13800 	/* skip for now, but return error when we find this in fixup_kfunc_call */
13801 	if (!insn->imm)
13802 		return 0;
13803 
13804 	err = bpf_fetch_kfunc_arg_meta(env, insn->imm, insn->off, &meta);
13805 	if (err == -EACCES && meta.func_name) {
13806 		verbose(env, "calling kernel function %s is not allowed\n", meta.func_name);
13807 		operation = bpf_diag_fmt(env, "kfunc %s", meta.func_name);
13808 		bpf_diag_policy(
13809 			env, insn_idx, operation, "this program cannot call the kfunc",
13810 			"Use a kfunc allowed for this program type and attach point, or change the program context.");
13811 	}
13812 	if (err)
13813 		return err;
13814 	desc_btf = meta.btf;
13815 	func_name = meta.func_name;
13816 	insn_aux = &env->insn_aux_data[insn_idx];
13817 
13818 	desc = find_kfunc_desc(env->prog, insn->imm, insn->off);
13819 	if (!desc) {
13820 		verifier_bug(env, "kfunc descriptor not found for func_id %u", insn->imm);
13821 		return -EFAULT;
13822 	}
13823 	meta.fn = &desc->proto;
13824 
13825 	insn_aux->is_iter_next = bpf_is_iter_next_kfunc(&meta);
13826 
13827 	if (!insn->off &&
13828 	    (insn->imm == special_kfunc_list[KF_bpf_res_spin_lock] ||
13829 	     insn->imm == special_kfunc_list[KF_bpf_res_spin_lock_irqsave])) {
13830 		struct bpf_verifier_state *branch;
13831 		struct bpf_reg_state *regs;
13832 
13833 		branch = push_stack(env, env->insn_idx + 1, env->insn_idx, false);
13834 		if (IS_ERR(branch)) {
13835 			verbose(env, "failed to push state for failed lock acquisition\n");
13836 			return PTR_ERR(branch);
13837 		}
13838 
13839 		regs = branch->frame[branch->curframe]->regs;
13840 
13841 		/* Clear r0-r5 registers in forked state */
13842 		for (i = 0; i < CALLER_SAVED_REGS; i++)
13843 			bpf_mark_reg_not_init(env, &regs[caller_saved[i]]);
13844 
13845 		mark_reg_unknown(env, regs, BPF_REG_0);
13846 		err = __mark_reg_s32_range(env, regs, BPF_REG_0, -MAX_ERRNO, -1);
13847 		if (err) {
13848 			verbose(env, "failed to mark s32 range for retval in forked state for lock\n");
13849 			return err;
13850 		}
13851 	} else if (!insn->off && insn->imm == special_kfunc_list[KF___bpf_trap]) {
13852 		verbose(env, "unexpected __bpf_trap() due to uninitialized variable?\n");
13853 		return -EFAULT;
13854 	}
13855 
13856 	if (is_kfunc_destructive(&meta) && !capable(CAP_SYS_BOOT)) {
13857 		verbose(env, "destructive kfunc calls require CAP_SYS_BOOT capability\n");
13858 		operation = bpf_diag_fmt(env, "destructive kfunc %s", meta.func_name);
13859 		bpf_diag_policy(
13860 			env, insn_idx, operation, "destructive kfuncs require CAP_SYS_BOOT",
13861 			"Load the program with CAP_SYS_BOOT, or avoid destructive kfuncs.");
13862 		return -EACCES;
13863 	}
13864 
13865 	if (is_kfunc_perfmon(&meta) && !env->allow_ptr_leaks) {
13866 		verbose(env, "%s is allowed only to CAP_PERFMON and CAP_SYS_ADMIN\n",
13867 			func_name);
13868 		operation = bpf_diag_fmt(env, "kfunc %s", func_name);
13869 		bpf_diag_policy(env, insn_idx, operation, "the kfunc requires CAP_PERFMON",
13870 				"Load the program with CAP_PERFMON, or avoid the kfunc.");
13871 		return -EPERM;
13872 	}
13873 
13874 	sleepable = bpf_is_kfunc_sleepable(&meta);
13875 	if (sleepable && !in_sleepable(env)) {
13876 		verbose(env, "program must be sleepable to call sleepable kfunc %s\n", func_name);
13877 		operation = bpf_diag_fmt(env, "sleepable kfunc %s", func_name);
13878 		bpf_diag_ctx_forbidden(env, insn_idx, operation,
13879 			"Mark the program sleepable if the program type allows it, or use a non-sleepable kfunc.");
13880 		return -EACCES;
13881 	}
13882 
13883 	/* Track non-sleepable context for kfuncs, same as for helpers. */
13884 	if (!in_sleepable_context(env))
13885 		insn_aux->non_sleepable = true;
13886 
13887 	/* Check the arguments */
13888 	err = check_kfunc_args(env, &meta, insn_idx);
13889 	if (err < 0)
13890 		return err;
13891 
13892 	if ((is_bpf_obj_drop_kfunc(meta.func_id) ||
13893 	     is_bpf_percpu_obj_drop_kfunc(meta.func_id)) && (is_tracing_prog_type(prog_type) ||
13894 	     /* is_tracing_prog_type() for now doesn't cover non-iterator tracing progs. */
13895 	     (prog_type == BPF_PROG_TYPE_TRACING && env->prog->expected_attach_type != BPF_TRACE_ITER
13896 	      && !env->prog->sleepable))) {
13897 		struct btf_struct_meta *struct_meta;
13898 
13899 		struct_meta = btf_find_struct_meta(meta.arg_btf, meta.arg_btf_id);
13900 		if (struct_meta && btf_record_has_nmi_unsafe_fields(struct_meta->record)) {
13901 			verbose(env, "%s cannot be used in tracing programs on types with NMI unsafe fields\n",
13902 				func_name);
13903 			return -EINVAL;
13904 		}
13905 	}
13906 
13907 	if (is_bpf_rbtree_add_kfunc(meta.func_id)) {
13908 		err = push_callback_call(env, insn, insn_idx, meta.subprogno,
13909 					 set_rbtree_add_callback_state);
13910 		if (err) {
13911 			verbose(env, "kfunc %s#%d failed callback verification\n",
13912 				func_name, meta.func_id);
13913 			return err;
13914 		}
13915 	}
13916 
13917 	if (is_bpf_wq_set_callback_kfunc(meta.func_id)) {
13918 		err = push_callback_call(env, insn, insn_idx, meta.subprogno,
13919 					 set_timer_callback_state);
13920 		if (err) {
13921 			verbose(env, "kfunc %s#%d failed callback verification\n",
13922 				func_name, meta.func_id);
13923 			return err;
13924 		}
13925 	}
13926 
13927 	if (is_task_work_add_kfunc(meta.func_id)) {
13928 		err = push_callback_call(env, insn, insn_idx, meta.subprogno,
13929 					 set_task_work_schedule_callback_state);
13930 		if (err) {
13931 			verbose(env, "kfunc %s#%d failed callback verification\n",
13932 				func_name, meta.func_id);
13933 			return err;
13934 		}
13935 	}
13936 
13937 	rcu_lock = is_kfunc_bpf_rcu_read_lock(&meta);
13938 	rcu_unlock = is_kfunc_bpf_rcu_read_unlock(&meta);
13939 
13940 	preempt_disable = is_kfunc_bpf_preempt_disable(&meta);
13941 	preempt_enable = is_kfunc_bpf_preempt_enable(&meta);
13942 
13943 	if (rcu_lock) {
13944 		env->cur_state->active_rcu_locks++;
13945 		bpf_diag_record_context(env, insn_idx, BPF_DIAG_CONTEXT_RCU, true,
13946 					env->cur_state->active_rcu_locks);
13947 	} else if (rcu_unlock) {
13948 		if (env->cur_state->active_rcu_locks == 0) {
13949 			verbose(env, "unmatched rcu read unlock (kernel function %s)\n", func_name);
13950 			bpf_diag_ctx_underflow(
13951 				env, insn_idx, func_name, BPF_DIAG_CONTEXT_RCU,
13952 				"Remove the extra bpf_rcu_read_unlock() call, or ensure this path first enters an RCU read lock region.");
13953 			return -EINVAL;
13954 		}
13955 		env->cur_state->active_rcu_locks--;
13956 		bpf_diag_record_context(env, insn_idx, BPF_DIAG_CONTEXT_RCU, false,
13957 					env->cur_state->active_rcu_locks);
13958 		if (!in_rcu_cs(env))
13959 			invalidate_rcu_protected_refs(env);
13960 	} else if (preempt_disable) {
13961 		env->cur_state->active_preempt_locks++;
13962 		bpf_diag_record_context(env, insn_idx, BPF_DIAG_CONTEXT_PREEMPT, true,
13963 					env->cur_state->active_preempt_locks);
13964 	} else if (preempt_enable) {
13965 		if (env->cur_state->active_preempt_locks == 0) {
13966 			verbose(env, "unmatched attempt to enable preemption (kernel function %s)\n", func_name);
13967 			bpf_diag_ctx_underflow(
13968 				env, insn_idx, func_name, BPF_DIAG_CONTEXT_PREEMPT,
13969 				"Remove the extra bpf_preempt_enable() call, or ensure this path first disables preemption.");
13970 			return -EINVAL;
13971 		}
13972 		env->cur_state->active_preempt_locks--;
13973 		bpf_diag_record_context(env, insn_idx, BPF_DIAG_CONTEXT_PREEMPT, false,
13974 					env->cur_state->active_preempt_locks);
13975 		if (!in_rcu_cs(env))
13976 			invalidate_rcu_protected_refs(env);
13977 	}
13978 
13979 	if (sleepable && !in_sleepable_context(env)) {
13980 		verbose(env, "kernel func %s is sleepable within %s\n",
13981 			func_name, non_sleepable_context_description(env));
13982 		operation = bpf_diag_fmt(env, "sleepable kfunc %s", func_name);
13983 		bpf_diag_ctx_forbidden(env, insn_idx, operation,
13984 			"Move the kfunc call outside the critical section, or use a non-sleepable kfunc.");
13985 		return -EACCES;
13986 	}
13987 
13988 	if (in_rbtree_lock_required_cb(env) && (rcu_lock || rcu_unlock)) {
13989 		verbose(env, "Calling bpf_rcu_read_{lock,unlock} in unnecessary rbtree callback\n");
13990 		return -EACCES;
13991 	}
13992 
13993 	if (is_kfunc_rcu_protected(&meta) && !in_rcu_cs(env)) {
13994 		verbose(env, "kernel func %s requires RCU critical section protection\n", func_name);
13995 		bpf_diag_ctx_required(
13996 			env, insn_idx, func_name, BPF_DIAG_CONTEXT_RCU,
13997 			"Call this kfunc between bpf_rcu_read_lock() and bpf_rcu_read_unlock(), keeping all exit paths balanced.");
13998 		return -EACCES;
13999 	}
14000 
14001 	/* In case of release function, we get register number of refcounted
14002 	 * PTR_TO_BTF_ID in bpf_kfunc_arg_meta, do the release now.
14003 	 */
14004 	if (meta.release_regno) {
14005 		err = release_reg(env, &regs[meta.release_regno], false, !!meta.dynptr.id);
14006 		if (err)
14007 			return err;
14008 	}
14009 
14010 	if (is_bpf_list_push_kfunc(meta.func_id) || is_bpf_rbtree_add_kfunc(meta.func_id)) {
14011 		id = regs[BPF_REG_2].id;
14012 		insn_aux->insert_off = regs[BPF_REG_2].var_off.value;
14013 		insn_aux->kptr_struct_meta = btf_find_struct_meta(meta.arg_btf, meta.arg_btf_id);
14014 		ref_convert_owning_non_owning(env, id);
14015 	}
14016 
14017 	if (meta.func_id == special_kfunc_list[KF_bpf_throw]) {
14018 		if (!bpf_jit_supports_exceptions()) {
14019 			verbose(env, "JIT does not support calling kfunc %s#%d\n",
14020 				func_name, meta.func_id);
14021 			return -ENOTSUPP;
14022 		}
14023 		env->seen_exception = true;
14024 
14025 		/* In the case of the default callback, the cookie value passed
14026 		 * to bpf_throw becomes the return value of the program.
14027 		 */
14028 		if (!env->exception_callback_subprog) {
14029 			err = check_return_code(env, BPF_REG_1, "R1");
14030 			if (err < 0)
14031 				return err;
14032 		}
14033 	}
14034 
14035 	bpf_diag_record_caller_saved(env, regs);
14036 	bpf_diag_mod_begin(env, &regs[BPF_REG_0], NULL, BPF_DIAG_MOD_WRITE);
14037 	for (i = 0; i < CALLER_SAVED_REGS; i++) {
14038 		u32 regno = caller_saved[i];
14039 
14040 		bpf_mark_reg_not_init(env, &regs[regno]);
14041 	}
14042 	invalidate_outgoing_stack_args(env, cur_func(env));
14043 
14044 	/* Check return type */
14045 	t = btf_type_skip_modifiers(desc_btf, meta.func_proto->type, NULL);
14046 
14047 	if (is_kfunc_acquire(&meta) && !btf_type_is_struct_ptr(meta.btf, t)) {
14048 		if (meta.btf != btf_vmlinux ||
14049 		    (!is_bpf_obj_new_kfunc(meta.func_id) &&
14050 		     !is_bpf_percpu_obj_new_kfunc(meta.func_id) &&
14051 		     !is_bpf_refcount_acquire_kfunc(meta.func_id))) {
14052 			verbose(env, "acquire kernel function does not return PTR_TO_BTF_ID\n");
14053 			return -EINVAL;
14054 		}
14055 	}
14056 
14057 	if (btf_type_is_scalar(t)) {
14058 		mark_reg_unknown(env, regs, BPF_REG_0);
14059 		if (meta.btf == btf_vmlinux && (meta.func_id == special_kfunc_list[KF_bpf_res_spin_lock] ||
14060 		    meta.func_id == special_kfunc_list[KF_bpf_res_spin_lock_irqsave]))
14061 			__mark_reg_const_zero(env, &regs[BPF_REG_0]);
14062 	} else if (btf_type_is_ptr(t)) {
14063 		ptr_type = btf_type_skip_modifiers(desc_btf, t->type, &ptr_type_id);
14064 		err = check_special_kfunc(env, &meta, regs, insn_aux, ptr_type, desc_btf);
14065 		if (err) {
14066 			if (err < 0)
14067 				return err;
14068 		} else if (btf_type_is_void(ptr_type)) {
14069 			/* kfunc returning 'void *' is equivalent to returning scalar */
14070 			mark_reg_unknown(env, regs, BPF_REG_0);
14071 		} else if (!__btf_type_is_struct(ptr_type)) {
14072 			if (!meta.ret_mem.found) {
14073 				__u32 sz;
14074 
14075 				if (!IS_ERR(btf_resolve_size(desc_btf, ptr_type, &sz))) {
14076 					meta.ret_mem.found = true;
14077 					meta.ret_mem.size = sz;
14078 					meta.r0_rdonly = true;
14079 				}
14080 
14081 				if (meta.func_id == special_kfunc_list[KF_bpf_session_cookie])
14082 					meta.r0_rdonly = false;
14083 			}
14084 			if (!meta.ret_mem.found) {
14085 				ptr_type_name = btf_name_by_offset(desc_btf,
14086 								   ptr_type->name_off);
14087 				verbose(env,
14088 					"kernel function %s returns pointer type %s %s is not supported\n",
14089 					func_name,
14090 					btf_type_str(ptr_type),
14091 					ptr_type_name);
14092 				return -EINVAL;
14093 			}
14094 
14095 			mark_reg_known_zero(env, regs, BPF_REG_0);
14096 			regs[BPF_REG_0].type = PTR_TO_MEM;
14097 			regs[BPF_REG_0].mem_size = meta.ret_mem.size;
14098 
14099 			if (meta.r0_rdonly)
14100 				regs[BPF_REG_0].type |= MEM_RDONLY;
14101 
14102 			/* Ensures we don't access the memory after a release_reference() */
14103 			if (meta.ref_obj.id) {
14104 				err = validate_ref_obj(env, &meta.ref_obj);
14105 				if (err)
14106 					return err;
14107 				regs[BPF_REG_0].parent_id = meta.ref_obj.id;
14108 			}
14109 
14110 			if (is_kfunc_rcu_protected(&meta))
14111 				regs[BPF_REG_0].type |= MEM_RCU;
14112 		} else {
14113 			enum bpf_reg_type type = PTR_TO_BTF_ID;
14114 
14115 			if (meta.func_id == special_kfunc_list[KF_bpf_get_kmem_cache])
14116 				type |= PTR_UNTRUSTED;
14117 			else if (is_kfunc_rcu_protected(&meta) ||
14118 				 (bpf_is_iter_next_kfunc(&meta) &&
14119 				  (get_iter_from_state(env->cur_state, &meta)
14120 					   ->type & MEM_RCU))) {
14121 				/*
14122 				 * If the iterator's constructor (the _new
14123 				 * function e.g., bpf_iter_task_new) has been
14124 				 * annotated with BPF kfunc flag
14125 				 * KF_RCU_PROTECTED and was called within a RCU
14126 				 * read-side critical section, also propagate
14127 				 * the MEM_RCU flag to the pointer returned from
14128 				 * the iterator's next function (e.g.,
14129 				 * bpf_iter_task_next).
14130 				 */
14131 				type |= MEM_RCU;
14132 			} else {
14133 				/*
14134 				 * Any PTR_TO_BTF_ID that is returned from a BPF
14135 				 * kfunc should by default be treated as
14136 				 * implicitly trusted.
14137 				 */
14138 				type |= PTR_TRUSTED;
14139 			}
14140 
14141 			mark_reg_known_zero(env, regs, BPF_REG_0);
14142 			regs[BPF_REG_0].btf = desc_btf;
14143 			regs[BPF_REG_0].type = type;
14144 			regs[BPF_REG_0].btf_id = ptr_type_id;
14145 		}
14146 
14147 		if (is_kfunc_ret_null(&meta)) {
14148 			regs[BPF_REG_0].type |= PTR_MAYBE_NULL;
14149 			/* For mark_ptr_or_null_reg, see 93c230e3f5bd6 */
14150 			regs[BPF_REG_0].id = ++env->id_gen;
14151 		}
14152 		if (is_kfunc_acquire(&meta)) {
14153 			id = acquire_reference(env, insn_idx, 0);
14154 			if (id < 0)
14155 				return id;
14156 			regs[BPF_REG_0].id = id;
14157 		} else if (is_rbtree_node_type(ptr_type) || is_list_node_type(ptr_type)) {
14158 			ref_set_non_owning(env, &regs[BPF_REG_0]);
14159 		}
14160 
14161 		if (reg_may_point_to_spin_lock(&regs[BPF_REG_0]) && !regs[BPF_REG_0].id)
14162 			regs[BPF_REG_0].id = ++env->id_gen;
14163 	} else if (btf_type_is_void(t)) {
14164 		if (meta.btf == btf_vmlinux) {
14165 			if (is_bpf_obj_drop_kfunc(meta.func_id) ||
14166 			    is_bpf_percpu_obj_drop_kfunc(meta.func_id)) {
14167 				insn_aux->kptr_struct_meta =
14168 					btf_find_struct_meta(meta.arg_btf,
14169 							     meta.arg_btf_id);
14170 			}
14171 		}
14172 	}
14173 
14174 	if (bpf_is_kfunc_pkt_changing(&meta))
14175 		clear_all_pkt_pointers(env);
14176 
14177 	nargs = btf_type_vlen(meta.func_proto);
14178 	if (nargs > MAX_BPF_FUNC_REG_ARGS) {
14179 		struct bpf_func_state *caller = cur_func(env);
14180 		struct bpf_subprog_info *caller_info = &env->subprog_info[caller->subprogno];
14181 		u16 out_stack_arg_cnt = nargs - MAX_BPF_FUNC_REG_ARGS;
14182 		u16 stack_arg_cnt = bpf_in_stack_arg_cnt(caller_info) + out_stack_arg_cnt;
14183 
14184 		if (stack_arg_cnt > caller_info->stack_arg_cnt)
14185 			caller_info->stack_arg_cnt = stack_arg_cnt;
14186 	}
14187 
14188 	/*
14189 	 * Record R0 before process_iter_next_call() snapshots the alternate
14190 	 * iterator path's diagnostic position.
14191 	 */
14192 	bpf_diag_mod_end(env);
14193 
14194 	if (bpf_is_iter_next_kfunc(&meta)) {
14195 		err = process_iter_next_call(env, insn_idx, &meta);
14196 		if (err)
14197 			return err;
14198 	}
14199 
14200 	if (meta.func_id == special_kfunc_list[KF_bpf_session_cookie])
14201 		env->prog->call_session_cookie = true;
14202 
14203 	if (bpf_is_throw_kfunc(insn))
14204 		return process_bpf_exit_full(env, NULL, true);
14205 
14206 	return 0;
14207 }
14208 
14209 static bool check_reg_sane_offset_scalar(struct bpf_verifier_env *env,
14210 					 const struct bpf_reg_state *reg,
14211 					 enum bpf_reg_type type)
14212 {
14213 	bool known = tnum_is_const(reg->var_off);
14214 	s64 val = reg->var_off.value;
14215 	s64 smin = reg_smin(reg);
14216 
14217 	if (known && (val >= BPF_MAX_VAR_OFF || val <= -BPF_MAX_VAR_OFF)) {
14218 		verbose(env, "math between %s pointer and %lld is not allowed\n",
14219 			reg_type_str(env, type), val);
14220 		return false;
14221 	}
14222 
14223 	if (smin == S64_MIN) {
14224 		verbose(env, "math between %s pointer and register with unbounded min value is not allowed\n",
14225 			reg_type_str(env, type));
14226 		return false;
14227 	}
14228 
14229 	if (smin >= BPF_MAX_VAR_OFF || smin <= -BPF_MAX_VAR_OFF) {
14230 		verbose(env, "value %lld makes %s pointer be out of bounds\n",
14231 			smin, reg_type_str(env, type));
14232 		return false;
14233 	}
14234 
14235 	return true;
14236 }
14237 
14238 static bool check_reg_sane_offset_ptr(struct bpf_verifier_env *env,
14239 				      const struct bpf_reg_state *reg,
14240 				      enum bpf_reg_type type)
14241 {
14242 	bool known = tnum_is_const(reg->var_off);
14243 	s64 val = reg->var_off.value;
14244 	s64 smin = reg_smin(reg);
14245 
14246 	if (known && (val >= BPF_MAX_VAR_OFF || val <= -BPF_MAX_VAR_OFF)) {
14247 		verbose(env, "%s pointer offset %lld is not allowed\n",
14248 			reg_type_str(env, type), val);
14249 		return false;
14250 	}
14251 
14252 	if (smin >= BPF_MAX_VAR_OFF || smin <= -BPF_MAX_VAR_OFF) {
14253 		verbose(env, "%s pointer offset %lld is not allowed\n",
14254 			reg_type_str(env, type), smin);
14255 		return false;
14256 	}
14257 
14258 	return true;
14259 }
14260 
14261 enum {
14262 	REASON_BOUNDS	= -1,
14263 	REASON_TYPE	= -2,
14264 	REASON_PATHS	= -3,
14265 	REASON_LIMIT	= -4,
14266 	REASON_STACK	= -5,
14267 };
14268 
14269 static int retrieve_ptr_limit(const struct bpf_reg_state *ptr_reg,
14270 			      u32 *alu_limit, bool mask_to_left)
14271 {
14272 	u32 max = 0, ptr_limit = 0;
14273 
14274 	switch (ptr_reg->type) {
14275 	case PTR_TO_STACK:
14276 		/* Offset 0 is out-of-bounds, but acceptable start for the
14277 		 * left direction, see BPF_REG_FP. Also, unknown scalar
14278 		 * offset where we would need to deal with min/max bounds is
14279 		 * currently prohibited for unprivileged.
14280 		 */
14281 		max = MAX_BPF_STACK + mask_to_left;
14282 		ptr_limit = -ptr_reg->var_off.value;
14283 		break;
14284 	case PTR_TO_MAP_VALUE:
14285 		max = ptr_reg->map_ptr->value_size;
14286 		ptr_limit = mask_to_left ? reg_smin(ptr_reg) : reg_umax(ptr_reg);
14287 		break;
14288 	default:
14289 		return REASON_TYPE;
14290 	}
14291 
14292 	if (ptr_limit >= max)
14293 		return REASON_LIMIT;
14294 	*alu_limit = ptr_limit;
14295 	return 0;
14296 }
14297 
14298 static bool can_skip_alu_sanitation(const struct bpf_verifier_env *env,
14299 				    const struct bpf_insn *insn)
14300 {
14301 	return env->bypass_spec_v1 ||
14302 		BPF_SRC(insn->code) == BPF_K ||
14303 		cur_aux(env)->nospec;
14304 }
14305 
14306 static int update_alu_sanitation_state(struct bpf_insn_aux_data *aux,
14307 				       u32 alu_state, u32 alu_limit)
14308 {
14309 	/* If we arrived here from different branches with different
14310 	 * state or limits to sanitize, then this won't work.
14311 	 */
14312 	if (aux->alu_state &&
14313 	    (aux->alu_state != alu_state ||
14314 	     aux->alu_limit != alu_limit))
14315 		return REASON_PATHS;
14316 
14317 	/* Corresponding fixup done in do_misc_fixups(). */
14318 	aux->alu_state = alu_state;
14319 	aux->alu_limit = alu_limit;
14320 	return 0;
14321 }
14322 
14323 static int sanitize_val_alu(struct bpf_verifier_env *env,
14324 			    struct bpf_insn *insn)
14325 {
14326 	struct bpf_insn_aux_data *aux = cur_aux(env);
14327 
14328 	if (can_skip_alu_sanitation(env, insn))
14329 		return 0;
14330 
14331 	return update_alu_sanitation_state(aux, BPF_ALU_NON_POINTER, 0);
14332 }
14333 
14334 static bool sanitize_needed(u8 opcode)
14335 {
14336 	return opcode == BPF_ADD || opcode == BPF_SUB;
14337 }
14338 
14339 struct bpf_sanitize_info {
14340 	struct bpf_insn_aux_data aux;
14341 	bool mask_to_left;
14342 };
14343 
14344 static int sanitize_speculative_path(struct bpf_verifier_env *env,
14345 				     const struct bpf_insn *insn,
14346 				     u32 next_idx, u32 curr_idx)
14347 {
14348 	struct bpf_verifier_state *branch;
14349 	struct bpf_reg_state *regs;
14350 
14351 	branch = push_stack(env, next_idx, curr_idx, true);
14352 	if (!IS_ERR(branch) && insn) {
14353 		regs = branch->frame[branch->curframe]->regs;
14354 		if (BPF_SRC(insn->code) == BPF_K) {
14355 			mark_reg_unknown(env, regs, insn->dst_reg);
14356 		} else if (BPF_SRC(insn->code) == BPF_X) {
14357 			mark_reg_unknown(env, regs, insn->dst_reg);
14358 			mark_reg_unknown(env, regs, insn->src_reg);
14359 		}
14360 	}
14361 	return PTR_ERR_OR_ZERO(branch);
14362 }
14363 
14364 static int sanitize_ptr_alu(struct bpf_verifier_env *env,
14365 			    struct bpf_insn *insn,
14366 			    const struct bpf_reg_state *ptr_reg,
14367 			    const struct bpf_reg_state *off_reg,
14368 			    struct bpf_reg_state *dst_reg,
14369 			    struct bpf_sanitize_info *info,
14370 			    const bool commit_window)
14371 {
14372 	struct bpf_insn_aux_data *aux = commit_window ? cur_aux(env) : &info->aux;
14373 	struct bpf_verifier_state *vstate = env->cur_state;
14374 	bool off_is_imm = tnum_is_const(off_reg->var_off);
14375 	bool off_is_neg = reg_smin(off_reg) < 0;
14376 	bool ptr_is_dst_reg = ptr_reg == dst_reg;
14377 	u8 opcode = BPF_OP(insn->code);
14378 	u32 alu_state, alu_limit;
14379 	struct bpf_reg_state tmp;
14380 	int err;
14381 
14382 	if (can_skip_alu_sanitation(env, insn))
14383 		return 0;
14384 
14385 	/* We already marked aux for masking from non-speculative
14386 	 * paths, thus we got here in the first place. We only care
14387 	 * to explore bad access from here.
14388 	 */
14389 	if (vstate->speculative)
14390 		goto do_sim;
14391 
14392 	if (!commit_window) {
14393 		if (!tnum_is_const(off_reg->var_off) &&
14394 		    (reg_smin(off_reg) < 0) != (reg_smax(off_reg) < 0))
14395 			return REASON_BOUNDS;
14396 
14397 		info->mask_to_left = (opcode == BPF_ADD &&  off_is_neg) ||
14398 				     (opcode == BPF_SUB && !off_is_neg);
14399 	}
14400 
14401 	err = retrieve_ptr_limit(ptr_reg, &alu_limit, info->mask_to_left);
14402 	if (err < 0)
14403 		return err;
14404 
14405 	if (commit_window) {
14406 		/* In commit phase we narrow the masking window based on
14407 		 * the observed pointer move after the simulated operation.
14408 		 */
14409 		alu_state = info->aux.alu_state;
14410 		alu_limit = abs(info->aux.alu_limit - alu_limit);
14411 	} else {
14412 		alu_state  = off_is_neg ? BPF_ALU_NEG_VALUE : 0;
14413 		alu_state |= off_is_imm ? BPF_ALU_IMMEDIATE : 0;
14414 		alu_state |= ptr_is_dst_reg ?
14415 			     BPF_ALU_SANITIZE_SRC : BPF_ALU_SANITIZE_DST;
14416 
14417 		/* Limit pruning on unknown scalars to enable deep search for
14418 		 * potential masking differences from other program paths.
14419 		 */
14420 		if (!off_is_imm)
14421 			env->explore_alu_limits = true;
14422 	}
14423 
14424 	err = update_alu_sanitation_state(aux, alu_state, alu_limit);
14425 	if (err < 0)
14426 		return err;
14427 do_sim:
14428 	/* If we're in commit phase, we're done here given we already
14429 	 * pushed the truncated dst_reg into the speculative verification
14430 	 * stack.
14431 	 *
14432 	 * Also, when register is a known constant, we rewrite register-based
14433 	 * operation to immediate-based, and thus do not need masking (and as
14434 	 * a consequence, do not need to simulate the zero-truncation either).
14435 	 */
14436 	if (commit_window || off_is_imm)
14437 		return 0;
14438 
14439 	/* Simulate and find potential out-of-bounds access under
14440 	 * speculative execution from truncation as a result of
14441 	 * masking when off was not within expected range. If off
14442 	 * sits in dst, then we temporarily need to move ptr there
14443 	 * to simulate dst (== 0) +/-= ptr. Needed, for example,
14444 	 * for cases where we use K-based arithmetic in one direction
14445 	 * and truncated reg-based in the other in order to explore
14446 	 * bad access.
14447 	 */
14448 	if (!ptr_is_dst_reg) {
14449 		tmp = *dst_reg;
14450 		*dst_reg = *ptr_reg;
14451 	}
14452 	err = sanitize_speculative_path(env, NULL, env->insn_idx + 1, env->insn_idx);
14453 	if (err < 0)
14454 		return REASON_STACK;
14455 	if (!ptr_is_dst_reg)
14456 		*dst_reg = tmp;
14457 	return 0;
14458 }
14459 
14460 static void sanitize_mark_insn_seen(struct bpf_verifier_env *env)
14461 {
14462 	struct bpf_verifier_state *vstate = env->cur_state;
14463 
14464 	/* If we simulate paths under speculation, we don't update the
14465 	 * insn as 'seen' such that when we verify unreachable paths in
14466 	 * the non-speculative domain, sanitize_dead_code() can still
14467 	 * rewrite/sanitize them.
14468 	 */
14469 	if (!vstate->speculative)
14470 		env->insn_aux_data[env->insn_idx].seen = env->pass_cnt;
14471 }
14472 
14473 static int sanitize_err(struct bpf_verifier_env *env, const struct bpf_insn *insn, int reason)
14474 {
14475 	static const char *err = "pointer arithmetic with it prohibited for !root";
14476 	const char *op = BPF_OP(insn->code) == BPF_ADD ? "add" : "sub";
14477 	u32 dst = insn->dst_reg, src = insn->src_reg;
14478 	struct bpf_reg_state *regs = cur_regs(env);
14479 
14480 	switch (reason) {
14481 	case REASON_BOUNDS:
14482 		verbose(env, "R%d has unknown scalar with mixed signed bounds, %s\n",
14483 			regs[src].type == SCALAR_VALUE ? src : dst, err);
14484 		break;
14485 	case REASON_TYPE:
14486 		verbose(env, "R%d has pointer with unsupported alu operation, %s\n",
14487 			regs[src].type == SCALAR_VALUE ? dst : src, err);
14488 		break;
14489 	case REASON_PATHS:
14490 		verbose(env, "R%d tried to %s from different maps, paths or scalars, %s\n",
14491 			dst, op, err);
14492 		break;
14493 	case REASON_LIMIT:
14494 		verbose(env, "R%d tried to %s beyond pointer bounds, %s\n",
14495 			dst, op, err);
14496 		break;
14497 	case REASON_STACK:
14498 		verbose(env, "R%d could not be pushed for speculative verification, %s\n",
14499 			dst, err);
14500 		return -ENOMEM;
14501 	default:
14502 		verifier_bug(env, "unknown reason (%d)", reason);
14503 		break;
14504 	}
14505 
14506 	return -EACCES;
14507 }
14508 
14509 /* check that stack access falls within stack limits and that 'reg' doesn't
14510  * have a variable offset.
14511  *
14512  * Variable offset is prohibited for unprivileged mode for simplicity since it
14513  * requires corresponding support in Spectre masking for stack ALU.  See also
14514  * retrieve_ptr_limit().
14515  */
14516 static int check_stack_access_for_ptr_arithmetic(
14517 				struct bpf_verifier_env *env,
14518 				int regno,
14519 				const struct bpf_reg_state *reg,
14520 				int off)
14521 {
14522 	if (!tnum_is_const(reg->var_off)) {
14523 		char tn_buf[48];
14524 
14525 		tnum_strn(tn_buf, sizeof(tn_buf), reg->var_off);
14526 		verbose(env, "R%d variable stack access prohibited for !root, var_off=%s off=%d\n",
14527 			regno, tn_buf, off);
14528 		return -EACCES;
14529 	}
14530 
14531 	if (off >= 0 || off < -MAX_BPF_STACK) {
14532 		verbose(env, "R%d stack pointer arithmetic goes out of range, "
14533 			"prohibited for !root; off=%d\n", regno, off);
14534 		return -EACCES;
14535 	}
14536 
14537 	return 0;
14538 }
14539 
14540 static int sanitize_check_bounds(struct bpf_verifier_env *env,
14541 				 const struct bpf_insn *insn,
14542 				 struct bpf_reg_state *dst_reg)
14543 {
14544 	u32 dst = insn->dst_reg;
14545 
14546 	/* For unprivileged we require that resulting offset must be in bounds
14547 	 * in order to be able to sanitize access later on.
14548 	 */
14549 	if (env->bypass_spec_v1)
14550 		return 0;
14551 
14552 	switch (dst_reg->type) {
14553 	case PTR_TO_STACK:
14554 		if (check_stack_access_for_ptr_arithmetic(env, dst, dst_reg,
14555 							  dst_reg->var_off.value))
14556 			return -EACCES;
14557 		break;
14558 	case PTR_TO_MAP_VALUE:
14559 		if (check_map_access(env, dst_reg, argno_from_reg(dst), 0, 1, false, ACCESS_HELPER)) {
14560 			verbose(env, "R%d pointer arithmetic of map value goes out of range, "
14561 				"prohibited for !root\n", dst);
14562 			return -EACCES;
14563 		}
14564 		break;
14565 	default:
14566 		return -EOPNOTSUPP;
14567 	}
14568 
14569 	return 0;
14570 }
14571 
14572 /* Handles arithmetic on a pointer and a scalar: computes new min/max and var_off.
14573  * Caller should also handle BPF_MOV case separately.
14574  * If we return -EACCES, caller may want to try again treating pointer as a
14575  * scalar.  So we only emit a diagnostic if !env->allow_ptr_leaks.
14576  */
14577 static int adjust_ptr_min_max_vals(struct bpf_verifier_env *env, struct bpf_insn *insn,
14578 				   u32 ptr_regno, const struct bpf_reg_state *ptr_reg,
14579 				   const struct bpf_reg_state *off_reg)
14580 {
14581 	struct bpf_verifier_state *vstate = env->cur_state;
14582 	struct bpf_func_state *state = vstate->frame[vstate->curframe];
14583 	struct bpf_reg_state *regs = state->regs, *dst_reg;
14584 	bool known = tnum_is_const(off_reg->var_off);
14585 	s64 smin_val = reg_smin(off_reg), smax_val = reg_smax(off_reg);
14586 	u64 umin_val = reg_umin(off_reg), umax_val = reg_umax(off_reg);
14587 	struct bpf_sanitize_info info = {};
14588 	u8 opcode = BPF_OP(insn->code);
14589 	u32 dst = insn->dst_reg;
14590 	const char *reason;
14591 	int ret, bounds_ret;
14592 
14593 	dst_reg = &regs[dst];
14594 
14595 	if ((known && (smin_val != smax_val || umin_val != umax_val)) ||
14596 	    smin_val > smax_val || umin_val > umax_val) {
14597 		/* Taint dst register if offset had invalid bounds derived from
14598 		 * e.g. dead branches.
14599 		 */
14600 		__mark_reg_unknown(env, dst_reg);
14601 		return 0;
14602 	}
14603 
14604 	if (BPF_CLASS(insn->code) != BPF_ALU64) {
14605 		/* 32-bit ALU ops on pointers produce (meaningless) scalars */
14606 		if (opcode == BPF_SUB && env->allow_ptr_leaks) {
14607 			__mark_reg_unknown(env, dst_reg);
14608 			return 0;
14609 		}
14610 
14611 		verbose(env,
14612 			"R%d 32-bit pointer arithmetic prohibited\n",
14613 			dst);
14614 		reason = bpf_diag_fmt(
14615 			env, "R%d holds %s. 32-bit ALU operations on pointers discard pointer tracking, so the verifier cannot keep the result as a safe pointer.",
14616 			ptr_regno, bpf_diag_reg_type_plain(env, ptr_reg->type));
14617 		bpf_diag_register_type(
14618 			env, env->insn_idx, ptr_regno, "32-bit pointer arithmetic", reason,
14619 			"Use a 64-bit ALU instruction with an allowed, bounded scalar offset.");
14620 		return -EACCES;
14621 	}
14622 
14623 	if (ptr_reg->type & PTR_MAYBE_NULL) {
14624 		verbose(env, "R%d pointer arithmetic on %s prohibited, null-check it first\n",
14625 			dst, reg_type_str(env, ptr_reg->type));
14626 		reason = bpf_diag_fmt(
14627 			env, "R%d may be NULL (%s). Pointer arithmetic is allowed only after the program proves the pointer is non-NULL on this path.",
14628 			ptr_regno, reg_type_str(env, ptr_reg->type));
14629 		bpf_diag_register_type(
14630 			env, env->insn_idx, ptr_regno, "pointer arithmetic before NULL check", reason,
14631 			"Make sure that a NULL check precedes any arithmetic performed on the pointer.");
14632 		return -EACCES;
14633 	}
14634 
14635 	switch (base_type(ptr_reg->type)) {
14636 	case PTR_TO_CTX:
14637 	case PTR_TO_MAP_VALUE:
14638 	case PTR_TO_MAP_KEY:
14639 	case PTR_TO_STACK:
14640 	case PTR_TO_PACKET_META:
14641 	case PTR_TO_PACKET:
14642 	case PTR_TO_TP_BUFFER:
14643 	case PTR_TO_BTF_ID:
14644 	case PTR_TO_MEM:
14645 	case PTR_TO_BUF:
14646 	case PTR_TO_FUNC:
14647 	case CONST_PTR_TO_DYNPTR:
14648 		break;
14649 	case PTR_TO_FLOW_KEYS:
14650 		if (known)
14651 			break;
14652 		fallthrough;
14653 	case CONST_PTR_TO_MAP:
14654 		/* smin_val represents the known value */
14655 		if (known && smin_val == 0 && opcode == BPF_ADD)
14656 			break;
14657 		fallthrough;
14658 	default:
14659 		verbose(env, "R%d pointer arithmetic on %s prohibited\n",
14660 			dst, reg_type_str(env, ptr_reg->type));
14661 		reason = bpf_diag_fmt(
14662 			env, "R%d holds %s. This pointer kind does not allow offset arithmetic.",
14663 			ptr_regno, bpf_diag_reg_type_plain(env, ptr_reg->type));
14664 		bpf_diag_register_type(
14665 			env, env->insn_idx, ptr_regno, "pointer arithmetic is not allowed", reason,
14666 			"Do not change this pointer's offset; use it only in operations accepted for its kind.");
14667 		return -EACCES;
14668 	}
14669 
14670 	/* For 'scalar += pointer', dst_reg inherits the complete pointer
14671 	 * register state. Individual fields may be adjusted later by pointer
14672 	 * arithmetic. Callers guarantee that below does not overwrite off_reg.
14673 	 */
14674 	if (dst_reg != ptr_reg)
14675 		*dst_reg = *ptr_reg;
14676 
14677 	/*
14678 	 * Accesses to untrusted PTR_TO_MEM are done through probe
14679 	 * instructions, hence no need to track offsets.
14680 	 */
14681 	if (base_type(ptr_reg->type) == PTR_TO_MEM && (ptr_reg->type & PTR_UNTRUSTED))
14682 		return 0;
14683 
14684 	if (!check_reg_sane_offset_scalar(env, off_reg, ptr_reg->type)) {
14685 		reason = bpf_diag_fmt(
14686 			env, "The scalar offset used with R%d is unbounded or outside the verifier's safe pointer-offset range [-%u, %u].",
14687 			ptr_regno, BPF_MAX_VAR_OFF, BPF_MAX_VAR_OFF);
14688 		bpf_diag_register_type(
14689 			env, env->insn_idx, ptr_regno, "pointer offset is not safe", reason,
14690 			"Clamp or bounds-check the scalar offset before applying it to the pointer.");
14691 		return -EINVAL;
14692 	}
14693 	if (!check_reg_sane_offset_ptr(env, ptr_reg, ptr_reg->type)) {
14694 		reason = bpf_diag_fmt(
14695 			env, "R%d already has an offset outside the verifier's safe range [-%u, %u] for %s.",
14696 			ptr_regno, BPF_MAX_VAR_OFF, BPF_MAX_VAR_OFF,
14697 			bpf_diag_reg_type_plain(env, ptr_reg->type));
14698 		bpf_diag_register_type(
14699 			env, env->insn_idx, ptr_regno, "pointer offset is not safe", reason,
14700 			"Keep the base pointer within the verifier's allowed offset range before applying more arithmetic.");
14701 		return -EINVAL;
14702 	}
14703 
14704 	if (sanitize_needed(opcode)) {
14705 		ret = sanitize_ptr_alu(env, insn, ptr_reg, off_reg, dst_reg,
14706 				       &info, false);
14707 		if (ret < 0)
14708 			return sanitize_err(env, insn, ret);
14709 	}
14710 
14711 	/*
14712 	 * Pointer types do not carry 32-bit bounds at the moment. Blank r32
14713 	 * only after sanitize_ptr_alu() may have snapshotted dst_reg into a
14714 	 * speculative path: otherwise reg_bounds_sanity_check() might hit some
14715 	 * constraints violations.
14716 	 */
14717 	__mark_reg32_unbounded(dst_reg);
14718 
14719 	switch (opcode) {
14720 	case BPF_ADD:
14721 		/*
14722 		 * dst_reg gets the pointer type and since some positive
14723 		 * integer value was added to the pointer, give it a new 'id'
14724 		 * if it's a PTR_TO_PACKET.
14725 		 * this creates a new 'base' pointer, off_reg (variable) gets
14726 		 * added into the variable offset, and we copy the fixed offset
14727 		 * from ptr_reg.
14728 		 */
14729 		dst_reg->r64 = cnum64_add(ptr_reg->r64, off_reg->r64);
14730 		dst_reg->var_off = tnum_add(ptr_reg->var_off, off_reg->var_off);
14731 		dst_reg->raw = ptr_reg->raw;
14732 		if (reg_is_pkt_pointer(ptr_reg)) {
14733 			if (!known)
14734 				dst_reg->id = ++env->id_gen;
14735 			/*
14736 			 * Clear range for unknown addends since we can't know
14737 			 * where the pkt pointer ended up. Also clear AT_PKT_END /
14738 			 * BEYOND_PKT_END from prior comparison as any pointer
14739 			 * arithmetic invalidates them.
14740 			 */
14741 			if (!known || dst_reg->range < 0)
14742 				memset(&dst_reg->raw, 0, sizeof(dst_reg->raw));
14743 		}
14744 		break;
14745 	case BPF_SUB:
14746 		if (dst_reg != ptr_reg) {
14747 			/* scalar -= pointer.  Creates an unknown scalar */
14748 			verbose(env, "R%d tried to subtract pointer from scalar\n",
14749 				dst);
14750 			reason = bpf_diag_fmt(
14751 				env, "This operation subtracts pointer register R%d from scalar register R%d. "
14752 				"The verifier only tracks pointer-minus-scalar arithmetic for allowed pointer types.",
14753 				ptr_regno, dst);
14754 			bpf_diag_register_type(
14755 				env, env->insn_idx, ptr_regno, "pointer subtracted from scalar", reason,
14756 				"Keep the pointer as the base; only add or subtract bounded scalars when permitted.");
14757 			return -EACCES;
14758 		}
14759 		/* We don't allow subtraction from FP, because (according to
14760 		 * test_verifier.c test "invalid fp arithmetic", JITs might not
14761 		 * be able to deal with it.
14762 		 */
14763 		if (ptr_reg->type == PTR_TO_STACK) {
14764 			verbose(env, "R%d subtraction from stack pointer prohibited\n",
14765 				dst);
14766 			reason = bpf_diag_fmt(
14767 				env, "R%d is a stack pointer. The verifier does not allow BPF_SUB to move stack pointers.",
14768 				ptr_regno);
14769 			bpf_diag_register_type(
14770 				env, env->insn_idx, ptr_regno, "subtraction from stack pointer", reason,
14771 				"Use addition from R10 to form stack addresses within the tracked stack frame.");
14772 			return -EACCES;
14773 		}
14774 		dst_reg->r64 = cnum64_add(ptr_reg->r64, cnum64_negate(off_reg->r64));
14775 		dst_reg->var_off = tnum_sub(ptr_reg->var_off, off_reg->var_off);
14776 		dst_reg->raw = ptr_reg->raw;
14777 		if (reg_is_pkt_pointer(ptr_reg)) {
14778 			if (!known)
14779 				dst_reg->id = ++env->id_gen;
14780 			/*
14781 			 * Clear range if the subtrahend may be negative since
14782 			 * pkt pointer could move past its bounds. A positive
14783 			 * subtrahend moves it backwards keeping positive range
14784 			 * intact. Also clear AT_PKT_END / BEYOND_PKT_END from
14785 			 * prior comparison as arithmetic invalidates them.
14786 			 */
14787 			if ((!known && smin_val < 0) || dst_reg->range < 0)
14788 				memset(&dst_reg->raw, 0, sizeof(dst_reg->raw));
14789 		}
14790 		break;
14791 	case BPF_AND:
14792 	case BPF_OR:
14793 	case BPF_XOR:
14794 		/* bitwise ops on pointers are troublesome, prohibit. */
14795 		verbose(env, "R%d bitwise operator %s on pointer prohibited\n",
14796 			dst, bpf_alu_string[opcode >> 4]);
14797 		reason = bpf_diag_fmt(
14798 			env, "R%d holds %s. Bitwise operator %s would destroy the pointer value the verifier is tracking.",
14799 			ptr_regno, bpf_diag_reg_type_plain(env, ptr_reg->type),
14800 			bpf_alu_string[opcode >> 4]);
14801 		bpf_diag_register_type(
14802 			env, env->insn_idx, ptr_regno, "bitwise operation on pointer", reason,
14803 			"Do bitwise operations on scalar values, not on pointer-valued registers.");
14804 		return -EACCES;
14805 	default:
14806 		/* other operators (e.g. MUL,LSH) produce non-pointer results */
14807 		verbose(env, "R%d pointer arithmetic with %s operator prohibited\n",
14808 			dst, bpf_alu_string[opcode >> 4]);
14809 		reason = bpf_diag_fmt(
14810 			env, "R%d holds %s. Operator %s is not one of the limited pointer arithmetic operations the verifier can track.",
14811 			ptr_regno, bpf_diag_reg_type_plain(env, ptr_reg->type),
14812 			bpf_alu_string[opcode >> 4]);
14813 		bpf_diag_register_type(
14814 			env, env->insn_idx, ptr_regno, "invalid pointer arithmetic operator", reason,
14815 			"Use only verifier-supported addition or subtraction with a bounded scalar offset, or perform this operation on a scalar value.");
14816 		return -EACCES;
14817 	}
14818 
14819 	if (!check_reg_sane_offset_ptr(env, dst_reg, ptr_reg->type)) {
14820 		reason = bpf_diag_fmt(
14821 			env, "After this arithmetic, R%d would be outside the verifier's safe offset range [-%u, %u] for %s.",
14822 			dst, BPF_MAX_VAR_OFF, BPF_MAX_VAR_OFF,
14823 			bpf_diag_reg_type_plain(env, ptr_reg->type));
14824 		bpf_diag_register_type(
14825 			env, env->insn_idx, ptr_regno, "pointer offset is not safe", reason,
14826 			"Tighten the scalar bounds before the arithmetic so the resulting pointer remains within the allowed range.");
14827 		return -EINVAL;
14828 	}
14829 	/*
14830 	 * A packet pointer that keeps its id or range is checked against a
14831 	 * range set from the checked pointer's umax, so var_off must not tighten
14832 	 * its umax. r32 must still match var_off for reg_bounds_sanity_check().
14833 	 */
14834 	if (reg_is_pkt_pointer(dst_reg) && (known || dst_reg->range > 0))
14835 		__update_reg32_bounds(dst_reg);
14836 	else
14837 		reg_bounds_sync(dst_reg);
14838 	bounds_ret = sanitize_check_bounds(env, insn, dst_reg);
14839 	if (bounds_ret == -EACCES)
14840 		return bounds_ret;
14841 	if (sanitize_needed(opcode)) {
14842 		ret = sanitize_ptr_alu(env, insn, dst_reg, off_reg, dst_reg,
14843 				       &info, true);
14844 		if (verifier_bug_if(!can_skip_alu_sanitation(env, insn)
14845 				    && !env->cur_state->speculative
14846 				    && bounds_ret
14847 				    && !ret,
14848 				    env, "Pointer type unsupported by sanitize_check_bounds() not rejected by retrieve_ptr_limit() as required")) {
14849 			return -EFAULT;
14850 		}
14851 		if (ret < 0)
14852 			return sanitize_err(env, insn, ret);
14853 	}
14854 
14855 	return 0;
14856 }
14857 
14858 static void scalar32_min_max_add(struct bpf_reg_state *dst_reg,
14859 				 struct bpf_reg_state *src_reg)
14860 {
14861 	dst_reg->r32 = cnum32_add(dst_reg->r32, src_reg->r32);
14862 }
14863 
14864 static void scalar_min_max_add(struct bpf_reg_state *dst_reg,
14865 			       struct bpf_reg_state *src_reg)
14866 {
14867 	dst_reg->r64 = cnum64_add(dst_reg->r64, src_reg->r64);
14868 }
14869 
14870 static void scalar32_min_max_sub(struct bpf_reg_state *dst_reg,
14871 				 struct bpf_reg_state *src_reg)
14872 {
14873 	dst_reg->r32 = cnum32_add(dst_reg->r32, cnum32_negate(src_reg->r32));
14874 }
14875 
14876 static void scalar_min_max_sub(struct bpf_reg_state *dst_reg,
14877 			       struct bpf_reg_state *src_reg)
14878 {
14879 	dst_reg->r64 = cnum64_add(dst_reg->r64, cnum64_negate(src_reg->r64));
14880 }
14881 
14882 static void scalar32_min_max_mul(struct bpf_reg_state *dst_reg,
14883 				 struct bpf_reg_state *src_reg)
14884 {
14885 	s32 smin = reg_s32_min(dst_reg);
14886 	s32 smax = reg_s32_max(dst_reg);
14887 	u32 umin = reg_u32_min(dst_reg);
14888 	u32 umax = reg_u32_max(dst_reg);
14889 	s32 tmp_prod[4];
14890 
14891 	if (check_mul_overflow(umax, reg_u32_max(src_reg), &umax) ||
14892 	    check_mul_overflow(umin, reg_u32_min(src_reg), &umin)) {
14893 		/* Overflow possible, we know nothing */
14894 		umin = 0;
14895 		umax = U32_MAX;
14896 	}
14897 	if (check_mul_overflow(smin, reg_s32_min(src_reg), &tmp_prod[0]) ||
14898 	    check_mul_overflow(smin, reg_s32_max(src_reg), &tmp_prod[1]) ||
14899 	    check_mul_overflow(smax, reg_s32_min(src_reg), &tmp_prod[2]) ||
14900 	    check_mul_overflow(smax, reg_s32_max(src_reg), &tmp_prod[3])) {
14901 		/* Overflow possible, we know nothing */
14902 		smin = S32_MIN;
14903 		smax = S32_MAX;
14904 	} else {
14905 		smin = min_array(tmp_prod, 4);
14906 		smax = max_array(tmp_prod, 4);
14907 	}
14908 
14909 	dst_reg->r32 = cnum32_intersect(cnum32_from_urange(umin, umax),
14910 					cnum32_from_srange(smin, smax));
14911 }
14912 
14913 static void scalar_min_max_mul(struct bpf_reg_state *dst_reg,
14914 			       struct bpf_reg_state *src_reg)
14915 {
14916 	s64 smin = reg_smin(dst_reg);
14917 	s64 smax = reg_smax(dst_reg);
14918 	u64 umin = reg_umin(dst_reg);
14919 	u64 umax = reg_umax(dst_reg);
14920 	s64 tmp_prod[4];
14921 
14922 	if (check_mul_overflow(umax, reg_umax(src_reg), &umax) ||
14923 	    check_mul_overflow(umin, reg_umin(src_reg), &umin)) {
14924 		/* Overflow possible, we know nothing */
14925 		umin = 0;
14926 		umax = U64_MAX;
14927 	}
14928 	if (check_mul_overflow(smin, reg_smin(src_reg), &tmp_prod[0]) ||
14929 	    check_mul_overflow(smin, reg_smax(src_reg), &tmp_prod[1]) ||
14930 	    check_mul_overflow(smax, reg_smin(src_reg), &tmp_prod[2]) ||
14931 	    check_mul_overflow(smax, reg_smax(src_reg), &tmp_prod[3])) {
14932 		/* Overflow possible, we know nothing */
14933 		smin = S64_MIN;
14934 		smax = S64_MAX;
14935 	} else {
14936 		smin = min_array(tmp_prod, 4);
14937 		smax = max_array(tmp_prod, 4);
14938 	}
14939 
14940 	dst_reg->r64 = cnum64_intersect(cnum64_from_urange(umin, umax),
14941 					cnum64_from_srange(smin, smax));
14942 }
14943 
14944 static void scalar32_min_max_udiv(struct bpf_reg_state *dst_reg,
14945 				  struct bpf_reg_state *src_reg)
14946 {
14947 	u32 src_val = reg_u32_min(src_reg); /* non-zero, const divisor */
14948 
14949 	reg_set_urange32(dst_reg, reg_u32_min(dst_reg) / src_val,
14950 			 reg_u32_max(dst_reg) / src_val);
14951 
14952 	/* Reset other ranges/tnum to unbounded/unknown. */
14953 	reset_reg64_and_tnum(dst_reg);
14954 }
14955 
14956 static void scalar_min_max_udiv(struct bpf_reg_state *dst_reg,
14957 				struct bpf_reg_state *src_reg)
14958 {
14959 	u64 src_val = reg_umin(src_reg); /* non-zero, const divisor */
14960 
14961 	reg_set_urange64(dst_reg, div64_u64(reg_umin(dst_reg), src_val),
14962 			 div64_u64(reg_umax(dst_reg), src_val));
14963 
14964 	/* Reset other ranges/tnum to unbounded/unknown. */
14965 	reset_reg32_and_tnum(dst_reg);
14966 }
14967 
14968 static void scalar32_min_max_sdiv(struct bpf_reg_state *dst_reg,
14969 				  struct bpf_reg_state *src_reg)
14970 {
14971 	s32 smin = reg_s32_min(dst_reg);
14972 	s32 smax = reg_s32_max(dst_reg);
14973 	s32 src_val = reg_s32_min(src_reg); /* non-zero, const divisor */
14974 	s32 res1, res2;
14975 
14976 	/* BPF div specification: S32_MIN / -1 = S32_MIN */
14977 	if (smin == S32_MIN && src_val == -1) {
14978 		/*
14979 		 * If the dividend range contains more than just S32_MIN,
14980 		 * we cannot precisely track the result, so it becomes unbounded.
14981 		 * e.g., [S32_MIN, S32_MIN+10]/(-1),
14982 		 *     = {S32_MIN} U [-(S32_MIN+10), -(S32_MIN+1)]
14983 		 *     = {S32_MIN} U [S32_MAX-9, S32_MAX] = [S32_MIN, S32_MAX]
14984 		 * Otherwise (if dividend is exactly S32_MIN), result remains S32_MIN.
14985 		 */
14986 		if (smax != S32_MIN) {
14987 			smin = S32_MIN;
14988 			smax = S32_MAX;
14989 		}
14990 		goto reset;
14991 	}
14992 
14993 	res1 = smin / src_val;
14994 	res2 = smax / src_val;
14995 	smin = min(res1, res2);
14996 	smax = max(res1, res2);
14997 
14998 reset:
14999 	reg_set_srange32(dst_reg, smin, smax);
15000 	/* Reset other ranges/tnum to unbounded/unknown. */
15001 	reset_reg64_and_tnum(dst_reg);
15002 }
15003 
15004 static void scalar_min_max_sdiv(struct bpf_reg_state *dst_reg,
15005 				struct bpf_reg_state *src_reg)
15006 {
15007 	s64 smin = reg_smin(dst_reg);
15008 	s64 smax = reg_smax(dst_reg);
15009 	s64 src_val = reg_smin(src_reg); /* non-zero, const divisor */
15010 	s64 res1, res2;
15011 
15012 	/* BPF div specification: S64_MIN / -1 = S64_MIN */
15013 	if (smin == S64_MIN && src_val == -1) {
15014 		/*
15015 		 * If the dividend range contains more than just S64_MIN,
15016 		 * we cannot precisely track the result, so it becomes unbounded.
15017 		 * e.g., [S64_MIN, S64_MIN+10]/(-1),
15018 		 *     = {S64_MIN} U [-(S64_MIN+10), -(S64_MIN+1)]
15019 		 *     = {S64_MIN} U [S64_MAX-9, S64_MAX] = [S64_MIN, S64_MAX]
15020 		 * Otherwise (if dividend is exactly S64_MIN), result remains S64_MIN.
15021 		 */
15022 		if (smax != S64_MIN) {
15023 			smin = S64_MIN;
15024 			smax = S64_MAX;
15025 		}
15026 		goto reset;
15027 	}
15028 
15029 	res1 = div64_s64(smin, src_val);
15030 	res2 = div64_s64(smax, src_val);
15031 	smin = min(res1, res2);
15032 	smax = max(res1, res2);
15033 
15034 reset:
15035 	reg_set_srange64(dst_reg, smin, smax);
15036 	/* Reset other ranges/tnum to unbounded/unknown. */
15037 	reset_reg32_and_tnum(dst_reg);
15038 }
15039 
15040 static void scalar32_min_max_umod(struct bpf_reg_state *dst_reg,
15041 				  struct bpf_reg_state *src_reg)
15042 {
15043 	u32 src_val = reg_u32_min(src_reg); /* non-zero, const divisor */
15044 	u32 res_max = src_val - 1;
15045 
15046 	/*
15047 	 * If dst_umax <= res_max, the result remains unchanged.
15048 	 * e.g., [2, 5] % 10 = [2, 5].
15049 	 */
15050 	if (reg_u32_max(dst_reg) <= res_max)
15051 		return;
15052 
15053 	reg_set_urange32(dst_reg, 0, min(reg_u32_max(dst_reg), res_max));
15054 
15055 	/* Reset other ranges/tnum to unbounded/unknown. */
15056 	reset_reg64_and_tnum(dst_reg);
15057 }
15058 
15059 static void scalar_min_max_umod(struct bpf_reg_state *dst_reg,
15060 				struct bpf_reg_state *src_reg)
15061 {
15062 	u64 src_val = reg_umin(src_reg); /* non-zero, const divisor */
15063 	u64 res_max = src_val - 1;
15064 
15065 	/*
15066 	 * If dst_umax <= res_max, the result remains unchanged.
15067 	 * e.g., [2, 5] % 10 = [2, 5].
15068 	 */
15069 	if (reg_umax(dst_reg) <= res_max)
15070 		return;
15071 
15072 	reg_set_urange64(dst_reg, 0, min(reg_umax(dst_reg), res_max));
15073 
15074 	/* Reset other ranges/tnum to unbounded/unknown. */
15075 	reset_reg32_and_tnum(dst_reg);
15076 }
15077 
15078 static void scalar32_min_max_smod(struct bpf_reg_state *dst_reg,
15079 				  struct bpf_reg_state *src_reg)
15080 {
15081 	s32 src_val = reg_s32_min(src_reg); /* non-zero, const divisor */
15082 
15083 	/*
15084 	 * Safe absolute value calculation:
15085 	 * If src_val == S32_MIN (-2147483648), src_abs becomes 2147483648.
15086 	 * Here use unsigned integer to avoid overflow.
15087 	 */
15088 	u32 src_abs = (src_val > 0) ? (u32)src_val : -(u32)src_val;
15089 
15090 	/*
15091 	 * Calculate the maximum possible absolute value of the result.
15092 	 * Even if src_abs is 2147483648 (S32_MIN), subtracting 1 gives
15093 	 * 2147483647 (S32_MAX), which fits perfectly in s32.
15094 	 */
15095 	s32 res_max_abs = src_abs - 1;
15096 
15097 	/*
15098 	 * If the dividend is already within the result range,
15099 	 * the result remains unchanged. e.g., [-2, 5] % 10 = [-2, 5].
15100 	 */
15101 	if (reg_s32_min(dst_reg) >= -res_max_abs && reg_s32_max(dst_reg) <= res_max_abs)
15102 		return;
15103 
15104 	/* General case: result has the same sign as the dividend. */
15105 	if (reg_s32_min(dst_reg) >= 0) {
15106 		reg_set_srange32(dst_reg, 0, min(reg_s32_max(dst_reg), res_max_abs));
15107 	} else if (reg_s32_max(dst_reg) <= 0) {
15108 		reg_set_srange32(dst_reg, max(reg_s32_min(dst_reg), -res_max_abs), 0);
15109 	} else {
15110 		reg_set_srange32(dst_reg, -res_max_abs, res_max_abs);
15111 	}
15112 
15113 	/* Reset other ranges/tnum to unbounded/unknown. */
15114 	reset_reg64_and_tnum(dst_reg);
15115 }
15116 
15117 static void scalar_min_max_smod(struct bpf_reg_state *dst_reg,
15118 				struct bpf_reg_state *src_reg)
15119 {
15120 	s64 src_val = reg_smin(src_reg); /* non-zero, const divisor */
15121 
15122 	/*
15123 	 * Safe absolute value calculation:
15124 	 * If src_val == S64_MIN (-2^63), src_abs becomes 2^63.
15125 	 * Here use unsigned integer to avoid overflow.
15126 	 */
15127 	u64 src_abs = (src_val > 0) ? (u64)src_val : -(u64)src_val;
15128 
15129 	/*
15130 	 * Calculate the maximum possible absolute value of the result.
15131 	 * Even if src_abs is 2^63 (S64_MIN), subtracting 1 gives
15132 	 * 2^63 - 1 (S64_MAX), which fits perfectly in s64.
15133 	 */
15134 	s64 res_max_abs = src_abs - 1;
15135 
15136 	/*
15137 	 * If the dividend is already within the result range,
15138 	 * the result remains unchanged. e.g., [-2, 5] % 10 = [-2, 5].
15139 	 */
15140 	if (reg_smin(dst_reg) >= -res_max_abs && reg_smax(dst_reg) <= res_max_abs)
15141 		return;
15142 
15143 	/* General case: result has the same sign as the dividend. */
15144 	if (reg_smin(dst_reg) >= 0) {
15145 		reg_set_srange64(dst_reg, 0, min(reg_smax(dst_reg), res_max_abs));
15146 	} else if (reg_smax(dst_reg) <= 0) {
15147 		reg_set_srange64(dst_reg, max(reg_smin(dst_reg), -res_max_abs), 0);
15148 	} else {
15149 		reg_set_srange64(dst_reg, -res_max_abs, res_max_abs);
15150 	}
15151 
15152 	/* Reset other ranges/tnum to unbounded/unknown. */
15153 	reset_reg32_and_tnum(dst_reg);
15154 }
15155 
15156 static void scalar32_min_max_and(struct bpf_reg_state *dst_reg,
15157 				 struct bpf_reg_state *src_reg)
15158 {
15159 	bool src_known = tnum_subreg_is_const(src_reg->var_off);
15160 	bool dst_known = tnum_subreg_is_const(dst_reg->var_off);
15161 	struct tnum var32_off = tnum_subreg(dst_reg->var_off);
15162 	u32 umax_val = reg_u32_max(src_reg);
15163 
15164 	if (src_known && dst_known) {
15165 		__mark_reg32_known(dst_reg, var32_off.value);
15166 		return;
15167 	}
15168 
15169 	/* We get our minimum from the var_off, since that's inherently
15170 	 * bitwise.  Our maximum is the minimum of the operands' maxima.
15171 	 */
15172 	reg_set_urange32(dst_reg,
15173 			 var32_off.value,
15174 			 min(reg_u32_max(dst_reg), umax_val));
15175 }
15176 
15177 static void scalar_min_max_and(struct bpf_reg_state *dst_reg,
15178 			       struct bpf_reg_state *src_reg)
15179 {
15180 	bool src_known = tnum_is_const(src_reg->var_off);
15181 	bool dst_known = tnum_is_const(dst_reg->var_off);
15182 	u64 umax_val = reg_umax(src_reg);
15183 
15184 	if (src_known && dst_known) {
15185 		__mark_reg_known(dst_reg, dst_reg->var_off.value);
15186 		return;
15187 	}
15188 
15189 	/* We get our minimum from the var_off, since that's inherently
15190 	 * bitwise.  Our maximum is the minimum of the operands' maxima.
15191 	 */
15192 	reg_set_urange64(dst_reg,
15193 			 dst_reg->var_off.value,
15194 			 min(reg_umax(dst_reg), umax_val));
15195 
15196 	/* We may learn something more from the var_off */
15197 	__update_reg_bounds(dst_reg);
15198 }
15199 
15200 static void scalar32_min_max_or(struct bpf_reg_state *dst_reg,
15201 				struct bpf_reg_state *src_reg)
15202 {
15203 	bool src_known = tnum_subreg_is_const(src_reg->var_off);
15204 	bool dst_known = tnum_subreg_is_const(dst_reg->var_off);
15205 	struct tnum var32_off = tnum_subreg(dst_reg->var_off);
15206 	u32 umin_val = reg_u32_min(src_reg);
15207 
15208 	if (src_known && dst_known) {
15209 		__mark_reg32_known(dst_reg, var32_off.value);
15210 		return;
15211 	}
15212 
15213 	/* We get our maximum from the var_off, and our minimum is the
15214 	 * maximum of the operands' minima
15215 	 */
15216 	reg_set_urange32(dst_reg,
15217 			 max(reg_u32_min(dst_reg), umin_val),
15218 			 var32_off.value | var32_off.mask);
15219 }
15220 
15221 static void scalar_min_max_or(struct bpf_reg_state *dst_reg,
15222 			      struct bpf_reg_state *src_reg)
15223 {
15224 	bool src_known = tnum_is_const(src_reg->var_off);
15225 	bool dst_known = tnum_is_const(dst_reg->var_off);
15226 	u64 umin_val = reg_umin(src_reg);
15227 
15228 	if (src_known && dst_known) {
15229 		__mark_reg_known(dst_reg, dst_reg->var_off.value);
15230 		return;
15231 	}
15232 
15233 	/* We get our maximum from the var_off, and our minimum is the
15234 	 * maximum of the operands' minima
15235 	 */
15236 	reg_set_urange64(dst_reg,
15237 			 max(reg_umin(dst_reg), umin_val),
15238 			 dst_reg->var_off.value | dst_reg->var_off.mask);
15239 
15240 	/* We may learn something more from the var_off */
15241 	__update_reg_bounds(dst_reg);
15242 }
15243 
15244 static void scalar32_min_max_xor(struct bpf_reg_state *dst_reg,
15245 				 struct bpf_reg_state *src_reg)
15246 {
15247 	bool src_known = tnum_subreg_is_const(src_reg->var_off);
15248 	bool dst_known = tnum_subreg_is_const(dst_reg->var_off);
15249 	struct tnum var32_off = tnum_subreg(dst_reg->var_off);
15250 
15251 	if (src_known && dst_known) {
15252 		__mark_reg32_known(dst_reg, var32_off.value);
15253 		return;
15254 	}
15255 
15256 	/* We get both minimum and maximum from the var32_off. */
15257 	reg_set_urange32(dst_reg, var32_off.value, var32_off.value | var32_off.mask);
15258 }
15259 
15260 static void scalar_min_max_xor(struct bpf_reg_state *dst_reg,
15261 			       struct bpf_reg_state *src_reg)
15262 {
15263 	bool src_known = tnum_is_const(src_reg->var_off);
15264 	bool dst_known = tnum_is_const(dst_reg->var_off);
15265 
15266 	if (src_known && dst_known) {
15267 		/* dst_reg->var_off.value has been updated earlier */
15268 		__mark_reg_known(dst_reg, dst_reg->var_off.value);
15269 		return;
15270 	}
15271 
15272 	/* We get both minimum and maximum from the var_off. */
15273 	reg_set_urange64(dst_reg,
15274 			 dst_reg->var_off.value,
15275 			 dst_reg->var_off.value | dst_reg->var_off.mask);
15276 }
15277 
15278 static void __scalar32_min_max_lsh(struct bpf_reg_state *dst_reg,
15279 				   u64 umin_val, u64 umax_val)
15280 {
15281 	/* If we might shift our top bit out, then we know nothing */
15282 	if (umax_val > 31 || reg_u32_max(dst_reg) > 1ULL << (31 - umax_val))
15283 		reg_set_urange32(dst_reg, 0, U32_MAX);
15284 	else
15285 		/* We lose all sign bit information (except what we can pick
15286 		 * up from var_off)
15287 		 */
15288 		reg_set_urange32(dst_reg, reg_u32_min(dst_reg) << umin_val,
15289 				 reg_u32_max(dst_reg) << umax_val);
15290 }
15291 
15292 static void scalar32_min_max_lsh(struct bpf_reg_state *dst_reg,
15293 				 struct bpf_reg_state *src_reg)
15294 {
15295 	u32 umax_val = reg_u32_max(src_reg);
15296 	u32 umin_val = reg_u32_min(src_reg);
15297 	/* u32 alu operation will zext upper bits */
15298 	struct tnum subreg = tnum_subreg(dst_reg->var_off);
15299 
15300 	__scalar32_min_max_lsh(dst_reg, umin_val, umax_val);
15301 	dst_reg->var_off = tnum_subreg(tnum_lshift(subreg, umin_val));
15302 	/* Not required but being careful mark reg64 bounds as unknown so
15303 	 * that we are forced to pick them up from tnum and zext later and
15304 	 * if some path skips this step we are still safe.
15305 	 */
15306 	__mark_reg64_unbounded(dst_reg);
15307 	__update_reg32_bounds(dst_reg);
15308 }
15309 
15310 static void __scalar64_min_max_lsh(struct bpf_reg_state *dst_reg,
15311 				   u64 umin_val, u64 umax_val)
15312 {
15313 	struct cnum64 u, s;
15314 
15315 	/* Special case <<32 because it is a common compiler pattern to sign
15316 	 * extend subreg by doing <<32 s>>32. smin/smax assignments are correct
15317 	 * because s32 bounds don't flip sign when shifting to the left by
15318 	 * 32bits.
15319 	 */
15320 	if (umin_val == 32 && umax_val == 32)
15321 		s = cnum64_from_srange((s64)reg_s32_min(dst_reg) << 32,
15322 				       (s64)reg_s32_max(dst_reg) << 32);
15323 	else
15324 		s = CNUM64_UNBOUNDED;
15325 
15326 	/* If we might shift our top bit out, then we know nothing */
15327 	if (reg_umax(dst_reg) > 1ULL << (63 - umax_val))
15328 		u = CNUM64_UNBOUNDED;
15329 	else
15330 		u = cnum64_from_urange(reg_umin(dst_reg) << umin_val,
15331 				       reg_umax(dst_reg) << umax_val);
15332 
15333 	dst_reg->r64 = cnum64_intersect(u, s);
15334 }
15335 
15336 static void scalar_min_max_lsh(struct bpf_reg_state *dst_reg,
15337 			       struct bpf_reg_state *src_reg)
15338 {
15339 	u64 umax_val = reg_umax(src_reg);
15340 	u64 umin_val = reg_umin(src_reg);
15341 
15342 	/* scalar64 calc uses 32bit unshifted bounds so must be called first */
15343 	__scalar64_min_max_lsh(dst_reg, umin_val, umax_val);
15344 	__scalar32_min_max_lsh(dst_reg, umin_val, umax_val);
15345 
15346 	dst_reg->var_off = tnum_lshift(dst_reg->var_off, umin_val);
15347 	/* We may learn something more from the var_off */
15348 	__update_reg_bounds(dst_reg);
15349 }
15350 
15351 static void scalar32_min_max_rsh(struct bpf_reg_state *dst_reg,
15352 				 struct bpf_reg_state *src_reg)
15353 {
15354 	struct tnum subreg = tnum_subreg(dst_reg->var_off);
15355 	u32 umax_val = reg_u32_max(src_reg);
15356 	u32 umin_val = reg_u32_min(src_reg);
15357 
15358 	/* BPF_RSH is an unsigned shift.  If the value in dst_reg might
15359 	 * be negative, then either:
15360 	 * 1) src_reg might be zero, so the sign bit of the result is
15361 	 *    unknown, so we lose our signed bounds
15362 	 * 2) it's known negative, thus the unsigned bounds capture the
15363 	 *    signed bounds
15364 	 * 3) the signed bounds cross zero, so they tell us nothing
15365 	 *    about the result
15366 	 * If the value in dst_reg is known nonnegative, then again the
15367 	 * unsigned bounds capture the signed bounds.
15368 	 * Thus, in all cases it suffices to blow away our signed bounds
15369 	 * and rely on inferring new ones from the unsigned bounds and
15370 	 * var_off of the result.
15371 	 */
15372 
15373 	dst_reg->var_off = tnum_rshift(subreg, umin_val);
15374 	reg_set_urange32(dst_reg, reg_u32_min(dst_reg) >> umax_val,
15375 			 reg_u32_max(dst_reg) >> umin_val);
15376 
15377 	__mark_reg64_unbounded(dst_reg);
15378 	__update_reg32_bounds(dst_reg);
15379 }
15380 
15381 static void scalar_min_max_rsh(struct bpf_reg_state *dst_reg,
15382 			       struct bpf_reg_state *src_reg)
15383 {
15384 	u64 umax_val = reg_umax(src_reg);
15385 	u64 umin_val = reg_umin(src_reg);
15386 
15387 	/* BPF_RSH is an unsigned shift.  If the value in dst_reg might
15388 	 * be negative, then either:
15389 	 * 1) src_reg might be zero, so the sign bit of the result is
15390 	 *    unknown, so we lose our signed bounds
15391 	 * 2) it's known negative, thus the unsigned bounds capture the
15392 	 *    signed bounds
15393 	 * 3) the signed bounds cross zero, so they tell us nothing
15394 	 *    about the result
15395 	 * If the value in dst_reg is known nonnegative, then again the
15396 	 * unsigned bounds capture the signed bounds.
15397 	 * Thus, in all cases it suffices to blow away our signed bounds
15398 	 * and rely on inferring new ones from the unsigned bounds and
15399 	 * var_off of the result.
15400 	 */
15401 	dst_reg->var_off = tnum_rshift(dst_reg->var_off, umin_val);
15402 	reg_set_urange64(dst_reg, reg_umin(dst_reg) >> umax_val,
15403 			 reg_umax(dst_reg) >> umin_val);
15404 
15405 	/* Its not easy to operate on alu32 bounds here because it depends
15406 	 * on bits being shifted in. Take easy way out and mark unbounded
15407 	 * so we can recalculate later from tnum.
15408 	 */
15409 	__mark_reg32_unbounded(dst_reg);
15410 	__update_reg_bounds(dst_reg);
15411 }
15412 
15413 static void scalar32_min_max_arsh(struct bpf_reg_state *dst_reg,
15414 				  struct bpf_reg_state *src_reg)
15415 {
15416 	u64 umin_val = reg_u32_min(src_reg);
15417 
15418 	/* Upon reaching here, src_known is true and
15419 	 * umax_val is equal to umin_val.
15420 	 * Blow away the dst_reg umin_value/umax_value and rely on
15421 	 * dst_reg var_off to refine the result.
15422 	 */
15423 	reg_set_srange32(dst_reg,
15424 			 (u32)(((s32)reg_s32_min(dst_reg)) >> umin_val),
15425 			 (u32)(((s32)reg_s32_max(dst_reg)) >> umin_val));
15426 
15427 	dst_reg->var_off = tnum_arshift(tnum_subreg(dst_reg->var_off), umin_val, 32);
15428 
15429 	__mark_reg64_unbounded(dst_reg);
15430 	__update_reg32_bounds(dst_reg);
15431 }
15432 
15433 static void scalar_min_max_arsh(struct bpf_reg_state *dst_reg,
15434 				struct bpf_reg_state *src_reg)
15435 {
15436 	u64 umin_val = reg_umin(src_reg);
15437 
15438 	/* Upon reaching here, src_known is true and umax_val is equal
15439 	 * to umin_val.
15440 	 */
15441 	reg_set_srange64(dst_reg, reg_smin(dst_reg) >> umin_val,
15442 			 reg_smax(dst_reg) >> umin_val);
15443 
15444 	dst_reg->var_off = tnum_arshift(dst_reg->var_off, umin_val, 64);
15445 
15446 	/* Its not easy to operate on alu32 bounds here because it depends
15447 	 * on bits being shifted in from upper 32-bits. Take easy way out
15448 	 * and mark unbounded so we can recalculate later from tnum.
15449 	 */
15450 	__mark_reg32_unbounded(dst_reg);
15451 	__update_reg_bounds(dst_reg);
15452 }
15453 
15454 static void scalar_byte_swap(struct bpf_reg_state *dst_reg, struct bpf_insn *insn)
15455 {
15456 	/*
15457 	 * Byte swap operation - update var_off using tnum_bswap.
15458 	 * Three cases:
15459 	 * 1. bswap(16|32|64): opcode=0xd7 (BPF_END | BPF_ALU64 | BPF_TO_LE)
15460 	 *    unconditional swap
15461 	 * 2. to_le(16|32|64): opcode=0xd4 (BPF_END | BPF_ALU | BPF_TO_LE)
15462 	 *    swap on big-endian, truncation or no-op on little-endian
15463 	 * 3. to_be(16|32|64): opcode=0xdc (BPF_END | BPF_ALU | BPF_TO_BE)
15464 	 *    swap on little-endian, truncation or no-op on big-endian
15465 	 */
15466 
15467 	bool alu64 = BPF_CLASS(insn->code) == BPF_ALU64;
15468 	bool to_le = BPF_SRC(insn->code) == BPF_TO_LE;
15469 	bool is_big_endian;
15470 #ifdef CONFIG_CPU_BIG_ENDIAN
15471 	is_big_endian = true;
15472 #else
15473 	is_big_endian = false;
15474 #endif
15475 	/* Apply bswap if alu64 or switch between big-endian and little-endian machines */
15476 	bool need_bswap = alu64 || (to_le == is_big_endian);
15477 
15478 	/*
15479 	 * If the register is mutated, manually reset its scalar ID to break
15480 	 * any existing ties and avoid incorrect bounds propagation.
15481 	 */
15482 	if (need_bswap || insn->imm == 16 || insn->imm == 32)
15483 		clear_scalar_id(dst_reg);
15484 
15485 	if (need_bswap) {
15486 		if (insn->imm == 16)
15487 			dst_reg->var_off = tnum_bswap16(dst_reg->var_off);
15488 		else if (insn->imm == 32)
15489 			dst_reg->var_off = tnum_bswap32(dst_reg->var_off);
15490 		else if (insn->imm == 64)
15491 			dst_reg->var_off = tnum_bswap64(dst_reg->var_off);
15492 		/*
15493 		 * Byteswap scrambles the range, so we must reset bounds.
15494 		 * Bounds will be re-derived from the new tnum later.
15495 		 */
15496 		__mark_reg_unbounded(dst_reg);
15497 	}
15498 	/* For bswap16/32, truncate dst register to match the swapped size */
15499 	if (insn->imm == 16 || insn->imm == 32)
15500 		coerce_reg_to_size(dst_reg, insn->imm / 8);
15501 }
15502 
15503 static bool is_safe_to_compute_dst_reg_range(struct bpf_insn *insn,
15504 					     const struct bpf_reg_state *src_reg)
15505 {
15506 	bool src_is_const = false;
15507 	u64 insn_bitness = (BPF_CLASS(insn->code) == BPF_ALU64) ? 64 : 32;
15508 
15509 	if (insn_bitness == 32) {
15510 		if (tnum_subreg_is_const(src_reg->var_off)
15511 		    && reg_s32_min(src_reg) == reg_s32_max(src_reg)
15512 		    && reg_u32_min(src_reg) == reg_u32_max(src_reg))
15513 			src_is_const = true;
15514 	} else {
15515 		if (tnum_is_const(src_reg->var_off)
15516 		    && reg_smin(src_reg) == reg_smax(src_reg)
15517 		    && reg_umin(src_reg) == reg_umax(src_reg))
15518 			src_is_const = true;
15519 	}
15520 
15521 	switch (BPF_OP(insn->code)) {
15522 	case BPF_ADD:
15523 	case BPF_SUB:
15524 	case BPF_NEG:
15525 	case BPF_AND:
15526 	case BPF_XOR:
15527 	case BPF_OR:
15528 	case BPF_MUL:
15529 	case BPF_END:
15530 		return true;
15531 
15532 	/*
15533 	 * Division and modulo operators range is only safe to compute when the
15534 	 * divisor is a constant.
15535 	 */
15536 	case BPF_DIV:
15537 	case BPF_MOD:
15538 		return src_is_const;
15539 
15540 	/* Shift operators range is only computable if shift dimension operand
15541 	 * is a constant. Shifts greater than 31 or 63 are undefined. This
15542 	 * includes shifts by a negative number.
15543 	 */
15544 	case BPF_LSH:
15545 	case BPF_RSH:
15546 	case BPF_ARSH:
15547 		return (src_is_const && reg_umax(src_reg) < insn_bitness);
15548 	default:
15549 		return false;
15550 	}
15551 }
15552 
15553 static int maybe_fork_scalars(struct bpf_verifier_env *env, struct bpf_insn *insn,
15554 			      struct bpf_reg_state *dst_reg)
15555 {
15556 	struct bpf_verifier_state *branch;
15557 	struct bpf_reg_state *regs;
15558 	bool alu32;
15559 
15560 	if (reg_smin(dst_reg) == -1 && reg_smax(dst_reg) == 0)
15561 		alu32 = false;
15562 	else if (reg_s32_min(dst_reg) == -1 && reg_s32_max(dst_reg) == 0)
15563 		alu32 = true;
15564 	else
15565 		return 0;
15566 
15567 	branch = push_stack(env, env->insn_idx, env->insn_idx, false);
15568 	if (IS_ERR(branch))
15569 		return PTR_ERR(branch);
15570 
15571 	regs = branch->frame[branch->curframe]->regs;
15572 	if (alu32) {
15573 		__mark_reg32_known(&regs[insn->dst_reg], 0);
15574 		__mark_reg32_known(dst_reg, -1ull);
15575 	} else {
15576 		__mark_reg_known(&regs[insn->dst_reg], 0);
15577 		__mark_reg_known(dst_reg, -1ull);
15578 	}
15579 	return 0;
15580 }
15581 
15582 /* WARNING: This function does calculations on 64-bit values, but the actual
15583  * execution may occur on 32-bit values. Therefore, things like bitshifts
15584  * need extra checks in the 32-bit case.
15585  */
15586 static int adjust_scalar_min_max_vals(struct bpf_verifier_env *env,
15587 				      struct bpf_insn *insn,
15588 				      struct bpf_reg_state *dst_reg,
15589 				      struct bpf_reg_state src_reg)
15590 {
15591 	u8 opcode = BPF_OP(insn->code);
15592 	s16 off = insn->off;
15593 	bool alu32 = (BPF_CLASS(insn->code) != BPF_ALU64);
15594 	int ret;
15595 
15596 	if (!is_safe_to_compute_dst_reg_range(insn, &src_reg)) {
15597 		__mark_reg_unknown(env, dst_reg);
15598 		return 0;
15599 	}
15600 
15601 	if (sanitize_needed(opcode)) {
15602 		ret = sanitize_val_alu(env, insn);
15603 		if (ret < 0)
15604 			return sanitize_err(env, insn, ret);
15605 	}
15606 
15607 	/* Calculate sign/unsigned bounds and tnum for alu32 and alu64 bit ops.
15608 	 * There are two classes of instructions: The first class we track both
15609 	 * alu32 and alu64 sign/unsigned bounds independently this provides the
15610 	 * greatest amount of precision when alu operations are mixed with jmp32
15611 	 * operations. These operations are BPF_ADD, BPF_SUB, BPF_MUL, BPF_ADD,
15612 	 * and BPF_OR. This is possible because these ops have fairly easy to
15613 	 * understand and calculate behavior in both 32-bit and 64-bit alu ops.
15614 	 * See alu32 verifier tests for examples. The second class of
15615 	 * operations, BPF_LSH, BPF_RSH, and BPF_ARSH, however are not so easy
15616 	 * with regards to tracking sign/unsigned bounds because the bits may
15617 	 * cross subreg boundaries in the alu64 case. When this happens we mark
15618 	 * the reg unbounded in the subreg bound space and use the resulting
15619 	 * tnum to calculate an approximation of the sign/unsigned bounds.
15620 	 */
15621 	switch (opcode) {
15622 	case BPF_ADD:
15623 		scalar32_min_max_add(dst_reg, &src_reg);
15624 		scalar_min_max_add(dst_reg, &src_reg);
15625 		dst_reg->var_off = tnum_add(dst_reg->var_off, src_reg.var_off);
15626 		break;
15627 	case BPF_SUB:
15628 		scalar32_min_max_sub(dst_reg, &src_reg);
15629 		scalar_min_max_sub(dst_reg, &src_reg);
15630 		dst_reg->var_off = tnum_sub(dst_reg->var_off, src_reg.var_off);
15631 		break;
15632 	case BPF_NEG:
15633 		env->fake_reg[0] = *dst_reg;
15634 		__mark_reg_known(dst_reg, 0);
15635 		scalar32_min_max_sub(dst_reg, &env->fake_reg[0]);
15636 		scalar_min_max_sub(dst_reg, &env->fake_reg[0]);
15637 		dst_reg->var_off = tnum_neg(env->fake_reg[0].var_off);
15638 		break;
15639 	case BPF_MUL:
15640 		dst_reg->var_off = tnum_mul(dst_reg->var_off, src_reg.var_off);
15641 		scalar32_min_max_mul(dst_reg, &src_reg);
15642 		scalar_min_max_mul(dst_reg, &src_reg);
15643 		break;
15644 	case BPF_DIV:
15645 		/* BPF div specification: x / 0 = 0 */
15646 		if ((alu32 && reg_u32_min(&src_reg) == 0) || (!alu32 && reg_umin(&src_reg) == 0)) {
15647 			___mark_reg_known(dst_reg, 0);
15648 			break;
15649 		}
15650 		if (alu32)
15651 			if (off == 1)
15652 				scalar32_min_max_sdiv(dst_reg, &src_reg);
15653 			else
15654 				scalar32_min_max_udiv(dst_reg, &src_reg);
15655 		else
15656 			if (off == 1)
15657 				scalar_min_max_sdiv(dst_reg, &src_reg);
15658 			else
15659 				scalar_min_max_udiv(dst_reg, &src_reg);
15660 		break;
15661 	case BPF_MOD:
15662 		/* BPF mod specification: x % 0 = x */
15663 		if ((alu32 && reg_u32_min(&src_reg) == 0) || (!alu32 && reg_umin(&src_reg) == 0))
15664 			break;
15665 		if (alu32)
15666 			if (off == 1)
15667 				scalar32_min_max_smod(dst_reg, &src_reg);
15668 			else
15669 				scalar32_min_max_umod(dst_reg, &src_reg);
15670 		else
15671 			if (off == 1)
15672 				scalar_min_max_smod(dst_reg, &src_reg);
15673 			else
15674 				scalar_min_max_umod(dst_reg, &src_reg);
15675 		break;
15676 	case BPF_AND:
15677 		if (tnum_is_const(src_reg.var_off)) {
15678 			ret = maybe_fork_scalars(env, insn, dst_reg);
15679 			if (ret)
15680 				return ret;
15681 		}
15682 		dst_reg->var_off = tnum_and(dst_reg->var_off, src_reg.var_off);
15683 		scalar32_min_max_and(dst_reg, &src_reg);
15684 		scalar_min_max_and(dst_reg, &src_reg);
15685 		break;
15686 	case BPF_OR:
15687 		if (tnum_is_const(src_reg.var_off)) {
15688 			ret = maybe_fork_scalars(env, insn, dst_reg);
15689 			if (ret)
15690 				return ret;
15691 		}
15692 		dst_reg->var_off = tnum_or(dst_reg->var_off, src_reg.var_off);
15693 		scalar32_min_max_or(dst_reg, &src_reg);
15694 		scalar_min_max_or(dst_reg, &src_reg);
15695 		break;
15696 	case BPF_XOR:
15697 		dst_reg->var_off = tnum_xor(dst_reg->var_off, src_reg.var_off);
15698 		scalar32_min_max_xor(dst_reg, &src_reg);
15699 		scalar_min_max_xor(dst_reg, &src_reg);
15700 		break;
15701 	case BPF_LSH:
15702 		if (alu32)
15703 			scalar32_min_max_lsh(dst_reg, &src_reg);
15704 		else
15705 			scalar_min_max_lsh(dst_reg, &src_reg);
15706 		break;
15707 	case BPF_RSH:
15708 		if (alu32)
15709 			scalar32_min_max_rsh(dst_reg, &src_reg);
15710 		else
15711 			scalar_min_max_rsh(dst_reg, &src_reg);
15712 		break;
15713 	case BPF_ARSH:
15714 		if (alu32)
15715 			scalar32_min_max_arsh(dst_reg, &src_reg);
15716 		else
15717 			scalar_min_max_arsh(dst_reg, &src_reg);
15718 		break;
15719 	case BPF_END:
15720 		scalar_byte_swap(dst_reg, insn);
15721 		break;
15722 	default:
15723 		break;
15724 	}
15725 
15726 	/*
15727 	 * ALU32 ops are zero extended into 64bit register.
15728 	 *
15729 	 * BPF_END is already handled inside the helper (truncation),
15730 	 * so skip zext here to avoid unexpected zero extension.
15731 	 * e.g., le64: opcode=(BPF_END|BPF_ALU|BPF_TO_LE), imm=0x40
15732 	 * This is a 64bit byte swap operation with alu32==true,
15733 	 * but we should not zero extend the result.
15734 	 */
15735 	if (alu32 && opcode != BPF_END)
15736 		zext_32_to_64(dst_reg);
15737 	reg_bounds_sync(dst_reg);
15738 	return 0;
15739 }
15740 
15741 /* Handles ALU ops other than BPF_END, BPF_NEG and BPF_MOV: computes new min/max
15742  * and var_off.
15743  */
15744 static int adjust_reg_min_max_vals(struct bpf_verifier_env *env,
15745 				   struct bpf_insn *insn)
15746 {
15747 	struct bpf_verifier_state *vstate = env->cur_state;
15748 	struct bpf_func_state *state = vstate->frame[vstate->curframe];
15749 	struct bpf_reg_state *regs = state->regs, *dst_reg, *src_reg;
15750 	struct bpf_reg_state *ptr_reg = NULL, off_reg = {0};
15751 	bool alu32 = (BPF_CLASS(insn->code) != BPF_ALU64);
15752 	struct bpf_insn_aux_data *aux = cur_aux(env);
15753 	u8 opcode = BPF_OP(insn->code);
15754 	int err;
15755 
15756 	dst_reg = &regs[insn->dst_reg];
15757 	if (BPF_SRC(insn->code) == BPF_X)
15758 		src_reg = &regs[insn->src_reg];
15759 	else
15760 		src_reg = NULL;
15761 
15762 	/* Case where at least one operand is an arena. */
15763 	if (dst_reg->type == PTR_TO_ARENA || (src_reg && src_reg->type == PTR_TO_ARENA)) {
15764 
15765 		if (dst_reg->type != PTR_TO_ARENA)
15766 			*dst_reg = *src_reg;
15767 
15768 		if (BPF_CLASS(insn->code) == BPF_ALU64) {
15769 			/*
15770 			 * Only arena pointers set needs_zext, but doing so
15771 			 * modifies the instruction at fixup time to an ALU32
15772 			 * and makes it unsuitable for 64-bit scalar args. We
15773 			 * prevent zext from being set if the instruction has
15774 			 * been previously called with non-arena registers.
15775 			 */
15776 			if (aux->prevent_zext) {
15777 				verbose(env, "same insn cannot be used with and without arena pointer\n");
15778 				return -EINVAL;
15779 			}
15780 
15781 			/*
15782 			 * 32-bit operations zero upper bits automatically.
15783 			 * 64-bit operations need to be converted to 32.
15784 			 */
15785 			aux->needs_zext = true;
15786 			aux->zext_dst = true;
15787 		}
15788 
15789 		/* Any arithmetic operations are allowed on arena pointers */
15790 		return 0;
15791 	}
15792 
15793 	/* Prevent the instruction from being used with arena pointers (see above). */
15794 	if (env->prog->aux->arena && BPF_CLASS(insn->code) == BPF_ALU64) {
15795 		if (aux->needs_zext) {
15796 			verbose(env, "same insn cannot be used with and without arena pointer\n");
15797 			return -EINVAL;
15798 		}
15799 
15800 		aux->prevent_zext = true;
15801 	}
15802 
15803 	if (dst_reg->type != SCALAR_VALUE)
15804 		ptr_reg = dst_reg;
15805 
15806 	if (BPF_SRC(insn->code) == BPF_X) {
15807 		if (src_reg->type != SCALAR_VALUE) {
15808 			if (dst_reg->type != SCALAR_VALUE) {
15809 				/* Combining two pointers by any ALU op yields
15810 				 * an arbitrary scalar. Disallow all math except
15811 				 * pointer subtraction
15812 				 */
15813 				if (opcode == BPF_SUB && env->allow_ptr_leaks) {
15814 					mark_reg_unknown(env, regs, insn->dst_reg);
15815 					return 0;
15816 				}
15817 				verbose(env, "R%d pointer %s pointer prohibited\n",
15818 					insn->dst_reg,
15819 					bpf_alu_string[opcode >> 4]);
15820 				return -EACCES;
15821 			} else {
15822 				/* scalar += pointer
15823 				 * This is legal, but we have to reverse our
15824 				 * src/dest handling in computing the range
15825 				 */
15826 				err = mark_chain_precision(env, insn->dst_reg);
15827 				if (err)
15828 					return err;
15829 				off_reg = *dst_reg;
15830 				return adjust_ptr_min_max_vals(env, insn, insn->src_reg, src_reg,
15831 							       &off_reg);
15832 			}
15833 		} else if (ptr_reg) {
15834 			/* pointer += scalar */
15835 			err = mark_chain_precision(env, insn->src_reg);
15836 			if (err)
15837 				return err;
15838 			return adjust_ptr_min_max_vals(env, insn, insn->dst_reg, dst_reg, src_reg);
15839 		} else if (dst_reg->precise) {
15840 			/* if dst_reg is precise, src_reg should be precise as well */
15841 			err = mark_chain_precision(env, insn->src_reg);
15842 			if (err)
15843 				return err;
15844 		}
15845 	} else {
15846 		/* Pretend the src is a reg with a known value, since we only
15847 		 * need to be able to read from this state.
15848 		 */
15849 		off_reg.type = SCALAR_VALUE;
15850 		__mark_reg_known(&off_reg, insn->imm);
15851 		src_reg = &off_reg;
15852 		if (ptr_reg) /* pointer += K */
15853 			return adjust_ptr_min_max_vals(env, insn, insn->dst_reg, ptr_reg, src_reg);
15854 	}
15855 
15856 	/* Got here implies adding two SCALAR_VALUEs */
15857 	if (WARN_ON_ONCE(ptr_reg)) {
15858 		print_verifier_state(env, vstate, vstate->curframe, true);
15859 		verbose(env, "verifier internal error: unexpected ptr_reg\n");
15860 		return -EFAULT;
15861 	}
15862 	if (WARN_ON(!src_reg)) {
15863 		print_verifier_state(env, vstate, vstate->curframe, true);
15864 		verbose(env, "verifier internal error: no src_reg\n");
15865 		return -EFAULT;
15866 	}
15867 	/*
15868 	 * For alu32 linked register tracking, we need to check dst_reg's
15869 	 * umax_value before the ALU operation. After adjust_scalar_min_max_vals(),
15870 	 * alu32 ops will have zero-extended the result, making umax_value <= U32_MAX.
15871 	 */
15872 	u64 dst_umax = reg_umax(dst_reg);
15873 
15874 	err = adjust_scalar_min_max_vals(env, insn, dst_reg, *src_reg);
15875 	if (err)
15876 		return err;
15877 	/*
15878 	 * Compilers can generate the code
15879 	 * r1 = r2
15880 	 * r1 += 0x1
15881 	 * if r2 < 1000 goto ...
15882 	 * use r1 in memory access
15883 	 * So remember constant delta between r2 and r1 and update r1 after
15884 	 * 'if' condition.
15885 	 */
15886 	if (env->bpf_capable &&
15887 	    (BPF_OP(insn->code) == BPF_ADD || BPF_OP(insn->code) == BPF_SUB) &&
15888 	    dst_reg->id && is_reg_const(src_reg, alu32) &&
15889 	    !(BPF_SRC(insn->code) == BPF_X && insn->src_reg == insn->dst_reg)) {
15890 		u64 val = reg_const_value(src_reg, alu32);
15891 		s32 off;
15892 
15893 		if (!alu32 && ((s64)val < S32_MIN || (s64)val > S32_MAX))
15894 			goto clear_id;
15895 
15896 		if (alu32 && (dst_umax > U32_MAX))
15897 			goto clear_id;
15898 
15899 		off = (s32)val;
15900 
15901 		if (BPF_OP(insn->code) == BPF_SUB) {
15902 			/* Negating S32_MIN would overflow */
15903 			if (off == S32_MIN)
15904 				goto clear_id;
15905 			off = -off;
15906 		}
15907 
15908 		if (dst_reg->id & BPF_ADD_CONST) {
15909 			/*
15910 			 * If the register already went through rX += val
15911 			 * we cannot accumulate another val into rx->off.
15912 			 */
15913 clear_id:
15914 			clear_scalar_id(dst_reg);
15915 		} else {
15916 			if (alu32)
15917 				dst_reg->id |= BPF_ADD_CONST32;
15918 			else
15919 				dst_reg->id |= BPF_ADD_CONST64;
15920 			dst_reg->delta = off;
15921 		}
15922 	} else {
15923 		/*
15924 		 * Make sure ID is cleared otherwise dst_reg min/max could be
15925 		 * incorrectly propagated into other registers by sync_linked_regs()
15926 		 */
15927 		clear_scalar_id(dst_reg);
15928 	}
15929 	return 0;
15930 }
15931 
15932 /* check validity of 32-bit and 64-bit arithmetic operations */
15933 static int check_alu_op(struct bpf_verifier_env *env, struct bpf_insn *insn)
15934 {
15935 	struct bpf_reg_state *regs = cur_regs(env);
15936 	u8 opcode = BPF_OP(insn->code);
15937 	int err;
15938 
15939 	bpf_diag_mod_begin(env, &regs[insn->dst_reg], NULL, BPF_DIAG_MOD_WRITE);
15940 
15941 	if (opcode == BPF_END || opcode == BPF_NEG) {
15942 		/* check src operand */
15943 		err = check_reg_arg(env, insn->dst_reg, SRC_OP);
15944 		if (err)
15945 			return err;
15946 
15947 		if (is_pointer_value(env, insn->dst_reg)) {
15948 			verbose(env, "R%d pointer arithmetic prohibited\n",
15949 				insn->dst_reg);
15950 			return -EACCES;
15951 		}
15952 
15953 		/* check dest operand */
15954 		if (regs[insn->dst_reg].type == SCALAR_VALUE) {
15955 			err = check_reg_arg(env, insn->dst_reg, DST_OP_NO_MARK);
15956 			err = err ?: adjust_scalar_min_max_vals(env, insn,
15957 							 &regs[insn->dst_reg],
15958 							 regs[insn->dst_reg]);
15959 		} else {
15960 			err = check_reg_arg(env, insn->dst_reg, DST_OP);
15961 		}
15962 		if (err)
15963 			return err;
15964 
15965 	} else if (opcode == BPF_MOV) {
15966 
15967 		if (BPF_SRC(insn->code) == BPF_X) {
15968 			if (insn->off == BPF_ADDR_SPACE_CAST) {
15969 				if (!env->prog->aux->arena) {
15970 					verbose(env, "addr_space_cast insn can only be used in a program that has an associated arena\n");
15971 					return -EINVAL;
15972 				}
15973 			}
15974 
15975 			/* check src operand */
15976 			err = check_reg_arg(env, insn->src_reg, SRC_OP);
15977 			if (err)
15978 				return err;
15979 		}
15980 
15981 		/* check dest operand, mark as required later */
15982 		err = check_reg_arg(env, insn->dst_reg, DST_OP_NO_MARK);
15983 		if (err)
15984 			return err;
15985 
15986 		if (BPF_SRC(insn->code) == BPF_X) {
15987 			struct bpf_reg_state *src_reg = regs + insn->src_reg;
15988 			struct bpf_reg_state *dst_reg = regs + insn->dst_reg;
15989 
15990 			if (BPF_CLASS(insn->code) == BPF_ALU64) {
15991 				if (insn->imm) {
15992 					/* off == BPF_ADDR_SPACE_CAST */
15993 					mark_reg_unknown(env, regs, insn->dst_reg);
15994 					if (insn->imm == 1) /* cast from as(1) to as(0) */
15995 						dst_reg->type = PTR_TO_ARENA;
15996 				} else if (insn->off == 0) {
15997 					/* case: R1 = R2
15998 					 * copy register state to dest reg
15999 					 */
16000 					assign_scalar_id_before_mov(env, src_reg);
16001 					*dst_reg = *src_reg;
16002 				} else {
16003 					/* case: R1 = (s8, s16 s32)R2 */
16004 					if (is_pointer_value(env, insn->src_reg)) {
16005 						verbose(env,
16006 							"R%d sign-extension part of pointer\n",
16007 							insn->src_reg);
16008 						return -EACCES;
16009 					} else if (src_reg->type == SCALAR_VALUE) {
16010 						bool no_sext;
16011 
16012 						no_sext = reg_umax(src_reg) < (1ULL << (insn->off - 1));
16013 						if (no_sext)
16014 							assign_scalar_id_before_mov(env, src_reg);
16015 						*dst_reg = *src_reg;
16016 						if (!no_sext)
16017 							clear_scalar_id(dst_reg);
16018 						coerce_reg_to_size_sx(dst_reg, insn->off >> 3);
16019 					} else {
16020 						mark_reg_unknown(env, regs, insn->dst_reg);
16021 					}
16022 				}
16023 			} else {
16024 				/* R1 = (u32) R2 */
16025 				if (is_pointer_value(env, insn->src_reg)) {
16026 					verbose(env,
16027 						"R%d partial copy of pointer\n",
16028 						insn->src_reg);
16029 					return -EACCES;
16030 				} else if (src_reg->type == SCALAR_VALUE) {
16031 					if (insn->off == 0) {
16032 						bool is_src_reg_u32 = get_reg_width(src_reg) <= 32;
16033 
16034 						if (is_src_reg_u32)
16035 							assign_scalar_id_before_mov(env, src_reg);
16036 						*dst_reg = *src_reg;
16037 						/* Make sure ID is cleared if src_reg is not in u32
16038 						 * range otherwise dst_reg min/max could be incorrectly
16039 						 * propagated into src_reg by sync_linked_regs()
16040 						 */
16041 						if (!is_src_reg_u32)
16042 							clear_scalar_id(dst_reg);
16043 					} else {
16044 						/* case: W1 = (s8, s16)W2 */
16045 						bool no_sext = reg_umax(src_reg) < (1ULL << (insn->off - 1));
16046 
16047 						if (no_sext)
16048 							assign_scalar_id_before_mov(env, src_reg);
16049 						*dst_reg = *src_reg;
16050 						if (!no_sext)
16051 							clear_scalar_id(dst_reg);
16052 						coerce_subreg_to_size_sx(dst_reg, insn->off >> 3);
16053 					}
16054 				} else {
16055 					mark_reg_unknown(env, regs,
16056 							 insn->dst_reg);
16057 				}
16058 				zext_32_to_64(dst_reg);
16059 				reg_bounds_sync(dst_reg);
16060 			}
16061 		} else {
16062 			/* case: R = imm
16063 			 * remember the value we stored into this reg
16064 			 */
16065 			/* clear any state __mark_reg_known doesn't set */
16066 			mark_reg_unknown(env, regs, insn->dst_reg);
16067 			regs[insn->dst_reg].type = SCALAR_VALUE;
16068 			if (BPF_CLASS(insn->code) == BPF_ALU64) {
16069 				__mark_reg_known(regs + insn->dst_reg,
16070 						 insn->imm);
16071 			} else {
16072 				__mark_reg_known(regs + insn->dst_reg,
16073 						 (u32)insn->imm);
16074 			}
16075 		}
16076 
16077 	} else {	/* all other ALU ops: and, sub, xor, add, ... */
16078 
16079 		if (BPF_SRC(insn->code) == BPF_X) {
16080 			/* check src1 operand */
16081 			err = check_reg_arg(env, insn->src_reg, SRC_OP);
16082 			if (err)
16083 				return err;
16084 		}
16085 
16086 		/* check src2 operand */
16087 		err = check_reg_arg(env, insn->dst_reg, SRC_OP);
16088 		if (err)
16089 			return err;
16090 
16091 		if ((opcode == BPF_MOD || opcode == BPF_DIV) &&
16092 		    BPF_SRC(insn->code) == BPF_K && insn->imm == 0) {
16093 			verbose(env, "div by zero\n");
16094 			return -EINVAL;
16095 		}
16096 
16097 		if ((opcode == BPF_LSH || opcode == BPF_RSH ||
16098 		     opcode == BPF_ARSH) && BPF_SRC(insn->code) == BPF_K) {
16099 			int size = BPF_CLASS(insn->code) == BPF_ALU64 ? 64 : 32;
16100 
16101 			if (insn->imm < 0 || insn->imm >= size) {
16102 				verbose(env, "invalid shift %d\n", insn->imm);
16103 				return -EINVAL;
16104 			}
16105 		}
16106 
16107 		/* check dest operand */
16108 		err = check_reg_arg(env, insn->dst_reg, DST_OP_NO_MARK);
16109 		err = err ?: adjust_reg_min_max_vals(env, insn);
16110 		if (err)
16111 			return err;
16112 	}
16113 
16114 	err = reg_bounds_sanity_check(env, &regs[insn->dst_reg], "alu");
16115 	if (err)
16116 		return err;
16117 
16118 	bpf_diag_mod_end(env);
16119 	return 0;
16120 }
16121 
16122 static void find_good_pkt_pointers(struct bpf_verifier_state *vstate,
16123 				   struct bpf_reg_state *dst_reg,
16124 				   enum bpf_reg_type type,
16125 				   bool range_right_open)
16126 {
16127 	struct bpf_func_state *state;
16128 	struct bpf_reg_state *reg;
16129 	int new_range;
16130 
16131 	if (reg_umax(dst_reg) == 0 && range_right_open)
16132 		/* This doesn't give us any range */
16133 		return;
16134 
16135 	if (reg_umax(dst_reg) > MAX_PACKET_OFF)
16136 		/* Risk of overflow.  For instance, ptr + (1<<63) may be less
16137 		 * than pkt_end, but that's because it's also less than pkt.
16138 		 */
16139 		return;
16140 
16141 	new_range = reg_umax(dst_reg);
16142 	if (range_right_open)
16143 		new_range++;
16144 
16145 	/* Examples for register markings:
16146 	 *
16147 	 * pkt_data in dst register:
16148 	 *
16149 	 *   r2 = r3;
16150 	 *   r2 += 8;
16151 	 *   if (r2 > pkt_end) goto <handle exception>
16152 	 *   <access okay>
16153 	 *
16154 	 *   r2 = r3;
16155 	 *   r2 += 8;
16156 	 *   if (r2 < pkt_end) goto <access okay>
16157 	 *   <handle exception>
16158 	 *
16159 	 *   Where:
16160 	 *     r2 == dst_reg, pkt_end == src_reg
16161 	 *     r2=pkt(id=n,off=8,r=0)
16162 	 *     r3=pkt(id=n,off=0,r=0)
16163 	 *
16164 	 * pkt_data in src register:
16165 	 *
16166 	 *   r2 = r3;
16167 	 *   r2 += 8;
16168 	 *   if (pkt_end >= r2) goto <access okay>
16169 	 *   <handle exception>
16170 	 *
16171 	 *   r2 = r3;
16172 	 *   r2 += 8;
16173 	 *   if (pkt_end <= r2) goto <handle exception>
16174 	 *   <access okay>
16175 	 *
16176 	 *   Where:
16177 	 *     pkt_end == dst_reg, r2 == src_reg
16178 	 *     r2=pkt(id=n,off=8,r=0)
16179 	 *     r3=pkt(id=n,off=0,r=0)
16180 	 *
16181 	 * Find register r3 and mark its range as r3=pkt(id=n,off=0,r=8)
16182 	 * or r3=pkt(id=n,off=0,r=8-1), so that range of bytes [r3, r3 + 8)
16183 	 * and [r3, r3 + 8-1) respectively is safe to access depending on
16184 	 * the check.
16185 	 */
16186 
16187 	/* If our ids match, then we must have the same max_value.  And we
16188 	 * don't care about the other reg's fixed offset, since if it's too big
16189 	 * the range won't allow anything.
16190 	 * reg_umax(dst_reg) is known < MAX_PACKET_OFF, therefore it fits in a u16.
16191 	 */
16192 	bpf_for_each_reg_in_vstate(vstate, state, reg, ({
16193 		if (reg->type == type && reg->id == dst_reg->id)
16194 			/* keep the maximum range already checked */
16195 			reg->range = max(reg->range, new_range);
16196 	}));
16197 }
16198 
16199 static void regs_refine_cond_op(struct bpf_reg_state *reg1, struct bpf_reg_state *reg2,
16200 				u8 opcode, bool is_jmp32);
16201 static u8 rev_opcode(u8 opcode);
16202 
16203 /*
16204  * Learn more information about live branches by simulating refinement on both branches.
16205  * regs_refine_cond_op() is sound, so producing ill-formed register bounds for the branch means
16206  * that branch is dead.
16207  */
16208 static int simulate_both_branches_taken(struct bpf_verifier_env *env, u8 opcode, bool is_jmp32)
16209 {
16210 	/* Fallthrough (FALSE) branch */
16211 	regs_refine_cond_op(&env->false_reg1, &env->false_reg2, rev_opcode(opcode), is_jmp32);
16212 	reg_bounds_sync(&env->false_reg1);
16213 	reg_bounds_sync(&env->false_reg2);
16214 	/*
16215 	 * If there is a range bounds violation in *any* of the abstract values in either
16216 	 * reg_states in the FALSE branch (i.e. reg1, reg2), the FALSE branch must be dead. Only
16217 	 * TRUE branch will be taken.
16218 	 */
16219 	if (range_bounds_violation(&env->false_reg1) || range_bounds_violation(&env->false_reg2))
16220 		return 1;
16221 
16222 	/* Jump (TRUE) branch */
16223 	regs_refine_cond_op(&env->true_reg1, &env->true_reg2, opcode, is_jmp32);
16224 	reg_bounds_sync(&env->true_reg1);
16225 	reg_bounds_sync(&env->true_reg2);
16226 	/*
16227 	 * If there is a range bounds violation in *any* of the abstract values in either
16228 	 * reg_states in the TRUE branch (i.e. true_reg1, true_reg2), the TRUE branch must be dead.
16229 	 * Only FALSE branch will be taken.
16230 	 */
16231 	if (range_bounds_violation(&env->true_reg1) || range_bounds_violation(&env->true_reg2))
16232 		return 0;
16233 
16234 	/* Both branches are possible, we can't determine which one will be taken. */
16235 	return -1;
16236 }
16237 
16238 /*
16239  * <reg1> <op> <reg2>, currently assuming reg2 is a constant
16240  */
16241 static int is_scalar_branch_taken(struct bpf_verifier_env *env, struct bpf_reg_state *reg1,
16242 				  struct bpf_reg_state *reg2, u8 opcode, bool is_jmp32)
16243 {
16244 	struct tnum t1 = is_jmp32 ? tnum_subreg(reg1->var_off) : reg1->var_off;
16245 	struct tnum t2 = is_jmp32 ? tnum_subreg(reg2->var_off) : reg2->var_off;
16246 	u64 umin1 = is_jmp32 ? (u64)reg_u32_min(reg1) : reg_umin(reg1);
16247 	u64 umax1 = is_jmp32 ? (u64)reg_u32_max(reg1) : reg_umax(reg1);
16248 	s64 smin1 = is_jmp32 ? (s64)reg_s32_min(reg1) : reg_smin(reg1);
16249 	s64 smax1 = is_jmp32 ? (s64)reg_s32_max(reg1) : reg_smax(reg1);
16250 	u64 umin2 = is_jmp32 ? (u64)reg_u32_min(reg2) : reg_umin(reg2);
16251 	u64 umax2 = is_jmp32 ? (u64)reg_u32_max(reg2) : reg_umax(reg2);
16252 	s64 smin2 = is_jmp32 ? (s64)reg_s32_min(reg2) : reg_smin(reg2);
16253 	s64 smax2 = is_jmp32 ? (s64)reg_s32_max(reg2) : reg_smax(reg2);
16254 
16255 	if (reg1 == reg2) {
16256 		switch (opcode) {
16257 		case BPF_JGE:
16258 		case BPF_JLE:
16259 		case BPF_JSGE:
16260 		case BPF_JSLE:
16261 		case BPF_JEQ:
16262 			return 1;
16263 		case BPF_JGT:
16264 		case BPF_JLT:
16265 		case BPF_JSGT:
16266 		case BPF_JSLT:
16267 		case BPF_JNE:
16268 			return 0;
16269 		case BPF_JSET:
16270 			if (tnum_is_const(t1))
16271 				return t1.value != 0;
16272 			else
16273 				return (smin1 <= 0 && smax1 >= 0) ? -1 : 1;
16274 		default:
16275 			return -1;
16276 		}
16277 	}
16278 
16279 	switch (opcode) {
16280 	case BPF_JEQ:
16281 		/* constants, umin/umax and smin/smax checks would be
16282 		 * redundant in this case because they all should match
16283 		 */
16284 		if (tnum_is_const(t1) && tnum_is_const(t2))
16285 			return t1.value == t2.value;
16286 		if (!tnum_overlap(t1, t2))
16287 			return 0;
16288 		/* non-overlapping ranges */
16289 		if (umin1 > umax2 || umax1 < umin2)
16290 			return 0;
16291 		if (smin1 > smax2 || smax1 < smin2)
16292 			return 0;
16293 		if (!is_jmp32) {
16294 			/* if 64-bit ranges are inconclusive, see if we can
16295 			 * utilize 32-bit subrange knowledge to eliminate
16296 			 * branches that can't be taken a priori
16297 			 */
16298 			if (reg_u32_min(reg1) > reg_u32_max(reg2) ||
16299 			    reg_u32_max(reg1) < reg_u32_min(reg2))
16300 				return 0;
16301 			if (reg_s32_min(reg1) > reg_s32_max(reg2) ||
16302 			    reg_s32_max(reg1) < reg_s32_min(reg2))
16303 				return 0;
16304 		}
16305 		break;
16306 	case BPF_JNE:
16307 		/* constants, umin/umax and smin/smax checks would be
16308 		 * redundant in this case because they all should match
16309 		 */
16310 		if (tnum_is_const(t1) && tnum_is_const(t2))
16311 			return t1.value != t2.value;
16312 		if (!tnum_overlap(t1, t2))
16313 			return 1;
16314 		/* non-overlapping ranges */
16315 		if (umin1 > umax2 || umax1 < umin2)
16316 			return 1;
16317 		if (smin1 > smax2 || smax1 < smin2)
16318 			return 1;
16319 		if (!is_jmp32) {
16320 			/* if 64-bit ranges are inconclusive, see if we can
16321 			 * utilize 32-bit subrange knowledge to eliminate
16322 			 * branches that can't be taken a priori
16323 			 */
16324 			if (reg_u32_min(reg1) > reg_u32_max(reg2) ||
16325 			    reg_u32_max(reg1) < reg_u32_min(reg2))
16326 				return 1;
16327 			if (reg_s32_min(reg1) > reg_s32_max(reg2) ||
16328 			    reg_s32_max(reg1) < reg_s32_min(reg2))
16329 				return 1;
16330 		}
16331 		break;
16332 	case BPF_JSET:
16333 		if (!is_reg_const(reg2, is_jmp32)) {
16334 			swap(reg1, reg2);
16335 			swap(t1, t2);
16336 		}
16337 		if (!is_reg_const(reg2, is_jmp32))
16338 			return -1;
16339 		if ((~t1.mask & t1.value) & t2.value)
16340 			return 1;
16341 		if (!((t1.mask | t1.value) & t2.value))
16342 			return 0;
16343 		break;
16344 	case BPF_JGT:
16345 		if (umin1 > umax2)
16346 			return 1;
16347 		else if (umax1 <= umin2)
16348 			return 0;
16349 		break;
16350 	case BPF_JSGT:
16351 		if (smin1 > smax2)
16352 			return 1;
16353 		else if (smax1 <= smin2)
16354 			return 0;
16355 		break;
16356 	case BPF_JLT:
16357 		if (umax1 < umin2)
16358 			return 1;
16359 		else if (umin1 >= umax2)
16360 			return 0;
16361 		break;
16362 	case BPF_JSLT:
16363 		if (smax1 < smin2)
16364 			return 1;
16365 		else if (smin1 >= smax2)
16366 			return 0;
16367 		break;
16368 	case BPF_JGE:
16369 		if (umin1 >= umax2)
16370 			return 1;
16371 		else if (umax1 < umin2)
16372 			return 0;
16373 		break;
16374 	case BPF_JSGE:
16375 		if (smin1 >= smax2)
16376 			return 1;
16377 		else if (smax1 < smin2)
16378 			return 0;
16379 		break;
16380 	case BPF_JLE:
16381 		if (umax1 <= umin2)
16382 			return 1;
16383 		else if (umin1 > umax2)
16384 			return 0;
16385 		break;
16386 	case BPF_JSLE:
16387 		if (smax1 <= smin2)
16388 			return 1;
16389 		else if (smin1 > smax2)
16390 			return 0;
16391 		break;
16392 	}
16393 
16394 	return simulate_both_branches_taken(env, opcode, is_jmp32);
16395 }
16396 
16397 static int flip_opcode(u32 opcode)
16398 {
16399 	/* How can we transform "a <op> b" into "b <op> a"? */
16400 	static const u8 opcode_flip[16] = {
16401 		/* these stay the same */
16402 		[BPF_JEQ  >> 4] = BPF_JEQ,
16403 		[BPF_JNE  >> 4] = BPF_JNE,
16404 		[BPF_JSET >> 4] = BPF_JSET,
16405 		/* these swap "lesser" and "greater" (L and G in the opcodes) */
16406 		[BPF_JGE  >> 4] = BPF_JLE,
16407 		[BPF_JGT  >> 4] = BPF_JLT,
16408 		[BPF_JLE  >> 4] = BPF_JGE,
16409 		[BPF_JLT  >> 4] = BPF_JGT,
16410 		[BPF_JSGE >> 4] = BPF_JSLE,
16411 		[BPF_JSGT >> 4] = BPF_JSLT,
16412 		[BPF_JSLE >> 4] = BPF_JSGE,
16413 		[BPF_JSLT >> 4] = BPF_JSGT
16414 	};
16415 	return opcode_flip[opcode >> 4];
16416 }
16417 
16418 static int is_pkt_ptr_branch_taken(struct bpf_reg_state *dst_reg,
16419 				   struct bpf_reg_state *src_reg,
16420 				   u8 opcode)
16421 {
16422 	struct bpf_reg_state *pkt;
16423 
16424 	if (src_reg->type == PTR_TO_PACKET_END) {
16425 		pkt = dst_reg;
16426 	} else if (dst_reg->type == PTR_TO_PACKET_END) {
16427 		pkt = src_reg;
16428 		opcode = flip_opcode(opcode);
16429 	} else {
16430 		return -1;
16431 	}
16432 
16433 	if (pkt->range >= 0)
16434 		return -1;
16435 
16436 	switch (opcode) {
16437 	case BPF_JLE:
16438 		/* pkt <= pkt_end */
16439 		fallthrough;
16440 	case BPF_JGT:
16441 		/* pkt > pkt_end */
16442 		if (pkt->range == BEYOND_PKT_END)
16443 			/* pkt has at last one extra byte beyond pkt_end */
16444 			return opcode == BPF_JGT;
16445 		break;
16446 	case BPF_JLT:
16447 		/* pkt < pkt_end */
16448 		fallthrough;
16449 	case BPF_JGE:
16450 		/* pkt >= pkt_end */
16451 		if (pkt->range == BEYOND_PKT_END || pkt->range == AT_PKT_END)
16452 			return opcode == BPF_JGE;
16453 		break;
16454 	}
16455 	return -1;
16456 }
16457 
16458 /* compute branch direction of the expression "if (<reg1> opcode <reg2>) goto target;"
16459  * and return:
16460  *  1 - branch will be taken and "goto target" will be executed
16461  *  0 - branch will not be taken and fall-through to next insn
16462  * -1 - unknown. Example: "if (reg1 < 5)" is unknown when register value
16463  *      range [0,10]
16464  */
16465 static int is_branch_taken(struct bpf_verifier_env *env, struct bpf_reg_state *reg1,
16466 			   struct bpf_reg_state *reg2, u8 opcode, bool is_jmp32)
16467 {
16468 	if (reg_is_pkt_pointer_any(reg1) && reg_is_pkt_pointer_any(reg2) && !is_jmp32)
16469 		return is_pkt_ptr_branch_taken(reg1, reg2, opcode);
16470 
16471 	if (__is_pointer_value(false, reg1) || __is_pointer_value(false, reg2)) {
16472 		u64 val;
16473 
16474 		/*
16475 		 * The low 32 bits of a valid pointer may well be zero, hence
16476 		 * nothing below applies to a 32-bit comparison.
16477 		 */
16478 		if (is_jmp32)
16479 			return -1;
16480 
16481 		/* arrange that reg2 is a scalar, and reg1 is a pointer */
16482 		if (!is_reg_const(reg2, is_jmp32)) {
16483 			opcode = flip_opcode(opcode);
16484 			swap(reg1, reg2);
16485 		}
16486 		/* and ensure that reg2 is a constant */
16487 		if (!is_reg_const(reg2, is_jmp32))
16488 			return -1;
16489 
16490 		if (!reg_not_null(env, reg1))
16491 			return -1;
16492 
16493 		/* If pointer is valid tests against zero will fail so we can
16494 		 * use this to direct branch taken.
16495 		 */
16496 		val = reg_const_value(reg2, is_jmp32);
16497 		if (val != 0)
16498 			return -1;
16499 
16500 		switch (opcode) {
16501 		case BPF_JEQ:
16502 			return 0;
16503 		case BPF_JNE:
16504 			return 1;
16505 		default:
16506 			return -1;
16507 		}
16508 	}
16509 
16510 	/* now deal with two scalars, but not necessarily constants */
16511 	return is_scalar_branch_taken(env, reg1, reg2, opcode, is_jmp32);
16512 }
16513 
16514 /* Opcode that corresponds to a *false* branch condition.
16515  * E.g., if r1 < r2, then reverse (false) condition is r1 >= r2
16516  */
16517 static u8 rev_opcode(u8 opcode)
16518 {
16519 	switch (opcode) {
16520 	case BPF_JEQ:		return BPF_JNE;
16521 	case BPF_JNE:		return BPF_JEQ;
16522 	/* JSET doesn't have it's reverse opcode in BPF, so add
16523 	 * BPF_X flag to denote the reverse of that operation
16524 	 */
16525 	case BPF_JSET:		return BPF_JSET | BPF_X;
16526 	case BPF_JSET | BPF_X:	return BPF_JSET;
16527 	case BPF_JGE:		return BPF_JLT;
16528 	case BPF_JGT:		return BPF_JLE;
16529 	case BPF_JLE:		return BPF_JGT;
16530 	case BPF_JLT:		return BPF_JGE;
16531 	case BPF_JSGE:		return BPF_JSLT;
16532 	case BPF_JSGT:		return BPF_JSLE;
16533 	case BPF_JSLE:		return BPF_JSGT;
16534 	case BPF_JSLT:		return BPF_JSGE;
16535 	default:		return 0;
16536 	}
16537 }
16538 
16539 /* Refine range knowledge for <reg1> <op> <reg>2 conditional operation. */
16540 static void regs_refine_cond_op(struct bpf_reg_state *reg1, struct bpf_reg_state *reg2,
16541 				u8 opcode, bool is_jmp32)
16542 {
16543 	struct tnum t;
16544 	u64 val;
16545 
16546 	/* In case of GE/GT/SGE/JST, reuse LE/LT/SLE/SLT logic from below */
16547 	switch (opcode) {
16548 	case BPF_JGE:
16549 	case BPF_JGT:
16550 	case BPF_JSGE:
16551 	case BPF_JSGT:
16552 		opcode = flip_opcode(opcode);
16553 		swap(reg1, reg2);
16554 		break;
16555 	default:
16556 		break;
16557 	}
16558 
16559 	switch (opcode) {
16560 	case BPF_JEQ:
16561 		if (is_jmp32) {
16562 			reg1->r32 = cnum32_intersect(reg1->r32, reg2->r32);
16563 			reg2->r32 = reg1->r32;
16564 
16565 			t = tnum_intersect(tnum_subreg(reg1->var_off), tnum_subreg(reg2->var_off));
16566 			reg1->var_off = tnum_with_subreg(reg1->var_off, t);
16567 			reg2->var_off = tnum_with_subreg(reg2->var_off, t);
16568 		} else {
16569 			reg1->r64 = cnum64_intersect(reg1->r64, reg2->r64);
16570 			reg2->r64 = reg1->r64;
16571 
16572 			reg1->var_off = tnum_intersect(reg1->var_off, reg2->var_off);
16573 			reg2->var_off = reg1->var_off;
16574 		}
16575 		break;
16576 	case BPF_JNE:
16577 		if (!is_reg_const(reg2, is_jmp32))
16578 			swap(reg1, reg2);
16579 		if (!is_reg_const(reg2, is_jmp32))
16580 			break;
16581 
16582 		/* try to recompute the bound of reg1 if reg2 is a const and
16583 		 * is exactly the edge of reg1.
16584 		 */
16585 		val = reg_const_value(reg2, is_jmp32);
16586 		if (is_jmp32) {
16587 			/* Complement of the range [val, val] as cnum32. */
16588 			cnum32_intersect_with(&reg1->r32, (struct cnum32){ val + 1, U32_MAX - 1 });
16589 		} else {
16590 			/* Complement of the range [val, val] as cnum64. */
16591 			cnum64_intersect_with(&reg1->r64, (struct cnum64){ val + 1, U64_MAX - 1 });
16592 		}
16593 		break;
16594 	case BPF_JSET:
16595 		if (!is_reg_const(reg2, is_jmp32))
16596 			swap(reg1, reg2);
16597 		if (!is_reg_const(reg2, is_jmp32))
16598 			break;
16599 		val = reg_const_value(reg2, is_jmp32);
16600 		/* BPF_JSET (i.e., TRUE branch, *not* BPF_JSET | BPF_X)
16601 		 * requires single bit to learn something useful. E.g., if we
16602 		 * know that `r1 & 0x3` is true, then which bits (0, 1, or both)
16603 		 * are actually set? We can learn something definite only if
16604 		 * it's a single-bit value to begin with.
16605 		 *
16606 		 * BPF_JSET | BPF_X (i.e., negation of BPF_JSET) doesn't have
16607 		 * this restriction. I.e., !(r1 & 0x3) means neither bit 0 nor
16608 		 * bit 1 is set, which we can readily use in adjustments.
16609 		 */
16610 		if (!is_power_of_2(val))
16611 			break;
16612 		if (is_jmp32) {
16613 			t = tnum_or(tnum_subreg(reg1->var_off), tnum_const(val));
16614 			reg1->var_off = tnum_with_subreg(reg1->var_off, t);
16615 		} else {
16616 			reg1->var_off = tnum_or(reg1->var_off, tnum_const(val));
16617 		}
16618 		break;
16619 	case BPF_JSET | BPF_X: /* reverse of BPF_JSET, see rev_opcode() */
16620 		if (!is_reg_const(reg2, is_jmp32))
16621 			swap(reg1, reg2);
16622 		if (!is_reg_const(reg2, is_jmp32))
16623 			break;
16624 		val = reg_const_value(reg2, is_jmp32);
16625 		/* Forget the ranges before narrowing tnums, to avoid invariant
16626 		 * violations if we're on a dead branch.
16627 		 */
16628 		__mark_reg_unbounded(reg1);
16629 		if (is_jmp32) {
16630 			t = tnum_and(tnum_subreg(reg1->var_off), tnum_const(~val));
16631 			reg1->var_off = tnum_with_subreg(reg1->var_off, t);
16632 		} else {
16633 			reg1->var_off = tnum_and(reg1->var_off, tnum_const(~val));
16634 		}
16635 		break;
16636 	case BPF_JLE:
16637 		if (is_jmp32) {
16638 			cnum32_intersect_with_urange(&reg1->r32, 0, reg_u32_max(reg2));
16639 			cnum32_intersect_with_urange(&reg2->r32, reg_u32_min(reg1), U32_MAX);
16640 		} else {
16641 			cnum64_intersect_with_urange(&reg1->r64, 0, reg_umax(reg2));
16642 			cnum64_intersect_with_urange(&reg2->r64, reg_umin(reg1), U64_MAX);
16643 		}
16644 		break;
16645 	case BPF_JLT:
16646 		if (is_jmp32) {
16647 			cnum32_intersect_with_urange(&reg1->r32, 0, reg_u32_max(reg2) - 1);
16648 			cnum32_intersect_with_urange(&reg2->r32, reg_u32_min(reg1) + 1, U32_MAX);
16649 		} else {
16650 			cnum64_intersect_with_urange(&reg1->r64, 0, reg_umax(reg2) - 1);
16651 			cnum64_intersect_with_urange(&reg2->r64, reg_umin(reg1) + 1, U64_MAX);
16652 		}
16653 		break;
16654 	case BPF_JSLE:
16655 		if (is_jmp32) {
16656 			cnum32_intersect_with_srange(&reg1->r32, S32_MIN, reg_s32_max(reg2));
16657 			cnum32_intersect_with_srange(&reg2->r32, reg_s32_min(reg1), S32_MAX);
16658 		} else {
16659 			cnum64_intersect_with_srange(&reg1->r64, S64_MIN, reg_smax(reg2));
16660 			cnum64_intersect_with_srange(&reg2->r64, reg_smin(reg1), S64_MAX);
16661 		}
16662 		break;
16663 	case BPF_JSLT:
16664 		if (is_jmp32) {
16665 			cnum32_intersect_with_srange(&reg1->r32, S32_MIN, reg_s32_max(reg2) - 1);
16666 			cnum32_intersect_with_srange(&reg2->r32, reg_s32_min(reg1) + 1, S32_MAX);
16667 		} else {
16668 			cnum64_intersect_with_srange(&reg1->r64, S64_MIN, reg_smax(reg2) - 1);
16669 			cnum64_intersect_with_srange(&reg2->r64, reg_smin(reg1) + 1, S64_MAX);
16670 		}
16671 		break;
16672 	default:
16673 		return;
16674 	}
16675 }
16676 
16677 /* Check for invariant violations on the registers for both branches of a condition */
16678 static int regs_bounds_sanity_check_branches(struct bpf_verifier_env *env)
16679 {
16680 	int err;
16681 
16682 	err = reg_bounds_sanity_check(env, &env->true_reg1, "true_reg1");
16683 	err = err ?: reg_bounds_sanity_check(env, &env->true_reg2, "true_reg2");
16684 	err = err ?: reg_bounds_sanity_check(env, &env->false_reg1, "false_reg1");
16685 	err = err ?: reg_bounds_sanity_check(env, &env->false_reg2, "false_reg2");
16686 	return err;
16687 }
16688 
16689 static void mark_ptr_or_null_reg(struct bpf_func_state *state,
16690 				 struct bpf_reg_state *reg, u32 id,
16691 				 bool is_null)
16692 {
16693 	if (type_may_be_null(reg->type) && reg->id == id &&
16694 	    (is_rcu_reg(reg) || !WARN_ON_ONCE(!reg->id))) {
16695 		/* Old offset should have been known-zero, because we don't
16696 		 * allow pointer arithmetic on pointers that might be NULL.
16697 		 * If we see this happening, don't convert the register.
16698 		 *
16699 		 * But in some cases, some helpers that return local kptrs
16700 		 * advance offset for the returned pointer. In those cases,
16701 		 * it is fine to expect to see reg->var_off.
16702 		 */
16703 		if (!(type_is_ptr_alloc_obj(reg->type) || type_is_non_owning_ref(reg->type)) &&
16704 		    WARN_ON_ONCE(!tnum_equals_const(reg->var_off, 0)))
16705 			return;
16706 		if (is_null) {
16707 			/* We don't need id from this point
16708 			 * onwards anymore, thus we should better reset it,
16709 			 * so that state pruning has chances to take effect.
16710 			 */
16711 			__mark_reg_known_zero(reg);
16712 			reg->type = SCALAR_VALUE;
16713 
16714 			return;
16715 		}
16716 
16717 		mark_ptr_not_null_reg(reg);
16718 
16719 		/*
16720 		 * reg->id is preserved for object relationship tracking
16721 		 * and spin_lock lock state tracking
16722 		 */
16723 	}
16724 }
16725 
16726 /* The logic is similar to find_good_pkt_pointers(), both could eventually
16727  * be folded together at some point.
16728  */
16729 static void mark_ptr_or_null_regs(struct bpf_verifier_state *vstate, u32 regno,
16730 				  bool is_null)
16731 {
16732 	struct bpf_func_state *state = vstate->frame[vstate->curframe];
16733 	struct bpf_reg_state *regs = state->regs, *reg;
16734 	u32 id = regs[regno].id;
16735 
16736 	if (is_null && find_reference_state(vstate, id))
16737 		/* regs[regno] is in the " == NULL" branch.
16738 		 * No one could have freed the reference state before
16739 		 * doing the NULL check.
16740 		 */
16741 		WARN_ON_ONCE(__release_reference_nomark(vstate, id));
16742 
16743 	bpf_for_each_reg_in_vstate(vstate, state, reg, ({
16744 		mark_ptr_or_null_reg(state, reg, id, is_null);
16745 	}));
16746 }
16747 
16748 static bool try_match_pkt_pointers(const struct bpf_insn *insn,
16749 				   struct bpf_reg_state *dst_reg,
16750 				   struct bpf_reg_state *src_reg,
16751 				   struct bpf_verifier_state *this_branch,
16752 				   struct bpf_verifier_state *other_branch)
16753 {
16754 	if (BPF_SRC(insn->code) != BPF_X)
16755 		return false;
16756 
16757 	/* Pointers are always 64-bit. */
16758 	if (BPF_CLASS(insn->code) == BPF_JMP32)
16759 		return false;
16760 
16761 	switch (BPF_OP(insn->code)) {
16762 	case BPF_JGT:
16763 		if ((dst_reg->type == PTR_TO_PACKET &&
16764 		     src_reg->type == PTR_TO_PACKET_END) ||
16765 		    (dst_reg->type == PTR_TO_PACKET_META &&
16766 		     reg_is_init_pkt_pointer(src_reg, PTR_TO_PACKET))) {
16767 			/* pkt_data' > pkt_end, pkt_meta' > pkt_data */
16768 			find_good_pkt_pointers(this_branch, dst_reg,
16769 					       dst_reg->type, false);
16770 			mark_pkt_end(other_branch, insn->dst_reg, true);
16771 		} else if ((dst_reg->type == PTR_TO_PACKET_END &&
16772 			    src_reg->type == PTR_TO_PACKET) ||
16773 			   (reg_is_init_pkt_pointer(dst_reg, PTR_TO_PACKET) &&
16774 			    src_reg->type == PTR_TO_PACKET_META)) {
16775 			/* pkt_end > pkt_data', pkt_data > pkt_meta' */
16776 			find_good_pkt_pointers(other_branch, src_reg,
16777 					       src_reg->type, true);
16778 			mark_pkt_end(this_branch, insn->src_reg, false);
16779 		} else {
16780 			return false;
16781 		}
16782 		break;
16783 	case BPF_JLT:
16784 		if ((dst_reg->type == PTR_TO_PACKET &&
16785 		     src_reg->type == PTR_TO_PACKET_END) ||
16786 		    (dst_reg->type == PTR_TO_PACKET_META &&
16787 		     reg_is_init_pkt_pointer(src_reg, PTR_TO_PACKET))) {
16788 			/* pkt_data' < pkt_end, pkt_meta' < pkt_data */
16789 			find_good_pkt_pointers(other_branch, dst_reg,
16790 					       dst_reg->type, true);
16791 			mark_pkt_end(this_branch, insn->dst_reg, false);
16792 		} else if ((dst_reg->type == PTR_TO_PACKET_END &&
16793 			    src_reg->type == PTR_TO_PACKET) ||
16794 			   (reg_is_init_pkt_pointer(dst_reg, PTR_TO_PACKET) &&
16795 			    src_reg->type == PTR_TO_PACKET_META)) {
16796 			/* pkt_end < pkt_data', pkt_data > pkt_meta' */
16797 			find_good_pkt_pointers(this_branch, src_reg,
16798 					       src_reg->type, false);
16799 			mark_pkt_end(other_branch, insn->src_reg, true);
16800 		} else {
16801 			return false;
16802 		}
16803 		break;
16804 	case BPF_JGE:
16805 		if ((dst_reg->type == PTR_TO_PACKET &&
16806 		     src_reg->type == PTR_TO_PACKET_END) ||
16807 		    (dst_reg->type == PTR_TO_PACKET_META &&
16808 		     reg_is_init_pkt_pointer(src_reg, PTR_TO_PACKET))) {
16809 			/* pkt_data' >= pkt_end, pkt_meta' >= pkt_data */
16810 			find_good_pkt_pointers(this_branch, dst_reg,
16811 					       dst_reg->type, true);
16812 			mark_pkt_end(other_branch, insn->dst_reg, false);
16813 		} else if ((dst_reg->type == PTR_TO_PACKET_END &&
16814 			    src_reg->type == PTR_TO_PACKET) ||
16815 			   (reg_is_init_pkt_pointer(dst_reg, PTR_TO_PACKET) &&
16816 			    src_reg->type == PTR_TO_PACKET_META)) {
16817 			/* pkt_end >= pkt_data', pkt_data >= pkt_meta' */
16818 			find_good_pkt_pointers(other_branch, src_reg,
16819 					       src_reg->type, false);
16820 			mark_pkt_end(this_branch, insn->src_reg, true);
16821 		} else {
16822 			return false;
16823 		}
16824 		break;
16825 	case BPF_JLE:
16826 		if ((dst_reg->type == PTR_TO_PACKET &&
16827 		     src_reg->type == PTR_TO_PACKET_END) ||
16828 		    (dst_reg->type == PTR_TO_PACKET_META &&
16829 		     reg_is_init_pkt_pointer(src_reg, PTR_TO_PACKET))) {
16830 			/* pkt_data' <= pkt_end, pkt_meta' <= pkt_data */
16831 			find_good_pkt_pointers(other_branch, dst_reg,
16832 					       dst_reg->type, false);
16833 			mark_pkt_end(this_branch, insn->dst_reg, true);
16834 		} else if ((dst_reg->type == PTR_TO_PACKET_END &&
16835 			    src_reg->type == PTR_TO_PACKET) ||
16836 			   (reg_is_init_pkt_pointer(dst_reg, PTR_TO_PACKET) &&
16837 			    src_reg->type == PTR_TO_PACKET_META)) {
16838 			/* pkt_end <= pkt_data', pkt_data <= pkt_meta' */
16839 			find_good_pkt_pointers(this_branch, src_reg,
16840 					       src_reg->type, true);
16841 			mark_pkt_end(other_branch, insn->src_reg, false);
16842 		} else {
16843 			return false;
16844 		}
16845 		break;
16846 	default:
16847 		return false;
16848 	}
16849 
16850 	return true;
16851 }
16852 
16853 static void __collect_linked_regs(struct linked_regs *reg_set, struct bpf_reg_state *reg,
16854 				  u32 id, u32 frameno, u32 spi_or_reg, bool is_reg)
16855 {
16856 	struct linked_reg *e;
16857 
16858 	if (reg->type != SCALAR_VALUE || (reg->id & ~BPF_ADD_CONST) != id)
16859 		return;
16860 
16861 	e = linked_regs_push(reg_set);
16862 	if (e) {
16863 		e->frameno = frameno;
16864 		e->is_reg = is_reg;
16865 		e->regno = spi_or_reg;
16866 	} else {
16867 		clear_scalar_id(reg);
16868 	}
16869 }
16870 
16871 /* For all R being scalar registers or spilled scalar registers
16872  * in verifier state, save R in linked_regs if R->id == id.
16873  * If there are too many Rs sharing same id, reset id for leftover Rs.
16874  */
16875 static void collect_linked_regs(struct bpf_verifier_env *env,
16876 				struct bpf_verifier_state *vstate,
16877 				u32 id,
16878 				struct linked_regs *linked_regs)
16879 {
16880 	struct bpf_insn_aux_data *aux = env->insn_aux_data;
16881 	struct bpf_func_state *func;
16882 	struct bpf_reg_state *reg;
16883 	u16 live_regs;
16884 	int i, j;
16885 
16886 	id = id & ~BPF_ADD_CONST;
16887 	for (i = vstate->curframe; i >= 0; i--) {
16888 		live_regs = aux[bpf_frame_insn_idx(vstate, i)].live_regs_before;
16889 		func = vstate->frame[i];
16890 		for (j = 0; j < BPF_REG_FP; j++) {
16891 			if (!(live_regs & BIT(j)))
16892 				continue;
16893 			reg = &func->regs[j];
16894 			__collect_linked_regs(linked_regs, reg, id, i, j, true);
16895 		}
16896 		for (j = 0; j < func->allocated_stack / BPF_REG_SIZE; j++) {
16897 			if (!bpf_is_spilled_reg(&func->stack[j]))
16898 				continue;
16899 			reg = &func->stack[j].spilled_ptr;
16900 			__collect_linked_regs(linked_regs, reg, id, i, j, false);
16901 		}
16902 	}
16903 }
16904 
16905 /* For all R in linked_regs, copy known_reg range into R
16906  * if R->id == known_reg->id.
16907  */
16908 static void sync_linked_regs(struct bpf_verifier_env *env, struct bpf_verifier_state *vstate,
16909 			     struct bpf_reg_state *known_reg, struct linked_regs *linked_regs)
16910 {
16911 	struct bpf_reg_state fake_reg;
16912 	struct bpf_reg_state *reg;
16913 	struct linked_reg *e;
16914 	int i;
16915 
16916 	for (i = 0; i < linked_regs->cnt; ++i) {
16917 		e = &linked_regs->entries[i];
16918 		reg = e->is_reg ? &vstate->frame[e->frameno]->regs[e->regno]
16919 				: &vstate->frame[e->frameno]->stack[e->spi].spilled_ptr;
16920 		if (reg->type != SCALAR_VALUE || reg == known_reg)
16921 			continue;
16922 		if ((reg->id & ~BPF_ADD_CONST) != (known_reg->id & ~BPF_ADD_CONST))
16923 			continue;
16924 		/*
16925 		 * Skip mixed 32/64-bit links: the delta relationship doesn't
16926 		 * hold across different ALU widths.
16927 		 */
16928 		if (((reg->id ^ known_reg->id) & BPF_ADD_CONST) == BPF_ADD_CONST)
16929 			continue;
16930 		if ((!(reg->id & BPF_ADD_CONST) && !(known_reg->id & BPF_ADD_CONST)) ||
16931 		    reg->delta == known_reg->delta) {
16932 			*reg = *known_reg;
16933 		} else {
16934 			s32 saved_off = reg->delta;
16935 			u32 saved_id = reg->id;
16936 
16937 			fake_reg.type = SCALAR_VALUE;
16938 			__mark_reg_known(&fake_reg, (s64)reg->delta - (s64)known_reg->delta);
16939 
16940 			/* reg = known_reg; reg += delta */
16941 			*reg = *known_reg;
16942 			/*
16943 			 * Must preserve off and id, otherwise another sync_linked_regs()
16944 			 * will be incorrect.
16945 			 */
16946 			reg->delta = saved_off;
16947 			reg->id = saved_id;
16948 
16949 			scalar32_min_max_add(reg, &fake_reg);
16950 			scalar_min_max_add(reg, &fake_reg);
16951 			reg->var_off = tnum_add(reg->var_off, fake_reg.var_off);
16952 			if ((reg->id | known_reg->id) & BPF_ADD_CONST32)
16953 				zext_32_to_64(reg);
16954 			reg_bounds_sync(reg);
16955 		}
16956 		if (e->is_reg)
16957 			mark_reg_scratched(env, e->regno);
16958 		else
16959 			mark_stack_slot_scratched(env, e->spi);
16960 	}
16961 }
16962 
16963 static int check_cond_jmp_op(struct bpf_verifier_env *env,
16964 			     struct bpf_insn *insn, int *insn_idx)
16965 {
16966 	struct bpf_verifier_state *this_branch = env->cur_state;
16967 	struct bpf_verifier_state *other_branch;
16968 	struct bpf_reg_state *regs = this_branch->frame[this_branch->curframe]->regs;
16969 	struct bpf_reg_state *dst_reg, *other_branch_regs, *src_reg = NULL;
16970 	struct bpf_reg_state *eq_branch_regs;
16971 	struct linked_regs linked_regs = {};
16972 	u8 opcode = BPF_OP(insn->code);
16973 	int insn_flags = 0;
16974 	bool is_jmp32;
16975 	int pred = -1;
16976 	int err;
16977 
16978 	/* Only conditional jumps are expected to reach here. */
16979 	if (opcode == BPF_JA || opcode > BPF_JCOND) {
16980 		verbose(env, "invalid BPF_JMP/JMP32 opcode %x\n", opcode);
16981 		return -EINVAL;
16982 	}
16983 
16984 	if (opcode == BPF_JCOND) {
16985 		struct bpf_verifier_state *cur_st = env->cur_state, *queued_st, *prev_st;
16986 		int idx = *insn_idx;
16987 
16988 		prev_st = find_prev_entry(env, cur_st->parent, idx);
16989 
16990 		/* branch out 'fallthrough' insn as a new state to explore */
16991 		queued_st = push_stack(env, idx + 1, idx, false);
16992 		if (IS_ERR(queued_st))
16993 			return PTR_ERR(queued_st);
16994 
16995 		queued_st->may_goto_depth++;
16996 		if (prev_st)
16997 			widen_imprecise_scalars(env, prev_st, queued_st);
16998 		*insn_idx += insn->off;
16999 		return 0;
17000 	}
17001 
17002 	/* check src2 operand */
17003 	err = check_reg_arg(env, insn->dst_reg, SRC_OP);
17004 	if (err)
17005 		return err;
17006 
17007 	dst_reg = &regs[insn->dst_reg];
17008 	if (BPF_SRC(insn->code) == BPF_X) {
17009 		/* check src1 operand */
17010 		err = check_reg_arg(env, insn->src_reg, SRC_OP);
17011 		if (err)
17012 			return err;
17013 
17014 		src_reg = &regs[insn->src_reg];
17015 		if (!(reg_is_pkt_pointer_any(dst_reg) && reg_is_pkt_pointer_any(src_reg)) &&
17016 		    is_pointer_value(env, insn->src_reg)) {
17017 			verbose(env, "R%d pointer comparison prohibited\n",
17018 				insn->src_reg);
17019 			return -EACCES;
17020 		}
17021 
17022 		if (src_reg->type == PTR_TO_STACK)
17023 			insn_flags |= INSN_F_SRC_REG_STACK;
17024 		if (dst_reg->type == PTR_TO_STACK)
17025 			insn_flags |= INSN_F_DST_REG_STACK;
17026 	} else {
17027 		src_reg = &env->fake_reg[0];
17028 		memset(src_reg, 0, sizeof(*src_reg));
17029 		src_reg->type = SCALAR_VALUE;
17030 		__mark_reg_known(src_reg, insn->imm);
17031 
17032 		if (dst_reg->type == PTR_TO_STACK)
17033 			insn_flags |= INSN_F_DST_REG_STACK;
17034 	}
17035 
17036 	if (insn_flags) {
17037 		err = bpf_push_jmp_history(env, this_branch, insn_flags, 0, 0, 0);
17038 		if (err)
17039 			return err;
17040 	}
17041 
17042 	/*
17043 	 * Collect the linked registers before env->{true,false}_reg{1,2} setup,
17044 	 * otherwise ids dropped by collect_linked_regs() would be resurrected
17045 	 * when env->{true,false}_reg{1,2} are copied back.
17046 	 */
17047 	if (BPF_SRC(insn->code) == BPF_X && src_reg->type == SCALAR_VALUE && src_reg->id)
17048 		collect_linked_regs(env, this_branch, src_reg->id, &linked_regs);
17049 	if (dst_reg->type == SCALAR_VALUE && dst_reg->id)
17050 		collect_linked_regs(env, this_branch, dst_reg->id, &linked_regs);
17051 
17052 	is_jmp32 = BPF_CLASS(insn->code) == BPF_JMP32;
17053 	env->false_reg1 = *dst_reg;
17054 	env->false_reg2 = *src_reg;
17055 	env->true_reg1 = *dst_reg;
17056 	env->true_reg2 = *src_reg;
17057 	pred = is_branch_taken(env, dst_reg, src_reg, opcode, is_jmp32);
17058 	if (pred >= 0) {
17059 		/* If we get here with a dst_reg pointer type it is because
17060 		 * above is_branch_taken() special cased the 0 comparison.
17061 		 */
17062 		if (!__is_pointer_value(false, dst_reg))
17063 			err = mark_chain_precision(env, insn->dst_reg);
17064 		if (BPF_SRC(insn->code) == BPF_X && !err &&
17065 		    !__is_pointer_value(false, src_reg))
17066 			err = mark_chain_precision(env, insn->src_reg);
17067 		if (err)
17068 			return err;
17069 	}
17070 
17071 	if (pred == 1) {
17072 		/* Only follow the goto, ignore fall-through. If needed, push
17073 		 * the fall-through branch for simulation under speculative
17074 		 * execution.
17075 		 */
17076 		if (!env->bypass_spec_v1) {
17077 			err = sanitize_speculative_path(env, insn, *insn_idx + 1, *insn_idx);
17078 			if (err < 0)
17079 				return err;
17080 		}
17081 		if (env->log.level & BPF_LOG_LEVEL)
17082 			print_insn_state(env, this_branch, this_branch->curframe);
17083 		*insn_idx += insn->off;
17084 		return 0;
17085 	} else if (pred == 0) {
17086 		/* Only follow the fall-through branch, since that's where the
17087 		 * program will go. If needed, push the goto branch for
17088 		 * simulation under speculative execution.
17089 		 */
17090 		if (!env->bypass_spec_v1) {
17091 			err = sanitize_speculative_path(env, insn, *insn_idx + insn->off + 1,
17092 							*insn_idx);
17093 			if (err < 0)
17094 				return err;
17095 		}
17096 		if (env->log.level & BPF_LOG_LEVEL)
17097 			print_insn_state(env, this_branch, this_branch->curframe);
17098 		return 0;
17099 	}
17100 
17101 	/* Push scalar registers sharing same ID to jump history,
17102 	 * do this before creating 'other_branch', so that both
17103 	 * 'this_branch' and 'other_branch' share this history
17104 	 * if parent state is created.
17105 	 */
17106 	if (linked_regs.cnt > 1) {
17107 		err = bpf_push_jmp_history(env, this_branch, 0, 0, 0, linked_regs_pack(&linked_regs));
17108 		if (err)
17109 			return err;
17110 	}
17111 
17112 	other_branch = push_stack(env, *insn_idx + insn->off + 1, *insn_idx, false);
17113 	if (IS_ERR(other_branch))
17114 		return PTR_ERR(other_branch);
17115 	other_branch_regs = other_branch->frame[other_branch->curframe]->regs;
17116 
17117 	err = regs_bounds_sanity_check_branches(env);
17118 	if (err)
17119 		return err;
17120 
17121 	*dst_reg = env->false_reg1;
17122 	*src_reg = env->false_reg2;
17123 	other_branch_regs[insn->dst_reg] = env->true_reg1;
17124 	if (BPF_SRC(insn->code) == BPF_X)
17125 		other_branch_regs[insn->src_reg] = env->true_reg2;
17126 
17127 	if (BPF_SRC(insn->code) == BPF_X &&
17128 	    src_reg->type == SCALAR_VALUE && src_reg->id &&
17129 	    !WARN_ON_ONCE(src_reg->id != other_branch_regs[insn->src_reg].id)) {
17130 		sync_linked_regs(env, this_branch, src_reg, &linked_regs);
17131 		sync_linked_regs(env, other_branch, &other_branch_regs[insn->src_reg],
17132 				 &linked_regs);
17133 	}
17134 	if (dst_reg->type == SCALAR_VALUE && dst_reg->id &&
17135 	    !WARN_ON_ONCE(dst_reg->id != other_branch_regs[insn->dst_reg].id)) {
17136 		sync_linked_regs(env, this_branch, dst_reg, &linked_regs);
17137 		sync_linked_regs(env, other_branch, &other_branch_regs[insn->dst_reg],
17138 				 &linked_regs);
17139 	}
17140 
17141 	/* if one pointer register is compared to another pointer
17142 	 * register check if PTR_MAYBE_NULL could be lifted.
17143 	 * E.g. register A - maybe null
17144 	 *      register B - not null
17145 	 * for JNE A, B, ... - A is not null in the false branch;
17146 	 * for JEQ A, B, ... - A is not null in the true branch.
17147 	 *
17148 	 * Since PTR_TO_BTF_ID points to a kernel struct that does
17149 	 * not need to be null checked by the BPF program, i.e.,
17150 	 * could be null even without PTR_MAYBE_NULL marking, so
17151 	 * only propagate nullness when neither reg is that type.
17152 	 */
17153 	if (!is_jmp32 && BPF_SRC(insn->code) == BPF_X &&
17154 	    __is_pointer_value(false, src_reg) && __is_pointer_value(false, dst_reg) &&
17155 	    base_type(src_reg->type) != PTR_TO_BTF_ID &&
17156 	    base_type(dst_reg->type) != PTR_TO_BTF_ID) {
17157 		eq_branch_regs = NULL;
17158 		switch (opcode) {
17159 		case BPF_JEQ:
17160 			eq_branch_regs = other_branch_regs;
17161 			break;
17162 		case BPF_JNE:
17163 			eq_branch_regs = regs;
17164 			break;
17165 		default:
17166 			/* do nothing */
17167 			break;
17168 		}
17169 		if (eq_branch_regs) {
17170 			/* src == dst && dst != NULL => src != NULL */
17171 			if (reg_not_null(env, dst_reg) && type_may_be_null(src_reg->type))
17172 				mark_ptr_not_null_reg(&eq_branch_regs[insn->src_reg]);
17173 			/* src == dst && src != NULL => dst != NULL */
17174 			if (reg_not_null(env, src_reg) && type_may_be_null(dst_reg->type))
17175 				mark_ptr_not_null_reg(&eq_branch_regs[insn->dst_reg]);
17176 		}
17177 	}
17178 
17179 	/* detect if R == 0 where R is returned from bpf_map_lookup_elem().
17180 	 * Also does the same detection for a register whose the value is
17181 	 * known to be 0.
17182 	 * NOTE: these optimizations below are related with pointer comparison
17183 	 *       which will never be JMP32.
17184 	 */
17185 	if (!is_jmp32 && (opcode == BPF_JEQ || opcode == BPF_JNE) &&
17186 	    type_may_be_null(dst_reg->type) &&
17187 	    ((BPF_SRC(insn->code) == BPF_K && insn->imm == 0) ||
17188 	     (BPF_SRC(insn->code) == BPF_X && bpf_register_is_null(src_reg)))) {
17189 		/*
17190 		 * For BPF_X the zero is a property of this execution path,
17191 		 * hence src_reg has to be precise.
17192 		 */
17193 		if (BPF_SRC(insn->code) == BPF_X) {
17194 			err = mark_chain_precision(env, insn->src_reg);
17195 			if (err)
17196 				return err;
17197 		}
17198 		/* Mark all identical registers in each branch as either
17199 		 * safe or unknown depending R == 0 or R != 0 conditional.
17200 		 */
17201 		mark_ptr_or_null_regs(this_branch, insn->dst_reg,
17202 				      opcode == BPF_JNE);
17203 		mark_ptr_or_null_regs(other_branch, insn->dst_reg,
17204 				      opcode == BPF_JEQ);
17205 	} else if (!try_match_pkt_pointers(insn, dst_reg, &regs[insn->src_reg],
17206 					   this_branch, other_branch) &&
17207 		   is_pointer_value(env, insn->dst_reg)) {
17208 		verbose(env, "R%d pointer comparison prohibited\n",
17209 			insn->dst_reg);
17210 		return -EACCES;
17211 	}
17212 	if (env->log.level & BPF_LOG_LEVEL)
17213 		print_insn_state(env, this_branch, this_branch->curframe);
17214 	return 0;
17215 }
17216 
17217 /* verify BPF_LD_IMM64 instruction */
17218 static int check_ld_imm(struct bpf_verifier_env *env, struct bpf_insn *insn)
17219 {
17220 	struct bpf_insn_aux_data *aux = cur_aux(env);
17221 	struct bpf_reg_state *regs = cur_regs(env);
17222 	struct bpf_reg_state *dst_reg;
17223 	struct bpf_map *map;
17224 	int err;
17225 
17226 	if (BPF_SIZE(insn->code) != BPF_DW) {
17227 		verbose(env, "invalid BPF_LD_IMM insn\n");
17228 		return -EINVAL;
17229 	}
17230 
17231 	err = check_reg_arg(env, insn->dst_reg, DST_OP);
17232 	if (err)
17233 		return err;
17234 
17235 	dst_reg = &regs[insn->dst_reg];
17236 	bpf_diag_mod_begin(env, dst_reg, NULL, BPF_DIAG_MOD_WRITE);
17237 	if (insn->src_reg == 0) {
17238 		u64 imm = ((u64)(insn + 1)->imm << 32) | (u32)insn->imm;
17239 
17240 		dst_reg->type = SCALAR_VALUE;
17241 		__mark_reg_known(&regs[insn->dst_reg], imm);
17242 		bpf_diag_mod_end(env);
17243 		return 0;
17244 	}
17245 
17246 	/* All special src_reg cases are listed below. From this point onwards
17247 	 * we either succeed and assign a corresponding dst_reg->type after
17248 	 * zeroing the offset, or fail and reject the program.
17249 	 */
17250 	mark_reg_known_zero(env, regs, insn->dst_reg);
17251 
17252 	if (insn->src_reg == BPF_PSEUDO_BTF_ID) {
17253 		dst_reg->type = aux->btf_var.reg_type;
17254 		switch (base_type(dst_reg->type)) {
17255 		case PTR_TO_MEM:
17256 			dst_reg->mem_size = aux->btf_var.mem_size;
17257 			break;
17258 		case PTR_TO_BTF_ID:
17259 			dst_reg->btf = aux->btf_var.btf;
17260 			dst_reg->btf_id = aux->btf_var.btf_id;
17261 			break;
17262 		default:
17263 			verifier_bug(env, "pseudo btf id: unexpected dst reg type");
17264 			return -EFAULT;
17265 		}
17266 		bpf_diag_mod_end(env);
17267 		return 0;
17268 	}
17269 
17270 	if (insn->src_reg == BPF_PSEUDO_FUNC) {
17271 		struct bpf_prog_aux *aux = env->prog->aux;
17272 		u32 subprogno = bpf_find_subprog(env,
17273 						 env->insn_idx + insn->imm + 1);
17274 
17275 		if (!aux->func_info) {
17276 			verbose(env, "missing btf func_info\n");
17277 			return -EINVAL;
17278 		}
17279 		if (aux->func_info_aux[subprogno].linkage != BTF_FUNC_STATIC) {
17280 			verbose(env, "callback function not static\n");
17281 			return -EINVAL;
17282 		}
17283 		/*
17284 		 * When env->subprog_cnt == 1 this instruction won't be rewritten
17285 		 * to hold a real function address. Assume that no usable program
17286 		 * combines e.g. main and timer callback and just reject here.
17287 		 */
17288 		if (subprogno == 0) {
17289 			verbose(env, "callback function cannot be the main program\n");
17290 			return -EINVAL;
17291 		}
17292 
17293 		dst_reg->type = PTR_TO_FUNC;
17294 		dst_reg->subprogno = subprogno;
17295 		bpf_diag_mod_end(env);
17296 		return 0;
17297 	}
17298 
17299 	map = env->used_maps[aux->map_index];
17300 
17301 	if (insn->src_reg == BPF_PSEUDO_MAP_VALUE ||
17302 	    insn->src_reg == BPF_PSEUDO_MAP_IDX_VALUE) {
17303 		if (map->map_type == BPF_MAP_TYPE_ARENA) {
17304 			__mark_reg_unknown(env, dst_reg);
17305 			dst_reg->map_ptr = map;
17306 			bpf_diag_mod_end(env);
17307 			return 0;
17308 		}
17309 		__mark_reg_known(dst_reg, aux->map_off);
17310 		dst_reg->type = PTR_TO_MAP_VALUE;
17311 		dst_reg->map_ptr = map;
17312 		WARN_ON_ONCE(map->map_type != BPF_MAP_TYPE_INSN_ARRAY &&
17313 			     map->max_entries != 1);
17314 		/* We want reg->id to be same (0) as map_value is not distinct */
17315 	} else if (insn->src_reg == BPF_PSEUDO_MAP_FD ||
17316 		   insn->src_reg == BPF_PSEUDO_MAP_IDX) {
17317 		dst_reg->type = CONST_PTR_TO_MAP;
17318 		dst_reg->map_ptr = map;
17319 	} else {
17320 		verifier_bug(env, "unexpected src reg value for ldimm64");
17321 		return -EFAULT;
17322 	}
17323 
17324 	bpf_diag_mod_end(env);
17325 	return 0;
17326 }
17327 
17328 static bool may_access_skb(enum bpf_prog_type type)
17329 {
17330 	switch (type) {
17331 	case BPF_PROG_TYPE_SOCKET_FILTER:
17332 	case BPF_PROG_TYPE_SCHED_CLS:
17333 	case BPF_PROG_TYPE_SCHED_ACT:
17334 		return true;
17335 	default:
17336 		return false;
17337 	}
17338 }
17339 
17340 /* verify safety of LD_ABS|LD_IND instructions:
17341  * - they can only appear in the programs where ctx == skb
17342  * - since they are wrappers of function calls, they scratch R1-R5 registers,
17343  *   preserve R6-R9, and store return value into R0
17344  *
17345  * Implicit input:
17346  *   ctx == skb == R6 == CTX
17347  *
17348  * Explicit input:
17349  *   SRC == any register
17350  *   IMM == 32-bit immediate
17351  *
17352  * Output:
17353  *   R0 - 8/16/32-bit skb data converted to cpu endianness
17354  */
17355 static int check_ld_abs(struct bpf_verifier_env *env, struct bpf_insn *insn)
17356 {
17357 	struct bpf_verifier_state *state = env->cur_state;
17358 	struct bpf_reg_state *regs = cur_regs(env);
17359 	static const int ctx_reg = BPF_REG_6;
17360 	u8 mode = BPF_MODE(insn->code);
17361 	int i, err;
17362 
17363 	if (!may_access_skb(resolve_prog_type(env->prog))) {
17364 		verbose(env, "BPF_LD_[ABS|IND] instructions not allowed for this program type\n");
17365 		return -EINVAL;
17366 	}
17367 
17368 	for (i = state->curframe; i; i--) {
17369 		if (state->frame[i]->in_callback_fn) {
17370 			verbose(env, "cannot use BPF_LD_[ABS|IND] within callback\n");
17371 			return -EINVAL;
17372 		}
17373 	}
17374 
17375 	if (!env->ops->gen_ld_abs) {
17376 		verifier_bug(env, "gen_ld_abs is null");
17377 		return -EFAULT;
17378 	}
17379 
17380 	/* check whether implicit source operand (register R6) is readable */
17381 	err = check_reg_arg(env, ctx_reg, SRC_OP);
17382 	if (err)
17383 		return err;
17384 
17385 	/* Disallow usage of BPF_LD_[ABS|IND] with reference tracking, as
17386 	 * gen_ld_abs() may terminate the program at runtime, leading to
17387 	 * reference leak.
17388 	 */
17389 	err = check_resource_leak(env, false, true, "BPF_LD_[ABS|IND]");
17390 	if (err)
17391 		return err;
17392 
17393 	if (regs[ctx_reg].type != PTR_TO_CTX) {
17394 		verbose(env,
17395 			"at the time of BPF_LD_ABS|IND R6 != pointer to skb\n");
17396 		return -EINVAL;
17397 	}
17398 
17399 	if (mode == BPF_IND) {
17400 		/* check explicit source operand */
17401 		err = check_reg_arg(env, insn->src_reg, SRC_OP);
17402 		if (err)
17403 			return err;
17404 	}
17405 
17406 	err = check_ptr_off_reg(env, &regs[ctx_reg], ctx_reg);
17407 	if (err < 0)
17408 		return err;
17409 
17410 	/* reset caller saved regs to unreadable */
17411 	bpf_diag_record_caller_saved(env, regs);
17412 	bpf_diag_mod_begin(env, &regs[BPF_REG_0], NULL, BPF_DIAG_MOD_WRITE);
17413 	for (i = 0; i < CALLER_SAVED_REGS; i++) {
17414 		bpf_mark_reg_not_init(env, &regs[caller_saved[i]]);
17415 		check_reg_arg(env, caller_saved[i], DST_OP_NO_MARK);
17416 	}
17417 
17418 	/* mark destination R0 register as readable, since it contains
17419 	 * the value fetched from the packet.
17420 	 * Already marked as written above.
17421 	 */
17422 	mark_reg_unknown(env, regs, BPF_REG_0);
17423 	bpf_diag_mod_end(env);
17424 	/*
17425 	 * See bpf_gen_ld_abs() which emits a hidden BPF_EXIT with r0=0
17426 	 * which must be explored by the verifier when in a subprog.
17427 	 */
17428 	if (env->cur_state->curframe) {
17429 		struct bpf_verifier_state *branch;
17430 
17431 		mark_reg_scratched(env, BPF_REG_0);
17432 		branch = push_stack(env, env->insn_idx + 1, env->insn_idx, false);
17433 		if (IS_ERR(branch))
17434 			return PTR_ERR(branch);
17435 		mark_reg_known_zero(env, regs, BPF_REG_0);
17436 		err = prepare_func_exit(env, &env->insn_idx);
17437 		if (err)
17438 			return err;
17439 		env->insn_idx--;
17440 	}
17441 	return 0;
17442 }
17443 
17444 static bool return_retval_range(struct bpf_verifier_env *env, struct bpf_retval_range *range)
17445 {
17446 	enum bpf_prog_type prog_type = resolve_prog_type(env->prog);
17447 
17448 	/* Default return value range. */
17449 	*range = retval_range(0, 1);
17450 
17451 	switch (prog_type) {
17452 	case BPF_PROG_TYPE_CGROUP_SOCK_ADDR:
17453 		switch (env->prog->expected_attach_type) {
17454 		case BPF_CGROUP_UDP4_RECVMSG:
17455 		case BPF_CGROUP_UDP6_RECVMSG:
17456 		case BPF_CGROUP_UNIX_RECVMSG:
17457 		case BPF_CGROUP_INET4_GETPEERNAME:
17458 		case BPF_CGROUP_INET6_GETPEERNAME:
17459 		case BPF_CGROUP_UNIX_GETPEERNAME:
17460 		case BPF_CGROUP_INET4_GETSOCKNAME:
17461 		case BPF_CGROUP_INET6_GETSOCKNAME:
17462 		case BPF_CGROUP_UNIX_GETSOCKNAME:
17463 			*range = retval_range(1, 1);
17464 			break;
17465 		case BPF_CGROUP_INET4_BIND:
17466 		case BPF_CGROUP_INET6_BIND:
17467 			*range = retval_range(0, 3);
17468 			break;
17469 		default:
17470 			break;
17471 		}
17472 		break;
17473 	case BPF_PROG_TYPE_CGROUP_SKB:
17474 		if (env->prog->expected_attach_type == BPF_CGROUP_INET_EGRESS)
17475 			*range = retval_range(0, 3);
17476 		break;
17477 	case BPF_PROG_TYPE_CGROUP_SOCK:
17478 	case BPF_PROG_TYPE_SOCK_OPS:
17479 	case BPF_PROG_TYPE_CGROUP_DEVICE:
17480 	case BPF_PROG_TYPE_CGROUP_SYSCTL:
17481 	case BPF_PROG_TYPE_CGROUP_SOCKOPT:
17482 		break;
17483 	case BPF_PROG_TYPE_RAW_TRACEPOINT:
17484 		if (!env->prog->aux->attach_btf_id)
17485 			return false;
17486 		*range = retval_range(0, 0);
17487 		break;
17488 	case BPF_PROG_TYPE_TRACING:
17489 		switch (env->prog->expected_attach_type) {
17490 		case BPF_TRACE_FENTRY:
17491 		case BPF_TRACE_FEXIT:
17492 		case BPF_TRACE_FSESSION:
17493 		case BPF_TRACE_FENTRY_MULTI:
17494 		case BPF_TRACE_FEXIT_MULTI:
17495 		case BPF_TRACE_FSESSION_MULTI:
17496 			*range = retval_range(0, 0);
17497 			break;
17498 		case BPF_TRACE_RAW_TP:
17499 		case BPF_MODIFY_RETURN:
17500 			return false;
17501 		case BPF_TRACE_ITER:
17502 		default:
17503 			break;
17504 		}
17505 		break;
17506 	case BPF_PROG_TYPE_KPROBE:
17507 		switch (env->prog->expected_attach_type) {
17508 		case BPF_TRACE_KPROBE_SESSION:
17509 		case BPF_TRACE_UPROBE_SESSION:
17510 			break;
17511 		default:
17512 			return false;
17513 		}
17514 		break;
17515 	case BPF_PROG_TYPE_SK_LOOKUP:
17516 		*range = retval_range(SK_DROP, SK_PASS);
17517 		break;
17518 
17519 	case BPF_PROG_TYPE_LSM:
17520 		if (env->prog->expected_attach_type != BPF_LSM_CGROUP) {
17521 			/* no range found, any return value is allowed */
17522 			if (!get_func_retval_range(env->prog, range))
17523 				return false;
17524 			/* no restricted range, any return value is allowed */
17525 			if (range->minval == S32_MIN && range->maxval == S32_MAX)
17526 				return false;
17527 			range->return_32bit = true;
17528 		} else if (!env->prog->aux->attach_func_proto->type) {
17529 			/* Make sure programs that attach to void
17530 			 * hooks don't try to modify return value.
17531 			 */
17532 			*range = retval_range(1, 1);
17533 		}
17534 		break;
17535 
17536 	case BPF_PROG_TYPE_NETFILTER:
17537 		*range = retval_range(NF_DROP, NF_ACCEPT);
17538 		break;
17539 	case BPF_PROG_TYPE_STRUCT_OPS:
17540 		*range = retval_range(0, 0);
17541 		break;
17542 	case BPF_PROG_TYPE_EXT:
17543 		/* freplace program can return anything as its return value
17544 		 * depends on the to-be-replaced kernel func or bpf program.
17545 		 */
17546 	default:
17547 		return false;
17548 	}
17549 
17550 	/* Continue calculating. */
17551 
17552 	return true;
17553 }
17554 
17555 static bool program_returns_void(struct bpf_verifier_env *env)
17556 {
17557 	const struct bpf_prog *prog = env->prog;
17558 	enum bpf_prog_type prog_type = prog->type;
17559 
17560 	switch (prog_type) {
17561 	case BPF_PROG_TYPE_LSM:
17562 		/* See return_retval_range, for BPF_LSM_CGROUP can be 0 or 0-1 depending on hook. */
17563 		if (prog->expected_attach_type != BPF_LSM_CGROUP &&
17564 		    !prog->aux->attach_func_proto->type)
17565 			return true;
17566 		break;
17567 	case BPF_PROG_TYPE_STRUCT_OPS:
17568 		if (!prog->aux->attach_func_proto->type)
17569 			return true;
17570 		break;
17571 	case BPF_PROG_TYPE_EXT:
17572 		/*
17573 		 * If the actual program is an extension, let it
17574 		 * return void - attaching will succeed only if the
17575 		 * program being replaced also returns void, and since
17576 		 * it has passed verification its actual type doesn't matter.
17577 		 */
17578 		if (subprog_returns_void(env, 0))
17579 			return true;
17580 		break;
17581 	default:
17582 		break;
17583 	}
17584 	return false;
17585 }
17586 
17587 static int check_return_code(struct bpf_verifier_env *env, int regno, const char *reg_name)
17588 {
17589 	const char *exit_ctx = "At program exit";
17590 	struct tnum enforce_attach_type_range = tnum_unknown;
17591 	const struct bpf_prog *prog = env->prog;
17592 	struct bpf_reg_state *reg = reg_state(env, regno);
17593 	struct bpf_retval_range range = retval_range(0, 1);
17594 	enum bpf_prog_type prog_type = resolve_prog_type(env->prog);
17595 	struct bpf_func_state *frame = env->cur_state->frame[0];
17596 	const struct btf_type *reg_type, *ret_type = NULL;
17597 	int err;
17598 
17599 	/* LSM and struct_ops func-ptr's return type could be "void" */
17600 	if (!frame->in_async_callback_fn && program_returns_void(env))
17601 		return 0;
17602 
17603 	if (prog_type == BPF_PROG_TYPE_STRUCT_OPS) {
17604 		/* Allow a struct_ops program to return a referenced kptr if it
17605 		 * matches the operator's return type and is in its unmodified
17606 		 * form. A scalar zero (i.e., a null pointer) is also allowed.
17607 		 */
17608 		reg_type = reg->btf ? btf_type_by_id(reg->btf, reg->btf_id) : NULL;
17609 		ret_type = btf_type_resolve_ptr(prog->aux->attach_btf,
17610 						prog->aux->attach_func_proto->type,
17611 						NULL);
17612 		if (ret_type && ret_type == reg_type && reg_is_referenced(env, reg))
17613 			return __check_ptr_off_reg(env, reg, argno_from_reg(regno), false);
17614 	}
17615 
17616 	/* eBPF calling convention is such that R0 is used
17617 	 * to return the value from eBPF program.
17618 	 * Make sure that it's readable at this time
17619 	 * of bpf_exit, which means that program wrote
17620 	 * something into it earlier
17621 	 */
17622 	err = check_reg_arg(env, regno, SRC_OP);
17623 	if (err)
17624 		return err;
17625 
17626 	if (is_pointer_value(env, regno)) {
17627 		verbose(env, "R%d leaks addr as return value\n", regno);
17628 		return -EACCES;
17629 	}
17630 
17631 	if (frame->in_async_callback_fn) {
17632 		exit_ctx = "At async callback return";
17633 		range = frame->callback_ret_range;
17634 		goto enforce_retval;
17635 	}
17636 
17637 	if (prog_type == BPF_PROG_TYPE_STRUCT_OPS && !ret_type)
17638 		return 0;
17639 
17640 	if (prog_type == BPF_PROG_TYPE_CGROUP_SKB && (env->prog->expected_attach_type == BPF_CGROUP_INET_EGRESS))
17641 		enforce_attach_type_range = tnum_range(2, 3);
17642 
17643 	if (!return_retval_range(env, &range))
17644 		return 0;
17645 
17646 enforce_retval:
17647 	if (reg->type != SCALAR_VALUE) {
17648 		verbose(env, "%s the register R%d is not a known value (%s)\n",
17649 			exit_ctx, regno, reg_type_str(env, reg->type));
17650 		return -EINVAL;
17651 	}
17652 
17653 	err = mark_chain_precision(env, regno);
17654 	if (err)
17655 		return err;
17656 
17657 	if (!retval_range_within(range, reg)) {
17658 		verbose_invalid_scalar(env, reg, range, exit_ctx, reg_name);
17659 		if (prog->expected_attach_type == BPF_LSM_CGROUP &&
17660 		    prog_type == BPF_PROG_TYPE_LSM &&
17661 		    !prog->aux->attach_func_proto->type)
17662 			verbose(env, "Note, BPF_LSM_CGROUP that attach to void LSM hooks can't modify return value!\n");
17663 		return -EINVAL;
17664 	}
17665 
17666 	if (!tnum_is_unknown(enforce_attach_type_range) &&
17667 	    tnum_in(enforce_attach_type_range, reg->var_off))
17668 		env->prog->enforce_expected_attach_type = 1;
17669 	return 0;
17670 }
17671 
17672 static int check_global_subprog_return_code(struct bpf_verifier_env *env)
17673 {
17674 	struct bpf_reg_state *reg = reg_state(env, BPF_REG_0);
17675 	struct bpf_func_state *cur_frame = cur_func(env);
17676 	int err;
17677 
17678 	if (subprog_returns_void(env, cur_frame->subprogno))
17679 		return 0;
17680 
17681 	err = check_reg_arg(env, BPF_REG_0, SRC_OP);
17682 	if (err)
17683 		return err;
17684 
17685 	/* Pointers to arena are safe to pass between subprograms. */
17686 	if (is_arena_reg(env, BPF_REG_0))
17687 		return 0;
17688 
17689 	if (is_pointer_value(env, BPF_REG_0)) {
17690 		verbose(env, "R%d leaks addr as return value\n", BPF_REG_0);
17691 		return -EACCES;
17692 	}
17693 
17694 	if (reg->type != SCALAR_VALUE) {
17695 		verbose(env, "At subprogram exit the register R0 is not a scalar value (%s)\n",
17696 			reg_type_str(env, reg->type));
17697 		return -EINVAL;
17698 	}
17699 
17700 	return 0;
17701 }
17702 
17703 /* Bitmask with 1s for all caller saved registers */
17704 #define ALL_CALLER_SAVED_REGS ((1u << CALLER_SAVED_REGS) - 1)
17705 
17706 /* True if do_misc_fixups() replaces calls to helper number 'imm',
17707  * replacement patch is presumed to follow bpf_fastcall contract
17708  * (see mark_fastcall_pattern_for_call() below).
17709  */
17710 bool bpf_verifier_inlines_helper_call(struct bpf_verifier_env *env, s32 imm)
17711 {
17712 	switch (imm) {
17713 #ifdef CONFIG_X86_64
17714 	case BPF_FUNC_get_smp_processor_id:
17715 #ifdef CONFIG_SMP
17716 	case BPF_FUNC_get_current_task_btf:
17717 	case BPF_FUNC_get_current_task:
17718 #endif
17719 		return env->prog->jit_requested && bpf_jit_supports_percpu_insn();
17720 #endif
17721 	default:
17722 		return false;
17723 	}
17724 }
17725 
17726 /* If @call is a kfunc or helper call, fills @cs and returns true,
17727  * otherwise returns false.
17728  */
17729 bool bpf_get_call_summary(struct bpf_verifier_env *env, struct bpf_insn *call,
17730 			  struct bpf_call_summary *cs)
17731 {
17732 	struct bpf_call_arg_meta meta;
17733 	const struct bpf_func_proto *fn;
17734 	int i;
17735 
17736 	if (bpf_helper_call(call)) {
17737 		if (bpf_get_helper_proto(env, call->imm, &fn) < 0)
17738 			/* error would be reported later */
17739 			return false;
17740 		cs->fastcall = fn->allow_fastcall &&
17741 			       (bpf_verifier_inlines_helper_call(env, call->imm) ||
17742 				bpf_jit_inlines_helper_call(call->imm));
17743 		cs->is_void = fn->ret_type == RET_VOID;
17744 		cs->num_params = 0;
17745 		for (i = 0; i < ARRAY_SIZE(fn->arg_type); ++i) {
17746 			if (fn->arg_type[i] == ARG_DONTCARE)
17747 				break;
17748 			cs->num_params++;
17749 		}
17750 		return true;
17751 	}
17752 
17753 	if (bpf_pseudo_kfunc_call(call)) {
17754 		int err;
17755 
17756 		err = bpf_fetch_kfunc_arg_meta(env, call->imm, call->off, &meta);
17757 		if (err < 0)
17758 			/* error would be reported later */
17759 			return false;
17760 		cs->num_params = btf_type_vlen(meta.func_proto);
17761 		cs->fastcall = meta.kfunc_flags & KF_FASTCALL;
17762 		cs->is_void = btf_type_is_void(btf_type_by_id(meta.btf, meta.func_proto->type));
17763 		return true;
17764 	}
17765 
17766 	return false;
17767 }
17768 
17769 /* LLVM define a bpf_fastcall function attribute.
17770  * This attribute means that function scratches only some of
17771  * the caller saved registers defined by ABI.
17772  * For BPF the set of such registers could be defined as follows:
17773  * - R0 is scratched only if function is non-void;
17774  * - R1-R5 are scratched only if corresponding parameter type is defined
17775  *   in the function prototype.
17776  *
17777  * The contract between kernel and clang allows to simultaneously use
17778  * such functions and maintain backwards compatibility with old
17779  * kernels that don't understand bpf_fastcall calls:
17780  *
17781  * - for bpf_fastcall calls clang allocates registers as-if relevant r0-r5
17782  *   registers are not scratched by the call;
17783  *
17784  * - as a post-processing step, clang visits each bpf_fastcall call and adds
17785  *   spill/fill for every live r0-r5;
17786  *
17787  * - stack offsets used for the spill/fill are allocated as lowest
17788  *   stack offsets in whole function and are not used for any other
17789  *   purposes;
17790  *
17791  * - when kernel loads a program, it looks for such patterns
17792  *   (bpf_fastcall function surrounded by spills/fills) and checks if
17793  *   spill/fill stack offsets are used exclusively in fastcall patterns;
17794  *
17795  * - if so, and if verifier or current JIT inlines the call to the
17796  *   bpf_fastcall function (e.g. a helper call), kernel removes unnecessary
17797  *   spill/fill pairs;
17798  *
17799  * - when old kernel loads a program, presence of spill/fill pairs
17800  *   keeps BPF program valid, albeit slightly less efficient.
17801  *
17802  * For example:
17803  *
17804  *   r1 = 1;
17805  *   r2 = 2;
17806  *   *(u64 *)(r10 - 8)  = r1;            r1 = 1;
17807  *   *(u64 *)(r10 - 16) = r2;            r2 = 2;
17808  *   call %[to_be_inlined]         -->   call %[to_be_inlined]
17809  *   r2 = *(u64 *)(r10 - 16);            r0 = r1;
17810  *   r1 = *(u64 *)(r10 - 8);             r0 += r2;
17811  *   r0 = r1;                            exit;
17812  *   r0 += r2;
17813  *   exit;
17814  *
17815  * The purpose of mark_fastcall_pattern_for_call is to:
17816  * - look for such patterns;
17817  * - mark spill and fill instructions in env->insn_aux_data[*].fastcall_pattern;
17818  * - mark set env->insn_aux_data[*].fastcall_spills_num for call instruction;
17819  * - update env->subprog_info[*]->fastcall_stack_off to find an offset
17820  *   at which bpf_fastcall spill/fill stack slots start;
17821  * - update env->subprog_info[*]->keep_fastcall_stack.
17822  *
17823  * The .fastcall_pattern and .fastcall_stack_off are used by
17824  * check_fastcall_stack_contract() to check if every stack access to
17825  * fastcall spill/fill stack slot originates from spill/fill
17826  * instructions, members of fastcall patterns.
17827  *
17828  * If such condition holds true for a subprogram, fastcall patterns could
17829  * be rewritten by remove_fastcall_spills_fills().
17830  * Otherwise bpf_fastcall patterns are not changed in the subprogram
17831  * (code, presumably, generated by an older clang version).
17832  *
17833  * For example, it is *not* safe to remove spill/fill below:
17834  *
17835  *   r1 = 1;
17836  *   *(u64 *)(r10 - 8)  = r1;            r1 = 1;
17837  *   call %[to_be_inlined]         -->   call %[to_be_inlined]
17838  *   r1 = *(u64 *)(r10 - 8);             r0 = *(u64 *)(r10 - 8);  <---- wrong !!!
17839  *   r0 = *(u64 *)(r10 - 8);             r0 += r1;
17840  *   r0 += r1;                           exit;
17841  *   exit;
17842  *
17843  * Both uses of the marks assume that a pattern is entered at its first
17844  * spill and thus executes as a unit, hence a pattern is not grown past
17845  * an instruction targeted by a jump.
17846  */
17847 static void mark_fastcall_pattern_for_call(struct bpf_verifier_env *env,
17848 					   struct bpf_subprog_info *subprog,
17849 					   int insn_idx, s16 lowest_off)
17850 {
17851 	struct bpf_insn *insns = env->prog->insnsi, *stx, *ldx;
17852 	struct bpf_insn *call = &env->prog->insnsi[insn_idx];
17853 	u32 clobbered_regs_mask;
17854 	struct bpf_call_summary cs;
17855 	u32 expected_regs_mask;
17856 	s16 off;
17857 	int i;
17858 
17859 	if (!bpf_get_call_summary(env, call, &cs))
17860 		return;
17861 
17862 	/* A bitmask specifying which caller saved registers are clobbered
17863 	 * by a call to a helper/kfunc *as if* this helper/kfunc follows
17864 	 * bpf_fastcall contract:
17865 	 * - includes R0 if function is non-void;
17866 	 * - includes R1-R5 if corresponding parameter has is described
17867 	 *   in the function prototype.
17868 	 */
17869 	clobbered_regs_mask = GENMASK(cs.num_params, cs.is_void ? 1 : 0);
17870 	/* e.g. if helper call clobbers r{0,1}, expect r{2,3,4,5} in the pattern */
17871 	expected_regs_mask = ~clobbered_regs_mask & ALL_CALLER_SAVED_REGS;
17872 
17873 	/* match pairs of form:
17874 	 *
17875 	 * *(u64 *)(r10 - Y) = rX   (where Y % 8 == 0)
17876 	 * ...
17877 	 * call %[to_be_inlined]
17878 	 * ...
17879 	 * rX = *(u64 *)(r10 - Y)
17880 	 */
17881 	for (i = 1, off = lowest_off; i <= ARRAY_SIZE(caller_saved); ++i, off += BPF_REG_SIZE) {
17882 		if (insn_idx - i < 0 || insn_idx + i >= env->prog->len)
17883 			break;
17884 		/* stx/ldx/call must not be a jump targets, a jump to the first stx is fine */
17885 		if (bpf_is_jump_target(env, insn_idx - i + 1) ||
17886 		    bpf_is_jump_target(env, insn_idx + i))
17887 			break;
17888 		stx = &insns[insn_idx - i];
17889 		ldx = &insns[insn_idx + i];
17890 		/* must be a stack spill/fill pair */
17891 		if (stx->code != (BPF_STX | BPF_MEM | BPF_DW) ||
17892 		    ldx->code != (BPF_LDX | BPF_MEM | BPF_DW) ||
17893 		    stx->dst_reg != BPF_REG_10 ||
17894 		    ldx->src_reg != BPF_REG_10)
17895 			break;
17896 		/* must be a spill/fill for the same reg */
17897 		if (stx->src_reg != ldx->dst_reg)
17898 			break;
17899 		/* must be one of the previously unseen registers */
17900 		if ((BIT(stx->src_reg) & expected_regs_mask) == 0)
17901 			break;
17902 		/* must be a spill/fill for the same expected offset,
17903 		 * no need to check offset alignment, BPF_DW stack access
17904 		 * is always 8-byte aligned.
17905 		 */
17906 		if (stx->off != off || ldx->off != off)
17907 			break;
17908 		expected_regs_mask &= ~BIT(stx->src_reg);
17909 		env->insn_aux_data[insn_idx - i].fastcall_pattern = 1;
17910 		env->insn_aux_data[insn_idx + i].fastcall_pattern = 1;
17911 	}
17912 	if (i == 1)
17913 		return;
17914 
17915 	/* Conditionally set 'fastcall_spills_num' to allow forward
17916 	 * compatibility when more helper functions are marked as
17917 	 * bpf_fastcall at compile time than current kernel supports, e.g:
17918 	 *
17919 	 *   1: *(u64 *)(r10 - 8) = r1
17920 	 *   2: call A                  ;; assume A is bpf_fastcall for current kernel
17921 	 *   3: r1 = *(u64 *)(r10 - 8)
17922 	 *   4: *(u64 *)(r10 - 8) = r1
17923 	 *   5: call B                  ;; assume B is not bpf_fastcall for current kernel
17924 	 *   6: r1 = *(u64 *)(r10 - 8)
17925 	 *
17926 	 * There is no need to block bpf_fastcall rewrite for such program.
17927 	 * Set 'fastcall_pattern' for both calls to keep check_fastcall_stack_contract() happy,
17928 	 * don't set 'fastcall_spills_num' for call B so that remove_fastcall_spills_fills()
17929 	 * does not remove spill/fill pair {4,6}.
17930 	 */
17931 	if (cs.fastcall)
17932 		env->insn_aux_data[insn_idx].fastcall_spills_num = i - 1;
17933 	else
17934 		subprog->keep_fastcall_stack = 1;
17935 	subprog->fastcall_stack_off = min(subprog->fastcall_stack_off, off);
17936 }
17937 
17938 static int mark_fastcall_patterns(struct bpf_verifier_env *env)
17939 {
17940 	struct bpf_subprog_info *subprog = env->subprog_info;
17941 	struct bpf_insn *insn;
17942 	s16 lowest_off;
17943 	int s, i;
17944 
17945 	for (s = 0; s < env->subprog_cnt; ++s, ++subprog) {
17946 		/* find lowest stack spill offset used in this subprog */
17947 		lowest_off = 0;
17948 		for (i = subprog->start; i < (subprog + 1)->start; ++i) {
17949 			insn = env->prog->insnsi + i;
17950 			if (insn->code != (BPF_STX | BPF_MEM | BPF_DW) ||
17951 			    insn->dst_reg != BPF_REG_10)
17952 				continue;
17953 			lowest_off = min(lowest_off, insn->off);
17954 		}
17955 		/* use this offset to find fastcall patterns */
17956 		for (i = subprog->start; i < (subprog + 1)->start; ++i) {
17957 			insn = env->prog->insnsi + i;
17958 			if (insn->code != (BPF_JMP | BPF_CALL))
17959 				continue;
17960 			mark_fastcall_pattern_for_call(env, subprog, i, lowest_off);
17961 		}
17962 	}
17963 	return 0;
17964 }
17965 
17966 static void adjust_btf_func(struct bpf_verifier_env *env)
17967 {
17968 	struct bpf_prog_aux *aux = env->prog->aux;
17969 	int i;
17970 
17971 	if (!aux->func_info)
17972 		return;
17973 
17974 	/* func_info is not available for hidden subprogs */
17975 	for (i = 0; i < env->subprog_cnt - env->hidden_subprog_cnt; i++)
17976 		aux->func_info[i].insn_off = env->subprog_info[i].start;
17977 }
17978 
17979 /* Find id in idset and increment its count, or add new entry */
17980 static void idset_cnt_inc(struct bpf_idset *idset, u32 id)
17981 {
17982 	u32 i;
17983 
17984 	for (i = 0; i < idset->num_ids; i++) {
17985 		if (idset->entries[i].id == id) {
17986 			idset->entries[i].cnt++;
17987 			return;
17988 		}
17989 	}
17990 	/* New id */
17991 	if (idset->num_ids < BPF_ID_MAP_SIZE) {
17992 		idset->entries[idset->num_ids].id = id;
17993 		idset->entries[idset->num_ids].cnt = 1;
17994 		idset->num_ids++;
17995 	}
17996 }
17997 
17998 /* Find id in idset and return its count, or 0 if not found */
17999 static u32 idset_cnt_get(struct bpf_idset *idset, u32 id)
18000 {
18001 	u32 i;
18002 
18003 	for (i = 0; i < idset->num_ids; i++) {
18004 		if (idset->entries[i].id == id)
18005 			return idset->entries[i].cnt;
18006 	}
18007 	return 0;
18008 }
18009 
18010 /*
18011  * Clear singular scalar ids in a state.
18012  * A register with a non-zero id is called singular if no other register shares
18013  * the same base id. Such registers can be treated as independent (id=0).
18014  */
18015 void bpf_clear_singular_ids(struct bpf_verifier_env *env,
18016 			    struct bpf_verifier_state *st)
18017 {
18018 	struct bpf_idset *idset = &env->idset_scratch;
18019 	struct bpf_func_state *func;
18020 	struct bpf_reg_state *reg;
18021 
18022 	idset->num_ids = 0;
18023 
18024 	bpf_for_each_reg_in_vstate(st, func, reg, ({
18025 		if (reg->type != SCALAR_VALUE)
18026 			continue;
18027 		if (!reg->id)
18028 			continue;
18029 		idset_cnt_inc(idset, reg->id & ~BPF_ADD_CONST);
18030 	}));
18031 
18032 	bpf_for_each_reg_in_vstate(st, func, reg, ({
18033 		if (reg->type != SCALAR_VALUE)
18034 			continue;
18035 		if (!reg->id)
18036 			continue;
18037 		if (idset_cnt_get(idset, reg->id & ~BPF_ADD_CONST) == 1)
18038 			clear_scalar_id(reg);
18039 	}));
18040 }
18041 
18042 /* Return true if it's OK to have the same insn return a different type. */
18043 static bool reg_type_mismatch_ok(enum bpf_reg_type type)
18044 {
18045 	switch (base_type(type)) {
18046 	case PTR_TO_CTX:
18047 	case PTR_TO_SOCKET:
18048 	case PTR_TO_SOCK_COMMON:
18049 	case PTR_TO_TCP_SOCK:
18050 	case PTR_TO_XDP_SOCK:
18051 	case PTR_TO_BTF_ID:
18052 	case PTR_TO_ARENA:
18053 		return false;
18054 	case PTR_TO_MEM:
18055 		return !bpf_may_fault_on_deref(type);
18056 	default:
18057 		return true;
18058 	}
18059 }
18060 
18061 /* If an instruction was previously used with particular pointer types, then we
18062  * need to be careful to avoid cases such as the below, where it may be ok
18063  * for one branch accessing the pointer, but not ok for the other branch:
18064  *
18065  * R1 = sock_ptr
18066  * goto X;
18067  * ...
18068  * R1 = some_other_valid_ptr;
18069  * goto X;
18070  * ...
18071  * R2 = *(u32 *)(R1 + 0);
18072  */
18073 static bool reg_type_mismatch(enum bpf_reg_type src, enum bpf_reg_type prev)
18074 {
18075 	return src != prev && (!reg_type_mismatch_ok(src) ||
18076 			       !reg_type_mismatch_ok(prev));
18077 }
18078 
18079 static bool is_ptr_to_mem(enum bpf_reg_type type)
18080 {
18081 	return base_type(type) == PTR_TO_MEM;
18082 }
18083 
18084 static enum bpf_reg_type merge_ptr_types(enum bpf_reg_type type_a,
18085 					 enum bpf_reg_type type_b)
18086 {
18087 	bool to_mem = is_ptr_to_mem(type_a) || is_ptr_to_mem(type_b);
18088 	enum bpf_reg_type type_merged = to_mem ? PTR_TO_MEM : PTR_TO_BTF_ID;
18089 
18090 	if (bpf_may_fault_on_deref(type_a) || bpf_may_fault_on_deref(type_b))
18091 		type_merged |= to_mem ? MEM_RDONLY | PTR_UNTRUSTED :
18092 					PTR_UNTRUSTED;
18093 	else
18094 		type_merged |= ((type_a | type_b) & MEM_RDONLY);
18095 	return type_merged;
18096 }
18097 
18098 static int save_aux_ptr_type(struct bpf_verifier_env *env, enum bpf_reg_type type,
18099 			     bool allow_trust_mismatch)
18100 {
18101 	enum bpf_reg_type *prev_type = &env->insn_aux_data[env->insn_idx].ptr_type;
18102 
18103 	if (*prev_type == NOT_INIT) {
18104 		/* Saw a valid insn
18105 		 * dst_reg = *(u32 *)(src_reg + off)
18106 		 * save type to validate intersecting paths
18107 		 */
18108 		*prev_type = type;
18109 	} else if (reg_type_mismatch(type, *prev_type)) {
18110 		/* Abuser program is trying to use the same insn
18111 		 * dst_reg = *(u32*) (src_reg + off)
18112 		 * with different pointer types:
18113 		 * src_reg == ctx in one branch and
18114 		 * src_reg == stack|map in some other branch.
18115 		 * Reject it.
18116 		 */
18117 		if (allow_trust_mismatch &&
18118 		    bpf_is_ptr_to_mem_or_btf_id(type) &&
18119 		    bpf_is_ptr_to_mem_or_btf_id(*prev_type)) {
18120 			/*
18121 			 * Have to support a use case when one path through
18122 			 * the program yields a TRUSTED pointer while another
18123 			 * is UNTRUSTED. Merge them into a type which keeps
18124 			 * the BPF_PROBE_MEM/BPF_PROBE_MEMSX rewrite when
18125 			 * either side needs it.
18126 			 */
18127 			*prev_type = merge_ptr_types(type, *prev_type);
18128 		} else {
18129 			verbose(env, "same insn cannot be used with different pointers\n");
18130 			return -EINVAL;
18131 		}
18132 	}
18133 
18134 	return 0;
18135 }
18136 
18137 enum {
18138 	PROCESS_BPF_EXIT = 1,
18139 	INSN_IDX_UPDATED = 2,
18140 };
18141 
18142 static int process_bpf_exit_full(struct bpf_verifier_env *env,
18143 				 bool *do_print_state,
18144 				 bool exception_exit)
18145 {
18146 	struct bpf_func_state *cur_frame = cur_func(env);
18147 
18148 	/* We must do check_reference_leak here before
18149 	 * prepare_func_exit to handle the case when
18150 	 * state->curframe > 0, it may be a callback function,
18151 	 * for which reference_state must match caller reference
18152 	 * state when it exits.
18153 	 */
18154 	int err = check_resource_leak(env, exception_exit,
18155 				      exception_exit || !env->cur_state->curframe,
18156 				      exception_exit ? "bpf_throw" :
18157 				      "BPF_EXIT instruction in main prog");
18158 	if (err)
18159 		return err;
18160 
18161 	/* The side effect of the prepare_func_exit which is
18162 	 * being skipped is that it frees bpf_func_state.
18163 	 * Typically, process_bpf_exit will only be hit with
18164 	 * outermost exit. copy_verifier_state in pop_stack will
18165 	 * handle freeing of any extra bpf_func_state left over
18166 	 * from not processing all nested function exits. We
18167 	 * also skip return code checks as they are not needed
18168 	 * for exceptional exits.
18169 	 */
18170 	if (exception_exit)
18171 		return PROCESS_BPF_EXIT;
18172 
18173 	if (env->cur_state->curframe) {
18174 		/* exit from nested function */
18175 		err = prepare_func_exit(env, &env->insn_idx);
18176 		if (err)
18177 			return err;
18178 		*do_print_state = true;
18179 		return INSN_IDX_UPDATED;
18180 	}
18181 
18182 	/*
18183 	 * Return from a regular global subprogram differs from return
18184 	 * from the main program or async/exception callback.
18185 	 * Main program exit implies return code restrictions
18186 	 * that depend on program type.
18187 	 * Exit from exception callback is equivalent to main program exit.
18188 	 * Exit from async callback implies return code restrictions
18189 	 * that depend on async scheduling mechanism.
18190 	 */
18191 	if (cur_frame->subprogno &&
18192 	    !cur_frame->in_async_callback_fn &&
18193 	    !cur_frame->in_exception_callback_fn)
18194 		err = check_global_subprog_return_code(env);
18195 	else
18196 		err = check_return_code(env, BPF_REG_0, "R0");
18197 	if (err)
18198 		return err;
18199 	return PROCESS_BPF_EXIT;
18200 }
18201 
18202 static int indirect_jump_min_max_index(struct bpf_verifier_env *env,
18203 				       int regno,
18204 				       struct bpf_map *map,
18205 				       u32 *pmin_index, u32 *pmax_index)
18206 {
18207 	struct bpf_reg_state *reg = reg_state(env, regno);
18208 	u64 min_index = reg_umin(reg);
18209 	u64 max_index = reg_umax(reg);
18210 	const u32 size = 8;
18211 
18212 	if (min_index > (u64) U32_MAX * size) {
18213 		verbose(env, "the sum of R%u umin_value %llu is too big\n", regno, reg_umin(reg));
18214 		return -ERANGE;
18215 	}
18216 	if (max_index > (u64) U32_MAX * size) {
18217 		verbose(env, "the sum of R%u umax_value %llu is too big\n", regno, reg_umax(reg));
18218 		return -ERANGE;
18219 	}
18220 
18221 	min_index /= size;
18222 	max_index /= size;
18223 
18224 	if (max_index >= map->max_entries) {
18225 		verbose(env, "R%u points to outside of jump table: [%llu,%llu] max_entries %u\n",
18226 			     regno, min_index, max_index, map->max_entries);
18227 		return -EINVAL;
18228 	}
18229 
18230 	*pmin_index = min_index;
18231 	*pmax_index = max_index;
18232 	return 0;
18233 }
18234 
18235 /* gotox *dst_reg */
18236 static int check_indirect_jump(struct bpf_verifier_env *env, struct bpf_insn *insn)
18237 {
18238 	struct bpf_verifier_state *other_branch;
18239 	struct bpf_reg_state *dst_reg;
18240 	struct bpf_map *map;
18241 	u32 min_index, max_index;
18242 	int err = 0;
18243 	int n;
18244 	int i;
18245 
18246 	dst_reg = reg_state(env, insn->dst_reg);
18247 	if (dst_reg->type != PTR_TO_INSN) {
18248 		verbose(env, "R%d has type %s, expected PTR_TO_INSN\n",
18249 			     insn->dst_reg, reg_type_str(env, dst_reg->type));
18250 		return -EINVAL;
18251 	}
18252 
18253 	map = dst_reg->map_ptr;
18254 	if (verifier_bug_if(!map, env, "R%d has an empty map pointer", insn->dst_reg))
18255 		return -EFAULT;
18256 
18257 	if (verifier_bug_if(map->map_type != BPF_MAP_TYPE_INSN_ARRAY, env,
18258 			    "R%d has incorrect map type %d", insn->dst_reg, map->map_type))
18259 		return -EFAULT;
18260 
18261 	err = indirect_jump_min_max_index(env, insn->dst_reg, map, &min_index, &max_index);
18262 	if (err)
18263 		return err;
18264 
18265 	/* Ensure that the buffer is large enough */
18266 	if (!env->gotox_tmp_buf || env->gotox_tmp_buf->cnt < max_index - min_index + 1) {
18267 		env->gotox_tmp_buf = bpf_iarray_realloc(env->gotox_tmp_buf,
18268 						        max_index - min_index + 1);
18269 		if (!env->gotox_tmp_buf)
18270 			return -ENOMEM;
18271 	}
18272 
18273 	n = bpf_copy_insn_array_uniq(map, min_index, max_index, env->gotox_tmp_buf->items);
18274 	if (n < 0)
18275 		return n;
18276 	if (n == 0) {
18277 		verbose(env, "register R%d doesn't point to any offset in map id=%d\n",
18278 			     insn->dst_reg, map->id);
18279 		return -EINVAL;
18280 	}
18281 
18282 	for (i = 0; i < n - 1; i++) {
18283 		mark_indirect_target(env, env->gotox_tmp_buf->items[i]);
18284 		other_branch = push_stack(env, env->gotox_tmp_buf->items[i],
18285 					  env->insn_idx, env->cur_state->speculative);
18286 		if (IS_ERR(other_branch))
18287 			return PTR_ERR(other_branch);
18288 	}
18289 	env->insn_idx = env->gotox_tmp_buf->items[n-1];
18290 	mark_indirect_target(env, env->insn_idx);
18291 	return INSN_IDX_UPDATED;
18292 }
18293 
18294 static int do_check_insn(struct bpf_verifier_env *env, bool *do_print_state)
18295 {
18296 	int err;
18297 	struct bpf_insn *insn = &env->prog->insnsi[env->insn_idx];
18298 	u8 class = BPF_CLASS(insn->code);
18299 
18300 	switch (class) {
18301 	case BPF_ALU:
18302 	case BPF_ALU64:
18303 		return check_alu_op(env, insn);
18304 
18305 	case BPF_LDX:
18306 		return check_load_mem(env, insn, false,
18307 				      BPF_MODE(insn->code) == BPF_MEMSX,
18308 				      true, "ldx");
18309 
18310 	case BPF_STX:
18311 		if (BPF_MODE(insn->code) == BPF_ATOMIC)
18312 			return check_atomic(env, insn);
18313 		return check_store_reg(env, insn, false);
18314 
18315 	case BPF_ST: {
18316 		/* Handle stack arg write (store immediate) */
18317 		if (is_stack_arg_st(insn)) {
18318 			struct bpf_verifier_state *vstate = env->cur_state;
18319 			struct bpf_func_state *state = vstate->frame[vstate->curframe];
18320 
18321 			return check_stack_arg_write(env, state, insn->off, NULL);
18322 		}
18323 
18324 		enum bpf_reg_type dst_reg_type;
18325 
18326 		err = check_reg_arg(env, insn->dst_reg, SRC_OP);
18327 		if (err)
18328 			return err;
18329 
18330 		dst_reg_type = cur_regs(env)[insn->dst_reg].type;
18331 
18332 		err = check_mem_access(env, env->insn_idx, cur_regs(env) + insn->dst_reg, argno_from_reg(insn->dst_reg),
18333 				       insn->off, BPF_SIZE(insn->code),
18334 				       BPF_WRITE, -1, false, false);
18335 		if (err)
18336 			return err;
18337 
18338 		return save_aux_ptr_type(env, dst_reg_type, false);
18339 	}
18340 	case BPF_JMP:
18341 	case BPF_JMP32: {
18342 		u8 opcode = BPF_OP(insn->code);
18343 
18344 		env->jmps_processed++;
18345 		if (opcode == BPF_CALL) {
18346 			if (env->cur_state->active_locks) {
18347 				if ((insn->src_reg == BPF_REG_0 &&
18348 				     insn->imm != BPF_FUNC_spin_unlock &&
18349 				     insn->imm != BPF_FUNC_kptr_xchg) ||
18350 				    (insn->src_reg == BPF_PSEUDO_KFUNC_CALL &&
18351 				     !kfunc_spin_allowed(env, insn->imm, insn->off))) {
18352 					verbose(env,
18353 						"function calls are not allowed while holding a lock\n");
18354 					bpf_diag_ctx_active(
18355 						env, env->insn_idx,
18356 						"function call", BPF_DIAG_CONTEXT_LOCK,
18357 						"Release the BPF spin lock before making this call, or move the call outside the locked region.");
18358 					return -EINVAL;
18359 				}
18360 			}
18361 			mark_reg_scratched(env, BPF_REG_0);
18362 			if (bpf_in_stack_arg_cnt(&env->subprog_info[cur_func(env)->subprogno]))
18363 				cur_func(env)->no_stack_arg_load = true;
18364 			if (insn->src_reg == BPF_PSEUDO_CALL)
18365 				return check_func_call(env, insn, &env->insn_idx);
18366 			if (insn->src_reg == BPF_PSEUDO_KFUNC_CALL)
18367 				return check_kfunc_call(env, insn, &env->insn_idx);
18368 			return check_helper_call(env, insn, &env->insn_idx);
18369 		} else if (opcode == BPF_JA) {
18370 			if (BPF_SRC(insn->code) == BPF_X)
18371 				return check_indirect_jump(env, insn);
18372 
18373 			if (class == BPF_JMP)
18374 				env->insn_idx += insn->off + 1;
18375 			else
18376 				env->insn_idx += insn->imm + 1;
18377 			return INSN_IDX_UPDATED;
18378 		} else if (opcode == BPF_EXIT) {
18379 			return process_bpf_exit_full(env, do_print_state, false);
18380 		}
18381 		return check_cond_jmp_op(env, insn, &env->insn_idx);
18382 	}
18383 	case BPF_LD: {
18384 		u8 mode = BPF_MODE(insn->code);
18385 
18386 		if (mode == BPF_ABS || mode == BPF_IND)
18387 			return check_ld_abs(env, insn);
18388 
18389 		if (mode == BPF_IMM) {
18390 			err = check_ld_imm(env, insn);
18391 			if (err)
18392 				return err;
18393 
18394 			env->insn_idx++;
18395 			sanitize_mark_insn_seen(env);
18396 		}
18397 		return 0;
18398 	}
18399 	}
18400 	/* all class values are handled above. silence compiler warning */
18401 	return -EFAULT;
18402 }
18403 
18404 static int do_check(struct bpf_verifier_env *env)
18405 {
18406 	bool pop_log = !(env->log.level & BPF_LOG_LEVEL2);
18407 	struct bpf_verifier_state *state = env->cur_state;
18408 	struct bpf_insn *insns = env->prog->insnsi;
18409 	int insn_cnt = env->prog->len;
18410 	bool do_print_state = false;
18411 	int prev_insn_idx = -1;
18412 
18413 	for (;;) {
18414 		struct bpf_insn *insn;
18415 		struct bpf_insn_aux_data *insn_aux;
18416 		int err;
18417 
18418 		/* reset current history entry on each new instruction */
18419 		env->cur_hist_ent = NULL;
18420 
18421 		env->prev_insn_idx = prev_insn_idx;
18422 		if (env->insn_idx >= insn_cnt) {
18423 			verbose(env, "invalid insn idx %d insn_cnt %d\n",
18424 				env->insn_idx, insn_cnt);
18425 			return -EFAULT;
18426 		}
18427 
18428 		insn = &insns[env->insn_idx];
18429 		insn_aux = &env->insn_aux_data[env->insn_idx];
18430 
18431 		account_processed_insn(env);
18432 
18433 		if (env->insn_processed > BPF_COMPLEXITY_LIMIT_INSNS) {
18434 			verbose(env,
18435 				"BPF program is too large. Processed %d insn\n",
18436 				env->insn_processed);
18437 			return -E2BIG;
18438 		}
18439 
18440 		state->last_insn_idx = env->prev_insn_idx;
18441 		state->insn_idx = env->insn_idx;
18442 		/*
18443 		 * Record the incoming edge so active and queued paths use the same
18444 		 * branch-recording path. A zero-offset conditional has identical
18445 		 * successors, so its outcome cannot be reconstructed from the edge.
18446 		 */
18447 		if (!state->speculative && prev_insn_idx >= 0 && prev_insn_idx < insn_cnt) {
18448 			struct bpf_insn *prev_insn = &insns[prev_insn_idx];
18449 			int fallthrough_idx = prev_insn_idx + 1;
18450 			int branch_idx = prev_insn_idx + bpf_jmp_offset(prev_insn) + 1;
18451 			u8 class = BPF_CLASS(prev_insn->code);
18452 			u8 opcode = BPF_OP(prev_insn->code);
18453 
18454 			if ((class == BPF_JMP || class == BPF_JMP32) &&
18455 			    opcode != BPF_JA && opcode != BPF_CALL && opcode != BPF_EXIT &&
18456 			    opcode <= BPF_JCOND && branch_idx != fallthrough_idx) {
18457 				if (env->insn_idx == branch_idx)
18458 					bpf_diag_record_branch(env, prev_insn_idx, true);
18459 				else if (env->insn_idx == fallthrough_idx)
18460 					bpf_diag_record_branch(env, prev_insn_idx, false);
18461 			}
18462 		}
18463 
18464 		if (bpf_is_prune_point(env, env->insn_idx)) {
18465 			err = bpf_is_state_visited(env, env->insn_idx);
18466 			if (err < 0)
18467 				return err;
18468 			if (err == 1) {
18469 				/* found equivalent state, can prune the search */
18470 				if (env->log.level & BPF_LOG_LEVEL) {
18471 					if (do_print_state)
18472 						verbose(env, "\nfrom %d to %d%s: safe\n",
18473 							env->prev_insn_idx, env->insn_idx,
18474 							env->cur_state->speculative ?
18475 							" (speculative execution)" : "");
18476 					else
18477 						verbose(env, "%d: safe\n", env->insn_idx);
18478 				}
18479 				goto process_bpf_exit;
18480 			}
18481 		}
18482 
18483 		if (bpf_is_jmp_point(env, env->insn_idx)) {
18484 			err = bpf_push_jmp_history(env, state, 0, 0, 0, 0);
18485 			if (err)
18486 				return err;
18487 		}
18488 
18489 		if (signal_pending(current))
18490 			return -EAGAIN;
18491 
18492 		if (need_resched())
18493 			cond_resched();
18494 
18495 		if (env->log.level & BPF_LOG_LEVEL2 && do_print_state) {
18496 			verbose(env, "\nfrom %d to %d%s:",
18497 				env->prev_insn_idx, env->insn_idx,
18498 				env->cur_state->speculative ?
18499 				" (speculative execution)" : "");
18500 			print_verifier_state(env, state, state->curframe, true);
18501 			do_print_state = false;
18502 		}
18503 
18504 		if (env->log.level & BPF_LOG_LEVEL) {
18505 			if (verifier_state_scratched(env))
18506 				print_insn_state(env, state, state->curframe);
18507 
18508 			verbose_linfo(env, env->insn_idx, "; ");
18509 			env->prev_log_pos = env->log.end_pos;
18510 			verbose(env, "%d: ", env->insn_idx);
18511 			bpf_verbose_insn(env, insn);
18512 			verbose(env, "\n");
18513 			env->prev_insn_print_pos = env->log.end_pos - env->prev_log_pos;
18514 			env->prev_log_pos = env->log.end_pos;
18515 		}
18516 
18517 		if (bpf_prog_is_offloaded(env->prog->aux)) {
18518 			err = bpf_prog_offload_verify_insn(env, env->insn_idx,
18519 							   env->prev_insn_idx);
18520 			if (err)
18521 				return err;
18522 		}
18523 
18524 		sanitize_mark_insn_seen(env);
18525 		prev_insn_idx = env->insn_idx;
18526 
18527 		/* Sanity check: precomputed constants must match verifier state */
18528 		if (!state->speculative && insn_aux->const_reg_mask) {
18529 			struct bpf_reg_state *regs = cur_regs(env);
18530 			u16 mask = insn_aux->const_reg_mask;
18531 
18532 			for (int r = 0; r < ARRAY_SIZE(insn_aux->const_reg_vals); r++) {
18533 				u32 cval = insn_aux->const_reg_vals[r];
18534 
18535 				if (!(mask & BIT(r)))
18536 					continue;
18537 				if (regs[r].type != SCALAR_VALUE)
18538 					continue;
18539 				if (!tnum_is_const(regs[r].var_off))
18540 					continue;
18541 				if (verifier_bug_if((u32)regs[r].var_off.value != cval,
18542 						    env, "const R%d: %u != %llu",
18543 						    r, cval, regs[r].var_off.value))
18544 					return -EFAULT;
18545 			}
18546 		}
18547 
18548 		/* Reduce verification complexity by stopping speculative path
18549 		 * verification when a nospec is encountered.
18550 		 */
18551 		if (state->speculative && insn_aux->nospec)
18552 			goto process_bpf_exit;
18553 
18554 		err = do_check_insn(env, &do_print_state);
18555 		if (error_recoverable_with_nospec(err) && state->speculative) {
18556 			/* Prevent this speculative path from ever reaching the
18557 			 * insn that would have been unsafe to execute.
18558 			 */
18559 			insn_aux->nospec = true;
18560 			/* If it was an ADD/SUB insn, potentially remove any
18561 			 * markings for alu sanitization.
18562 			 */
18563 			insn_aux->alu_state = 0;
18564 			goto process_bpf_exit;
18565 		} else if (err < 0) {
18566 			return err;
18567 		} else if (err == PROCESS_BPF_EXIT) {
18568 			goto process_bpf_exit;
18569 		} else if (err == INSN_IDX_UPDATED) {
18570 		} else if (err == 0) {
18571 			env->insn_idx++;
18572 		}
18573 
18574 		if (state->speculative && insn_aux->nospec_result) {
18575 			/* If we are on a path that performed a jump-op, this
18576 			 * may skip a nospec patched-in after the jump. This can
18577 			 * currently never happen because nospec_result is only
18578 			 * used for the write-ops
18579 			 * `*(size*)(dst_reg+off)=src_reg|imm32` and helper
18580 			 * calls. These must never skip the following insn
18581 			 * (i.e., bpf_insn_successors()'s opcode_info.can_jump
18582 			 * is false). Still, add a warning to document this in
18583 			 * case nospec_result is used elsewhere in the future.
18584 			 *
18585 			 * All non-branch instructions have a single
18586 			 * fall-through edge. For these, nospec_result should
18587 			 * already work.
18588 			 */
18589 			if (verifier_bug_if((BPF_CLASS(insn->code) == BPF_JMP ||
18590 					     BPF_CLASS(insn->code) == BPF_JMP32) &&
18591 					    BPF_OP(insn->code) != BPF_CALL, env,
18592 					    "speculation barrier after jump instruction may not have the desired effect"))
18593 				return -EFAULT;
18594 process_bpf_exit:
18595 			account_current_path(env);
18596 			mark_verifier_state_scratched(env);
18597 			err = bpf_update_branch_counts(env, env->cur_state);
18598 			if (err)
18599 				return err;
18600 			err = pop_stack(env, &prev_insn_idx, &env->insn_idx,
18601 					pop_log);
18602 			if (err < 0) {
18603 				if (err != -ENOENT)
18604 					return err;
18605 				break;
18606 			} else {
18607 				do_print_state = true;
18608 				continue;
18609 			}
18610 		}
18611 	}
18612 
18613 	return 0;
18614 }
18615 
18616 static int find_btf_percpu_datasec(struct btf *btf)
18617 {
18618 	const struct btf_type *t;
18619 	const char *tname;
18620 	int i, n;
18621 
18622 	/*
18623 	 * Both vmlinux and module each have their own ".data..percpu"
18624 	 * DATASECs in BTF. So for module's case, we need to skip vmlinux BTF
18625 	 * types to look at only module's own BTF types.
18626 	 */
18627 	n = btf_nr_types(btf);
18628 	for (i = btf_named_start_id(btf, true); i < n; i++) {
18629 		t = btf_type_by_id(btf, i);
18630 		if (BTF_INFO_KIND(t->info) != BTF_KIND_DATASEC)
18631 			continue;
18632 
18633 		tname = btf_name_by_offset(btf, t->name_off);
18634 		if (!strcmp(tname, ".data..percpu"))
18635 			return i;
18636 	}
18637 
18638 	return -ENOENT;
18639 }
18640 
18641 /*
18642  * Add btf to the env->used_btfs array. If needed, refcount the
18643  * corresponding kernel module. To simplify caller's logic
18644  * in case of error or if btf was added before the function
18645  * decreases the btf refcount.
18646  */
18647 static int __add_used_btf(struct bpf_verifier_env *env, struct btf *btf)
18648 {
18649 	struct btf_mod_pair *btf_mod;
18650 	int ret = 0;
18651 	int i;
18652 
18653 	/* check whether we recorded this BTF (and maybe module) already */
18654 	for (i = 0; i < env->used_btf_cnt; i++)
18655 		if (env->used_btfs[i].btf == btf)
18656 			goto ret_put;
18657 
18658 	if (env->signature) {
18659 		verbose(env, "signed program cannot bind any BTF\n");
18660 		ret = -EACCES;
18661 		goto ret_put;
18662 	}
18663 	if (env->used_btf_cnt >= MAX_USED_BTFS) {
18664 		verbose(env, "The total number of btfs per program has reached the limit of %u\n",
18665 			MAX_USED_BTFS);
18666 		ret = -E2BIG;
18667 		goto ret_put;
18668 	}
18669 
18670 	btf_mod = &env->used_btfs[env->used_btf_cnt];
18671 	btf_mod->btf = btf;
18672 	btf_mod->module = NULL;
18673 
18674 	/* if we reference variables from kernel module, bump its refcount */
18675 	if (btf_is_module(btf)) {
18676 		btf_mod->module = btf_try_get_module(btf);
18677 		if (!btf_mod->module) {
18678 			ret = -ENXIO;
18679 			goto ret_put;
18680 		}
18681 	}
18682 
18683 	env->used_btf_cnt++;
18684 	return 0;
18685 
18686 ret_put:
18687 	/* Either error or this BTF was already added */
18688 	btf_put(btf);
18689 	return ret;
18690 }
18691 
18692 /* replace pseudo btf_id with kernel symbol address */
18693 static int __check_pseudo_btf_id(struct bpf_verifier_env *env,
18694 				 struct bpf_insn *insn,
18695 				 struct bpf_insn_aux_data *aux,
18696 				 struct btf *btf)
18697 {
18698 	const struct btf_var_secinfo *vsi;
18699 	const struct btf_type *datasec;
18700 	const struct btf_type *t;
18701 	const char *sym_name;
18702 	bool percpu = false;
18703 	u32 type, id = insn->imm;
18704 	s32 datasec_id;
18705 	u64 addr;
18706 	int i;
18707 
18708 	t = btf_type_by_id(btf, id);
18709 	if (!t) {
18710 		verbose(env, "ldimm64 insn specifies invalid btf_id %d.\n", id);
18711 		return -ENOENT;
18712 	}
18713 
18714 	if (!btf_type_is_var(t) && !btf_type_is_func(t)) {
18715 		verbose(env, "pseudo btf_id %d in ldimm64 isn't KIND_VAR or KIND_FUNC\n", id);
18716 		return -EINVAL;
18717 	}
18718 
18719 	sym_name = btf_name_by_offset(btf, t->name_off);
18720 	addr = kallsyms_lookup_name(sym_name);
18721 	if (!addr) {
18722 		verbose(env, "ldimm64 failed to find the address for kernel symbol '%s'.\n",
18723 			sym_name);
18724 		return -ENOENT;
18725 	}
18726 	insn[0].imm = (u32)addr;
18727 	insn[1].imm = addr >> 32;
18728 
18729 	if (btf_type_is_func(t)) {
18730 		aux->btf_var.reg_type = PTR_TO_MEM | MEM_RDONLY;
18731 		aux->btf_var.mem_size = 0;
18732 		return 0;
18733 	}
18734 
18735 	datasec_id = find_btf_percpu_datasec(btf);
18736 	if (datasec_id > 0) {
18737 		datasec = btf_type_by_id(btf, datasec_id);
18738 		for_each_vsi(i, datasec, vsi) {
18739 			if (vsi->type == id) {
18740 				percpu = true;
18741 				break;
18742 			}
18743 		}
18744 	}
18745 
18746 	type = t->type;
18747 	t = btf_type_skip_modifiers(btf, type, NULL);
18748 	if (percpu) {
18749 		aux->btf_var.reg_type = PTR_TO_BTF_ID | MEM_PERCPU;
18750 		aux->btf_var.btf = btf;
18751 		aux->btf_var.btf_id = type;
18752 	} else if (!btf_type_is_struct(t)) {
18753 		const struct btf_type *ret;
18754 		const char *tname;
18755 		u32 tsize;
18756 
18757 		/* resolve the type size of ksym. */
18758 		ret = btf_resolve_size(btf, t, &tsize);
18759 		if (IS_ERR(ret)) {
18760 			tname = btf_name_by_offset(btf, t->name_off);
18761 			verbose(env, "ldimm64 unable to resolve the size of type '%s': %ld\n",
18762 				tname, PTR_ERR(ret));
18763 			return -EINVAL;
18764 		}
18765 		aux->btf_var.reg_type = PTR_TO_MEM | MEM_RDONLY;
18766 		aux->btf_var.mem_size = tsize;
18767 	} else {
18768 		aux->btf_var.reg_type = PTR_TO_BTF_ID;
18769 		aux->btf_var.btf = btf;
18770 		aux->btf_var.btf_id = type;
18771 	}
18772 
18773 	return 0;
18774 }
18775 
18776 static int check_pseudo_btf_id(struct bpf_verifier_env *env,
18777 			       struct bpf_insn *insn,
18778 			       struct bpf_insn_aux_data *aux)
18779 {
18780 	struct btf *btf;
18781 	int btf_fd;
18782 	int err;
18783 
18784 	btf_fd = insn[1].imm;
18785 	if (btf_fd) {
18786 		btf = btf_get_by_fd(btf_fd);
18787 		if (IS_ERR(btf)) {
18788 			verbose(env, "invalid module BTF object FD specified.\n");
18789 			return -EINVAL;
18790 		}
18791 	} else {
18792 		if (!btf_vmlinux) {
18793 			verbose(env, "kernel is missing BTF, make sure CONFIG_DEBUG_INFO_BTF=y is specified in Kconfig.\n");
18794 			return -EINVAL;
18795 		}
18796 		btf_get(btf_vmlinux);
18797 		btf = btf_vmlinux;
18798 	}
18799 
18800 	err = __check_pseudo_btf_id(env, insn, aux, btf);
18801 	if (err) {
18802 		btf_put(btf);
18803 		return err;
18804 	}
18805 
18806 	return __add_used_btf(env, btf);
18807 }
18808 
18809 static bool is_tracing_prog_type(enum bpf_prog_type type)
18810 {
18811 	switch (type) {
18812 	case BPF_PROG_TYPE_KPROBE:
18813 	case BPF_PROG_TYPE_TRACEPOINT:
18814 	case BPF_PROG_TYPE_PERF_EVENT:
18815 	case BPF_PROG_TYPE_RAW_TRACEPOINT:
18816 	case BPF_PROG_TYPE_RAW_TRACEPOINT_WRITABLE:
18817 		return true;
18818 	default:
18819 		return false;
18820 	}
18821 }
18822 
18823 static bool bpf_map_is_cgroup_storage(struct bpf_map *map)
18824 {
18825 	return (map->map_type == BPF_MAP_TYPE_CGROUP_STORAGE ||
18826 		map->map_type == BPF_MAP_TYPE_PERCPU_CGROUP_STORAGE);
18827 }
18828 
18829 static int check_map_prog_compatibility(struct bpf_verifier_env *env,
18830 					struct bpf_map *map,
18831 					struct bpf_prog *prog)
18832 
18833 {
18834 	enum bpf_prog_type prog_type = resolve_prog_type(prog);
18835 
18836 	if (map->excl_prog_sha &&
18837 	    memcmp(map->excl_prog_sha, prog->digest, SHA256_DIGEST_SIZE)) {
18838 		verbose(env, "program's hash doesn't match map's excl_prog_hash\n");
18839 		return -EACCES;
18840 	}
18841 
18842 	if (btf_record_has_field(map->record, BPF_LIST_HEAD) ||
18843 	    btf_record_has_field(map->record, BPF_RB_ROOT)) {
18844 		if (is_tracing_prog_type(prog_type)) {
18845 			verbose(env, "tracing progs cannot use bpf_{list_head,rb_root} yet\n");
18846 			return -EINVAL;
18847 		}
18848 	}
18849 
18850 	if (btf_record_has_field(map->record, BPF_SPIN_LOCK | BPF_RES_SPIN_LOCK)) {
18851 		if (prog_type == BPF_PROG_TYPE_SOCKET_FILTER) {
18852 			verbose(env, "socket filter progs cannot use bpf_spin_lock yet\n");
18853 			return -EINVAL;
18854 		}
18855 	}
18856 
18857 	if (btf_record_has_field(map->record, BPF_SPIN_LOCK)) {
18858 		if (is_tracing_prog_type(prog_type)) {
18859 			verbose(env, "tracing progs cannot use bpf_spin_lock yet\n");
18860 			return -EINVAL;
18861 		}
18862 	}
18863 
18864 	if ((bpf_prog_is_offloaded(prog->aux) || bpf_map_is_offloaded(map)) &&
18865 	    !bpf_offload_prog_map_match(prog, map)) {
18866 		verbose(env, "offload device mismatch between prog and map\n");
18867 		return -EINVAL;
18868 	}
18869 
18870 	if (map->map_type == BPF_MAP_TYPE_STRUCT_OPS) {
18871 		verbose(env, "bpf_struct_ops map cannot be used in prog\n");
18872 		return -EINVAL;
18873 	}
18874 
18875 	if (prog->sleepable)
18876 		switch (map->map_type) {
18877 		case BPF_MAP_TYPE_HASH:
18878 		case BPF_MAP_TYPE_RHASH:
18879 		case BPF_MAP_TYPE_LRU_HASH:
18880 		case BPF_MAP_TYPE_ARRAY:
18881 		case BPF_MAP_TYPE_PERCPU_HASH:
18882 		case BPF_MAP_TYPE_PERCPU_ARRAY:
18883 		case BPF_MAP_TYPE_LRU_PERCPU_HASH:
18884 		case BPF_MAP_TYPE_LPM_TRIE:
18885 		case BPF_MAP_TYPE_ARRAY_OF_MAPS:
18886 		case BPF_MAP_TYPE_HASH_OF_MAPS:
18887 		case BPF_MAP_TYPE_RINGBUF:
18888 		case BPF_MAP_TYPE_USER_RINGBUF:
18889 		case BPF_MAP_TYPE_INODE_STORAGE:
18890 		case BPF_MAP_TYPE_SK_STORAGE:
18891 		case BPF_MAP_TYPE_TASK_STORAGE:
18892 		case BPF_MAP_TYPE_CGRP_STORAGE:
18893 		case BPF_MAP_TYPE_QUEUE:
18894 		case BPF_MAP_TYPE_STACK:
18895 		case BPF_MAP_TYPE_ARENA:
18896 		case BPF_MAP_TYPE_INSN_ARRAY:
18897 		case BPF_MAP_TYPE_PROG_ARRAY:
18898 			break;
18899 		default:
18900 			verbose(env,
18901 				"Sleepable programs can only use array, hash, ringbuf and local storage maps\n");
18902 			return -EINVAL;
18903 		}
18904 
18905 	if (bpf_map_is_cgroup_storage(map) &&
18906 	    bpf_cgroup_storage_assign(env->prog->aux, map)) {
18907 		verbose(env, "only one cgroup storage of each type is allowed\n");
18908 		return -EBUSY;
18909 	}
18910 
18911 	if (map->map_type == BPF_MAP_TYPE_ARENA) {
18912 		if (env->prog->aux->arena) {
18913 			verbose(env, "Only one arena per program\n");
18914 			return -EBUSY;
18915 		}
18916 		if (!env->allow_ptr_leaks || !env->bpf_capable) {
18917 			verbose(env, "CAP_BPF and CAP_PERFMON are required to use arena\n");
18918 			return -EPERM;
18919 		}
18920 		if (!env->prog->jit_requested) {
18921 			verbose(env, "JIT is required to use arena\n");
18922 			return -EOPNOTSUPP;
18923 		}
18924 		if (!bpf_jit_supports_arena()) {
18925 			verbose(env, "JIT doesn't support arena\n");
18926 			return -EOPNOTSUPP;
18927 		}
18928 		env->prog->aux->arena = (void *)map;
18929 		env->prog->jit_required = true;
18930 		if (!bpf_arena_get_user_vm_start(env->prog->aux->arena)) {
18931 			verbose(env, "arena's user address must be set via map_extra or mmap()\n");
18932 			return -EINVAL;
18933 		}
18934 	}
18935 
18936 	return 0;
18937 }
18938 
18939 static int __add_used_map(struct bpf_verifier_env *env, struct bpf_map *map)
18940 {
18941 	int i, err;
18942 
18943 	/* check whether we recorded this map already */
18944 	for (i = 0; i < env->used_map_cnt; i++)
18945 		if (env->used_maps[i] == map)
18946 			return i;
18947 
18948 	if (env->signature &&
18949 	    env->prog->aux->sig.verdict == BPF_SIG_VERIFIED) {
18950 		verbose(env, "signed program cannot bind map '%s' not covered by the signature\n",
18951 			map->name);
18952 		return -EACCES;
18953 	}
18954 	if (env->used_map_cnt >= MAX_USED_MAPS) {
18955 		verbose(env, "The total number of maps per program has reached the limit of %u\n",
18956 			MAX_USED_MAPS);
18957 		return -E2BIG;
18958 	}
18959 
18960 	err = check_map_prog_compatibility(env, map, env->prog);
18961 	if (err)
18962 		return err;
18963 
18964 	if (env->prog->sleepable)
18965 		atomic64_inc(&map->sleepable_refcnt);
18966 
18967 	/* hold the map. If the program is rejected by verifier,
18968 	 * the map will be released by release_maps() or it
18969 	 * will be used by the valid program until it's unloaded
18970 	 * and all maps are released in bpf_free_used_maps()
18971 	 */
18972 	bpf_map_inc(map);
18973 
18974 	env->used_maps[env->used_map_cnt++] = map;
18975 
18976 	if (map->map_type == BPF_MAP_TYPE_INSN_ARRAY) {
18977 		err = bpf_insn_array_init(map, env->prog);
18978 		if (err) {
18979 			verbose(env, "Failed to properly initialize insn array\n");
18980 			return err;
18981 		}
18982 		env->insn_array_maps[env->insn_array_map_cnt++] = map;
18983 		env->prog->jit_required = true;
18984 	}
18985 
18986 	return env->used_map_cnt - 1;
18987 }
18988 
18989 /* Add map behind fd to used maps list, if it's not already there, and return
18990  * its index.
18991  * Returns <0 on error, or >= 0 index, on success.
18992  */
18993 static int add_used_map(struct bpf_verifier_env *env, int fd)
18994 {
18995 	struct bpf_map *map;
18996 	CLASS(fd, f)(fd);
18997 
18998 	map = __bpf_map_get(f);
18999 	if (IS_ERR(map)) {
19000 		verbose(env, "fd %d is not pointing to valid bpf_map\n", fd);
19001 		return PTR_ERR(map);
19002 	}
19003 
19004 	return __add_used_map(env, map);
19005 }
19006 
19007 static int fd_array_get_map_idx_continuous(struct bpf_verifier_env *env, u32 idx)
19008 {
19009 	struct bpf_map *map;
19010 
19011 	if (idx >= env->fd_array_cnt) {
19012 		verbose(env, "fd_idx %u out of bounds, fd_array_cnt %u\n",
19013 			idx, env->fd_array_cnt);
19014 		return -EINVAL;
19015 	}
19016 	map = fd_slot_map(env->fd_array[idx]);
19017 	if (!map) {
19018 		verbose(env, "fd_idx %u is not a map\n", idx);
19019 		return -EINVAL;
19020 	}
19021 	return __add_used_map(env, map);
19022 }
19023 
19024 static int fd_array_get_map_idx_sparse(struct bpf_verifier_env *env, u32 idx)
19025 {
19026 	int fd;
19027 
19028 	if (copy_from_bpfptr_offset(&fd, env->fd_array_raw,
19029 				    (size_t)idx * sizeof(fd), sizeof(fd)))
19030 		return -EFAULT;
19031 	return add_used_map(env, fd);
19032 }
19033 
19034 static int fd_array_get_map_idx(struct bpf_verifier_env *env, u32 idx)
19035 {
19036 	if (env->fd_array)
19037 		return fd_array_get_map_idx_continuous(env, idx);
19038 	if (env->signature) {
19039 		verbose(env, "signed program must bind maps via a continuous fd_array (fd_array_cnt)\n");
19040 		return -EACCES;
19041 	}
19042 	if (!bpfptr_is_null(env->fd_array_raw))
19043 		return fd_array_get_map_idx_sparse(env, idx);
19044 
19045 	verbose(env, "fd_idx without fd_array is invalid\n");
19046 	return -EPROTO;
19047 }
19048 
19049 static int check_alu_fields(struct bpf_verifier_env *env, struct bpf_insn *insn)
19050 {
19051 	u8 class = BPF_CLASS(insn->code);
19052 	u8 opcode = BPF_OP(insn->code);
19053 
19054 	switch (opcode) {
19055 	case BPF_NEG:
19056 		if (BPF_SRC(insn->code) != BPF_K || insn->src_reg != BPF_REG_0 ||
19057 		    insn->off != 0 || insn->imm != 0) {
19058 			verbose(env, "BPF_NEG uses reserved fields\n");
19059 			return -EINVAL;
19060 		}
19061 		return 0;
19062 	case BPF_END:
19063 		if (insn->src_reg != BPF_REG_0 || insn->off != 0 ||
19064 		    (insn->imm != 16 && insn->imm != 32 && insn->imm != 64) ||
19065 		    (class == BPF_ALU64 && BPF_SRC(insn->code) != BPF_TO_LE)) {
19066 			verbose(env, "BPF_END uses reserved fields\n");
19067 			return -EINVAL;
19068 		}
19069 		return 0;
19070 	case BPF_MOV:
19071 		if (BPF_SRC(insn->code) == BPF_X) {
19072 			if (class == BPF_ALU) {
19073 				if ((insn->off != 0 && insn->off != 8 && insn->off != 16) ||
19074 				    insn->imm) {
19075 					verbose(env, "BPF_MOV uses reserved fields\n");
19076 					return -EINVAL;
19077 				}
19078 			} else if (insn->off == BPF_ADDR_SPACE_CAST) {
19079 				if (insn->imm != 1 && insn->imm != 1u << 16) {
19080 					verbose(env, "addr_space_cast insn can only convert between address space 1 and 0\n");
19081 					return -EINVAL;
19082 				}
19083 			} else if ((insn->off != 0 && insn->off != 8 &&
19084 				    insn->off != 16 && insn->off != 32) || insn->imm) {
19085 				verbose(env, "BPF_MOV uses reserved fields\n");
19086 				return -EINVAL;
19087 			}
19088 		} else if (insn->src_reg != BPF_REG_0 || insn->off != 0) {
19089 			verbose(env, "BPF_MOV uses reserved fields\n");
19090 			return -EINVAL;
19091 		}
19092 		return 0;
19093 	case BPF_ADD:
19094 	case BPF_SUB:
19095 	case BPF_AND:
19096 	case BPF_OR:
19097 	case BPF_XOR:
19098 	case BPF_LSH:
19099 	case BPF_RSH:
19100 	case BPF_ARSH:
19101 	case BPF_MUL:
19102 	case BPF_DIV:
19103 	case BPF_MOD:
19104 		if (BPF_SRC(insn->code) == BPF_X) {
19105 			if (insn->imm != 0 || (insn->off != 0 && insn->off != 1) ||
19106 			    (insn->off == 1 && opcode != BPF_MOD && opcode != BPF_DIV)) {
19107 				verbose(env, "BPF_ALU uses reserved fields\n");
19108 				return -EINVAL;
19109 			}
19110 		} else if (insn->src_reg != BPF_REG_0 ||
19111 			   (insn->off != 0 && insn->off != 1) ||
19112 			   (insn->off == 1 && opcode != BPF_MOD && opcode != BPF_DIV)) {
19113 			verbose(env, "BPF_ALU uses reserved fields\n");
19114 			return -EINVAL;
19115 		}
19116 		return 0;
19117 	default:
19118 		verbose(env, "invalid BPF_ALU opcode %x\n", opcode);
19119 		return -EINVAL;
19120 	}
19121 }
19122 
19123 static int check_jmp_fields(struct bpf_verifier_env *env, struct bpf_insn *insn)
19124 {
19125 	u8 class = BPF_CLASS(insn->code);
19126 	u8 opcode = BPF_OP(insn->code);
19127 
19128 	switch (opcode) {
19129 	case BPF_CALL:
19130 		if (BPF_SRC(insn->code) != BPF_K ||
19131 		    (insn->src_reg != BPF_PSEUDO_KFUNC_CALL && insn->off != 0) ||
19132 		    (insn->src_reg != BPF_REG_0 && insn->src_reg != BPF_PSEUDO_CALL &&
19133 		     insn->src_reg != BPF_PSEUDO_KFUNC_CALL) ||
19134 		    insn->dst_reg != BPF_REG_0 || class == BPF_JMP32) {
19135 			verbose(env, "BPF_CALL uses reserved fields\n");
19136 			return -EINVAL;
19137 		}
19138 		return 0;
19139 	case BPF_JA:
19140 		if (BPF_SRC(insn->code) == BPF_X) {
19141 			if (insn->src_reg != BPF_REG_0 || insn->imm != 0 || insn->off != 0) {
19142 				verbose(env, "BPF_JA|BPF_X uses reserved fields\n");
19143 				return -EINVAL;
19144 			}
19145 		} else if (insn->src_reg != BPF_REG_0 || insn->dst_reg != BPF_REG_0 ||
19146 			   (class == BPF_JMP && insn->imm != 0) ||
19147 			   (class == BPF_JMP32 && insn->off != 0)) {
19148 			verbose(env, "BPF_JA uses reserved fields\n");
19149 			return -EINVAL;
19150 		}
19151 		return 0;
19152 	case BPF_EXIT:
19153 		if (BPF_SRC(insn->code) != BPF_K || insn->imm != 0 ||
19154 		    insn->src_reg != BPF_REG_0 || insn->dst_reg != BPF_REG_0 ||
19155 		    class == BPF_JMP32) {
19156 			verbose(env, "BPF_EXIT uses reserved fields\n");
19157 			return -EINVAL;
19158 		}
19159 		return 0;
19160 	case BPF_JCOND:
19161 		if (insn->code != (BPF_JMP | BPF_JCOND) || insn->src_reg != BPF_MAY_GOTO ||
19162 		    insn->dst_reg || insn->imm) {
19163 			verbose(env, "invalid may_goto imm %d\n", insn->imm);
19164 			return -EINVAL;
19165 		}
19166 		return 0;
19167 	default:
19168 		if (BPF_SRC(insn->code) == BPF_X) {
19169 			if (insn->imm != 0) {
19170 				verbose(env, "BPF_JMP/JMP32 uses reserved fields\n");
19171 				return -EINVAL;
19172 			}
19173 		} else if (insn->src_reg != BPF_REG_0) {
19174 			verbose(env, "BPF_JMP/JMP32 uses reserved fields\n");
19175 			return -EINVAL;
19176 		}
19177 		return 0;
19178 	}
19179 }
19180 
19181 static int check_insn_fields(struct bpf_verifier_env *env, struct bpf_insn *insn)
19182 {
19183 	switch (BPF_CLASS(insn->code)) {
19184 	case BPF_ALU:
19185 	case BPF_ALU64:
19186 		return check_alu_fields(env, insn);
19187 	case BPF_LDX:
19188 		if ((BPF_MODE(insn->code) != BPF_MEM && BPF_MODE(insn->code) != BPF_MEMSX) ||
19189 		    insn->imm != 0) {
19190 			verbose(env, "BPF_LDX uses reserved fields\n");
19191 			return -EINVAL;
19192 		}
19193 		return 0;
19194 	case BPF_STX:
19195 		if (BPF_MODE(insn->code) == BPF_ATOMIC)
19196 			return 0;
19197 		if (BPF_MODE(insn->code) != BPF_MEM || insn->imm != 0) {
19198 			verbose(env, "BPF_STX uses reserved fields\n");
19199 			return -EINVAL;
19200 		}
19201 		return 0;
19202 	case BPF_ST:
19203 		if (BPF_MODE(insn->code) != BPF_MEM || insn->src_reg != BPF_REG_0) {
19204 			verbose(env, "BPF_ST uses reserved fields\n");
19205 			return -EINVAL;
19206 		}
19207 		return 0;
19208 	case BPF_JMP:
19209 	case BPF_JMP32:
19210 		return check_jmp_fields(env, insn);
19211 	case BPF_LD: {
19212 		u8 mode = BPF_MODE(insn->code);
19213 
19214 		if (mode == BPF_ABS || mode == BPF_IND) {
19215 			if (insn->dst_reg != BPF_REG_0 || insn->off != 0 ||
19216 			    BPF_SIZE(insn->code) == BPF_DW ||
19217 			    (mode == BPF_ABS && insn->src_reg != BPF_REG_0)) {
19218 				verbose(env, "BPF_LD_[ABS|IND] uses reserved fields\n");
19219 				return -EINVAL;
19220 			}
19221 		} else if (mode != BPF_IMM) {
19222 			verbose(env, "invalid BPF_LD mode\n");
19223 			return -EINVAL;
19224 		}
19225 		return 0;
19226 	}
19227 	default:
19228 		verbose(env, "unknown insn class %d\n", BPF_CLASS(insn->code));
19229 		return -EINVAL;
19230 	}
19231 }
19232 
19233 /*
19234  * Check that insns are sane and rewrite pseudo imm in ld_imm64 instructions:
19235  *
19236  * 1. if it accesses map FD, replace it with actual map pointer.
19237  * 2. if it accesses btf_id of a VAR, replace it with pointer to the var.
19238  *
19239  * NOTE: btf_vmlinux is required for converting pseudo btf_id.
19240  */
19241 static int check_and_resolve_insns(struct bpf_verifier_env *env)
19242 {
19243 	struct bpf_insn *insn = env->prog->insnsi;
19244 	int insn_cnt = env->prog->len;
19245 	int i, err;
19246 
19247 	err = bpf_prog_calc_tag(env->prog);
19248 	if (err)
19249 		return err;
19250 
19251 	for (i = 0; i < insn_cnt; i++, insn++) {
19252 		if (insn->dst_reg >= MAX_BPF_REG &&
19253 		    !is_stack_arg_st(insn) && !is_stack_arg_stx(insn)) {
19254 			verbose(env, "R%d is invalid\n", insn->dst_reg);
19255 			return -EINVAL;
19256 		}
19257 		if (insn->src_reg >= MAX_BPF_REG && !is_stack_arg_ldx(insn)) {
19258 			verbose(env, "R%d is invalid\n", insn->src_reg);
19259 			return -EINVAL;
19260 		}
19261 		if (insn[0].code == (BPF_LD | BPF_IMM | BPF_DW)) {
19262 			struct bpf_insn_aux_data *aux;
19263 			struct bpf_map *map;
19264 			int map_idx;
19265 			u64 addr;
19266 
19267 			if (i == insn_cnt - 1 || insn[1].code != 0 ||
19268 			    insn[1].dst_reg != 0 || insn[1].src_reg != 0 ||
19269 			    insn[1].off != 0) {
19270 				verbose(env, "invalid bpf_ld_imm64 insn\n");
19271 				return -EINVAL;
19272 			}
19273 
19274 			if (insn[0].off != 0) {
19275 				verbose(env, "BPF_LD_IMM64 uses reserved fields\n");
19276 				return -EINVAL;
19277 			}
19278 
19279 			if (insn[0].src_reg == 0)
19280 				/* valid generic load 64-bit imm */
19281 				goto next_insn;
19282 
19283 			if (insn[0].src_reg == BPF_PSEUDO_BTF_ID) {
19284 				aux = &env->insn_aux_data[i];
19285 				err = check_pseudo_btf_id(env, insn, aux);
19286 				if (err)
19287 					return err;
19288 				goto next_insn;
19289 			}
19290 
19291 			if (insn[0].src_reg == BPF_PSEUDO_FUNC) {
19292 				aux = &env->insn_aux_data[i];
19293 				aux->ptr_type = PTR_TO_FUNC;
19294 				goto next_insn;
19295 			}
19296 
19297 			/* In final convert_pseudo_ld_imm64() step, this is
19298 			 * converted into regular 64-bit imm load insn.
19299 			 */
19300 			switch (insn[0].src_reg) {
19301 			case BPF_PSEUDO_MAP_VALUE:
19302 			case BPF_PSEUDO_MAP_IDX_VALUE:
19303 				break;
19304 			case BPF_PSEUDO_MAP_FD:
19305 			case BPF_PSEUDO_MAP_IDX:
19306 				if (insn[1].imm == 0)
19307 					break;
19308 				fallthrough;
19309 			default:
19310 				verbose(env, "unrecognized bpf_ld_imm64 insn\n");
19311 				return -EINVAL;
19312 			}
19313 
19314 			switch (insn[0].src_reg) {
19315 			case BPF_PSEUDO_MAP_IDX_VALUE:
19316 			case BPF_PSEUDO_MAP_IDX:
19317 				map_idx = fd_array_get_map_idx(env, insn[0].imm);
19318 				break;
19319 			default:
19320 				if (env->signature) {
19321 					verbose(env, "signed program cannot reference a map by fd, only via fd_array index\n");
19322 					return -EINVAL;
19323 				}
19324 				map_idx = add_used_map(env, insn[0].imm);
19325 				break;
19326 			}
19327 
19328 			if (map_idx < 0)
19329 				return map_idx;
19330 			map = env->used_maps[map_idx];
19331 
19332 			aux = &env->insn_aux_data[i];
19333 			aux->map_index = map_idx;
19334 
19335 			if (insn[0].src_reg == BPF_PSEUDO_MAP_FD ||
19336 			    insn[0].src_reg == BPF_PSEUDO_MAP_IDX) {
19337 				addr = (unsigned long)map;
19338 			} else {
19339 				u32 off = insn[1].imm;
19340 
19341 				if (!map->ops->map_direct_value_addr) {
19342 					verbose(env, "no direct value access support for this map type\n");
19343 					return -EINVAL;
19344 				}
19345 
19346 				err = map->ops->map_direct_value_addr(map, &addr, off);
19347 				if (err) {
19348 					verbose(env, "invalid access to map value pointer, value_size=%u off=%u\n",
19349 						map->value_size, off);
19350 					return err;
19351 				}
19352 
19353 				aux->map_off = off;
19354 				addr += off;
19355 			}
19356 
19357 			insn[0].imm = (u32)addr;
19358 			insn[1].imm = addr >> 32;
19359 
19360 next_insn:
19361 			insn++;
19362 			i++;
19363 			continue;
19364 		}
19365 
19366 		/* Basic sanity check before we invest more work here. */
19367 		if (!bpf_opcode_in_insntable(insn->code)) {
19368 			verbose(env, "unknown opcode %02x\n", insn->code);
19369 			return -EINVAL;
19370 		}
19371 
19372 		err = check_insn_fields(env, insn);
19373 		if (err)
19374 			return err;
19375 	}
19376 
19377 	/* now all pseudo BPF_LD_IMM64 instructions load valid
19378 	 * 'struct bpf_map *' into a register instead of user map_fd.
19379 	 * These pointers will be used later by verifier to validate map access.
19380 	 */
19381 	return 0;
19382 }
19383 
19384 /* drop refcnt of maps used by the rejected program */
19385 static void release_maps(struct bpf_verifier_env *env)
19386 {
19387 	__bpf_free_used_maps(env->prog->aux, env->used_maps,
19388 			     env->used_map_cnt);
19389 }
19390 
19391 /* drop refcnt of maps used by the rejected program */
19392 static void release_btfs(struct bpf_verifier_env *env)
19393 {
19394 	__bpf_free_used_btfs(env->used_btfs, env->used_btf_cnt);
19395 }
19396 
19397 /* convert pseudo BPF_LD_IMM64 into generic BPF_LD_IMM64 */
19398 static void convert_pseudo_ld_imm64(struct bpf_verifier_env *env)
19399 {
19400 	struct bpf_insn *insn = env->prog->insnsi;
19401 	int insn_cnt = env->prog->len;
19402 	int i;
19403 
19404 	for (i = 0; i < insn_cnt; i++, insn++) {
19405 		if (insn->code != (BPF_LD | BPF_IMM | BPF_DW))
19406 			continue;
19407 		if (insn->src_reg == BPF_PSEUDO_FUNC)
19408 			continue;
19409 		insn->src_reg = 0;
19410 	}
19411 }
19412 
19413 static void release_insn_arrays(struct bpf_verifier_env *env)
19414 {
19415 	int i;
19416 
19417 	for (i = 0; i < env->insn_array_map_cnt; i++)
19418 		bpf_insn_array_release(env->insn_array_maps[i]);
19419 }
19420 
19421 /* The verifier does more data flow analysis than llvm and will not
19422  * explore branches that are dead at run time. Malicious programs can
19423  * have dead code too. Therefore replace all dead at-run-time code
19424  * with 'ja -1'.
19425  *
19426  * Just nops are not optimal, e.g. if they would sit at the end of the
19427  * program and through another bug we would manage to jump there, then
19428  * we'd execute beyond program memory otherwise. Returning exception
19429  * code also wouldn't work since we can have subprogs where the dead
19430  * code could be located.
19431  */
19432 static void sanitize_dead_code(struct bpf_verifier_env *env)
19433 {
19434 	struct bpf_insn_aux_data *aux_data = env->insn_aux_data;
19435 	struct bpf_insn trap = BPF_JMP_IMM(BPF_JA, 0, 0, -1);
19436 	struct bpf_insn *insn = env->prog->insnsi;
19437 	const int insn_cnt = env->prog->len;
19438 	int i;
19439 
19440 	for (i = 0; i < insn_cnt; i++) {
19441 		if (aux_data[i].seen)
19442 			continue;
19443 		memcpy(insn + i, &trap, sizeof(trap));
19444 		aux_data[i].zext_dst = false;
19445 	}
19446 }
19447 
19448 static void free_states(struct bpf_verifier_env *env)
19449 {
19450 	struct bpf_verifier_state_list *sl;
19451 	struct list_head *head, *pos, *tmp;
19452 	struct bpf_scc_info *info;
19453 	int i, j;
19454 
19455 	bpf_free_verifier_state(env->cur_state, true);
19456 	env->cur_state = NULL;
19457 	while (!pop_stack(env, NULL, NULL, false));
19458 
19459 	list_for_each_safe(pos, tmp, &env->free_list) {
19460 		sl = container_of(pos, struct bpf_verifier_state_list, node);
19461 		bpf_free_verifier_state(&sl->state, false);
19462 		kfree(sl);
19463 	}
19464 	INIT_LIST_HEAD(&env->free_list);
19465 
19466 	for (i = 0; i < env->scc_cnt; ++i) {
19467 		info = env->scc_info[i];
19468 		if (!info)
19469 			continue;
19470 		for (j = 0; j < info->num_visits; j++)
19471 			bpf_free_backedges(&info->visits[j]);
19472 		kvfree(info);
19473 		env->scc_info[i] = NULL;
19474 	}
19475 
19476 	if (!env->explored_states)
19477 		return;
19478 
19479 	for (i = 0; i < state_htab_size(env); i++) {
19480 		head = &env->explored_states[i];
19481 
19482 		list_for_each_safe(pos, tmp, head) {
19483 			sl = container_of(pos, struct bpf_verifier_state_list, node);
19484 			bpf_free_verifier_state(&sl->state, false);
19485 			kfree(sl);
19486 		}
19487 		INIT_LIST_HEAD(&env->explored_states[i]);
19488 	}
19489 }
19490 
19491 static int do_check_common(struct bpf_verifier_env *env, int subprog, bool is_sleepable)
19492 {
19493 	bool pop_log = !(env->log.level & BPF_LOG_LEVEL2);
19494 	struct bpf_subprog_info *sub = subprog_info(env, subprog);
19495 	struct bpf_prog_aux *aux = env->prog->aux;
19496 	struct bpf_verifier_state *state;
19497 	struct bpf_reg_state *regs;
19498 	u32 old_insns_total = sub->insns_total;
19499 	u32 insn_processed = env->insn_processed;
19500 	int ret, i;
19501 
19502 	env->prev_linfo = NULL;
19503 	env->pass_cnt++;
19504 
19505 	state = kzalloc_obj(struct bpf_verifier_state, GFP_KERNEL_ACCOUNT);
19506 	if (!state)
19507 		return -ENOMEM;
19508 	state->curframe = 0;
19509 	state->speculative = false;
19510 	state->branches = 1;
19511 	state->in_sleepable = is_sleepable;
19512 	state->frame[0] = kzalloc_obj(struct bpf_func_state, GFP_KERNEL_ACCOUNT);
19513 	if (!state->frame[0]) {
19514 		kfree(state);
19515 		return -ENOMEM;
19516 	}
19517 	env->cur_state = state;
19518 	init_func_state(env, state->frame[0],
19519 			BPF_MAIN_FUNC /* callsite */,
19520 			0 /* frameno */,
19521 			subprog);
19522 	state->first_insn_idx = env->subprog_info[subprog].start;
19523 	state->last_insn_idx = -1;
19524 
19525 	regs = state->frame[state->curframe]->regs;
19526 	if (subprog || env->prog->type == BPF_PROG_TYPE_EXT) {
19527 		const char *sub_name = bpf_subprog_name(env, subprog);
19528 		struct bpf_subprog_arg_info *arg;
19529 		struct bpf_reg_state *reg;
19530 
19531 		if (env->log.level & BPF_LOG_LEVEL)
19532 			verbose(env, "Validating %s() func#%d...\n", sub_name, subprog);
19533 		ret = btf_prepare_func_args(env, subprog);
19534 		if (ret)
19535 			goto out;
19536 
19537 		if (subprog_is_exc_cb(env, subprog)) {
19538 			state->frame[0]->in_exception_callback_fn = true;
19539 
19540 			/*
19541 			 * Global functions are scalar or void, make sure
19542 			 * we return a scalar.
19543 			 */
19544 			if (subprog_returns_void(env, subprog)) {
19545 				verbose(env, "exception cb cannot return void\n");
19546 				ret = -EINVAL;
19547 				goto out;
19548 			}
19549 
19550 			/* Also ensure the callback only has a single scalar argument. */
19551 			if (sub->arg_cnt != 1 || sub->args[0].arg_type != ARG_ANYTHING) {
19552 				verbose(env, "exception cb only supports single integer argument\n");
19553 				ret = -EINVAL;
19554 				goto out;
19555 			}
19556 		}
19557 		for (i = BPF_REG_1; i <= min_t(u32, sub->arg_cnt, MAX_BPF_FUNC_REG_ARGS); i++) {
19558 			arg = &sub->args[i - BPF_REG_1];
19559 			reg = &regs[i];
19560 
19561 			if (arg->arg_type == ARG_PTR_TO_CTX) {
19562 				reg->type = PTR_TO_CTX;
19563 				mark_reg_known_zero(env, regs, i);
19564 			} else if (arg->arg_type == ARG_ANYTHING) {
19565 				reg->type = SCALAR_VALUE;
19566 				mark_reg_unknown(env, regs, i);
19567 			} else if (arg->arg_type == ARG_PTR_TO_DYNPTR) {
19568 				/* assume unspecial LOCAL dynptr type */
19569 				__mark_dynptr_reg(reg, BPF_DYNPTR_TYPE_LOCAL, true, ++env->id_gen, 0);
19570 			} else if (base_type(arg->arg_type) == ARG_PTR_TO_MEM) {
19571 				reg->type = PTR_TO_MEM;
19572 				reg->type |= arg->arg_type &
19573 					     (PTR_MAYBE_NULL | PTR_UNTRUSTED | MEM_RDONLY);
19574 				mark_reg_known_zero(env, regs, i);
19575 				reg->mem_size = arg->mem_size;
19576 				if (arg->arg_type & PTR_MAYBE_NULL)
19577 					reg->id = ++env->id_gen;
19578 			} else if (base_type(arg->arg_type) == ARG_PTR_TO_BTF_ID) {
19579 				reg->type = PTR_TO_BTF_ID;
19580 				if (arg->arg_type & PTR_MAYBE_NULL)
19581 					reg->type |= PTR_MAYBE_NULL;
19582 				if (arg->arg_type & PTR_UNTRUSTED)
19583 					reg->type |= PTR_UNTRUSTED;
19584 				if (arg->arg_type & PTR_TRUSTED)
19585 					reg->type |= PTR_TRUSTED;
19586 				mark_reg_known_zero(env, regs, i);
19587 				reg->btf = bpf_get_btf_vmlinux(); /* can't fail at this point */
19588 				reg->btf_id = arg->btf_id;
19589 				reg->id = ++env->id_gen;
19590 			} else if (base_type(arg->arg_type) == ARG_PTR_TO_ARENA) {
19591 				/* caller can pass either PTR_TO_ARENA or SCALAR */
19592 				mark_reg_unknown(env, regs, i);
19593 			} else {
19594 				verifier_bug(env, "unhandled arg#%d type %d",
19595 					     i - BPF_REG_1 + 1, arg->arg_type);
19596 				ret = -EFAULT;
19597 				goto out;
19598 			}
19599 		}
19600 		if (env->prog->type == BPF_PROG_TYPE_EXT && sub->arg_cnt > MAX_BPF_FUNC_REG_ARGS) {
19601 			verbose(env, "freplace programs with >%d args not supported yet\n",
19602 				MAX_BPF_FUNC_REG_ARGS);
19603 			ret = -EINVAL;
19604 			goto out;
19605 		}
19606 	} else {
19607 		/* if main BPF program has associated BTF info, validate that
19608 		 * it's matching expected signature, and otherwise mark BTF
19609 		 * info for main program as unreliable
19610 		 */
19611 		if (env->prog->aux->func_info_aux) {
19612 			ret = btf_prepare_func_args(env, 0);
19613 			if (ret || sub->arg_cnt != 1 || sub->args[0].arg_type != ARG_PTR_TO_CTX) {
19614 				env->prog->aux->func_info_aux[0].unreliable = true;
19615 				sub->arg_cnt = 1;
19616 				sub->stack_arg_cnt = 0;
19617 			}
19618 		}
19619 
19620 		/* 1st arg to a function */
19621 		regs[BPF_REG_1].type = PTR_TO_CTX;
19622 		mark_reg_known_zero(env, regs, BPF_REG_1);
19623 	}
19624 
19625 	/* Acquire references for struct_ops program arguments tagged with "__ref" */
19626 	if (!subprog && env->prog->type == BPF_PROG_TYPE_STRUCT_OPS) {
19627 		for (i = 0; i < aux->ctx_arg_info_size; i++) {
19628 			ret = aux->ctx_arg_info[i].refcounted ? acquire_reference(env, 0, 0) : 0;
19629 			if (ret < 0)
19630 				goto out;
19631 
19632 			aux->ctx_arg_info[i].ref_id = ret;
19633 		}
19634 	}
19635 
19636 	ret = do_check(env);
19637 out:
19638 	account_current_path(env);
19639 	if (!ret) {
19640 		if (pop_log)
19641 			bpf_vlog_reset(&env->log, 0);
19642 		bpf_diag_event_log_restore(env, 0);
19643 	}
19644 	free_states(env);
19645 
19646 	/*
19647 	 * The override is needed to account for async subprograms, which
19648 	 * are verified with their own set of stack frames and thus are
19649 	 * not accounted as callees by account_current_path().
19650 	 * Accumulate their total counts as total counts of the main or
19651 	 * global subprog hosting the async call.
19652 	 * Start from the saved total of earlier contexts: adding to the current
19653 	 * total would count this pass's synchronous paths twice.
19654 	 */
19655 	sub->insns_total = old_insns_total + (env->insn_processed - insn_processed);
19656 	return ret;
19657 }
19658 
19659 /* Lazily verify all global functions based on their BTF, if they are called
19660  * from main BPF program or any of subprograms transitively.
19661  * BPF global subprogs called from dead code are not validated.
19662  * All callable global functions must pass verification.
19663  * Otherwise the whole program is rejected.
19664  * Consider:
19665  * int bar(int);
19666  * int foo(int f)
19667  * {
19668  *    return bar(f);
19669  * }
19670  * int bar(int b)
19671  * {
19672  *    ...
19673  * }
19674  * foo() will be verified first for R1=any_scalar_value. During verification it
19675  * will be assumed that bar() already verified successfully and call to bar()
19676  * from foo() will be checked for type match only. Later bar() will be verified
19677  * independently to check that it's safe for R1=any_scalar_value.
19678  */
19679 static int do_check_subprogs(struct bpf_verifier_env *env)
19680 {
19681 	struct bpf_prog_aux *aux = env->prog->aux;
19682 	struct bpf_func_info_aux *sub_aux;
19683 	int context, i, ret, new_cnt;
19684 
19685 	if (!aux->func_info)
19686 		return 0;
19687 
19688 	/*
19689 	 * Callbacks cannot throw, so the exception callback always runs in the
19690 	 * main program's context. It is presumed to be always called.
19691 	 */
19692 	if (env->exception_callback_subprog) {
19693 		sub_aux = subprog_aux(env, env->exception_callback_subprog);
19694 		sub_aux->called[env->prog->sleepable] = true;
19695 	}
19696 
19697 again:
19698 	new_cnt = 0;
19699 	for (i = 1; i < env->subprog_cnt; i++) {
19700 		if (!bpf_subprog_is_global(env, i))
19701 			continue;
19702 
19703 		sub_aux = subprog_aux(env, i);
19704 		for (context = 0; context < ARRAY_SIZE(sub_aux->called); context++) {
19705 			if (!sub_aux->called[context] || sub_aux->verified[context])
19706 				continue;
19707 
19708 			env->insn_idx = env->subprog_info[i].start;
19709 			WARN_ON_ONCE(env->insn_idx == 0);
19710 			ret = do_check_common(env, i, context);
19711 			if (ret)
19712 				return ret;
19713 			if (env->log.level & BPF_LOG_LEVEL)
19714 				verbose(env, "Func#%d ('%s') is safe for any args "
19715 					"that match its prototype\n",
19716 					i, bpf_subprog_name(env, i));
19717 
19718 			sub_aux->verified[context] = true;
19719 			new_cnt++;
19720 		}
19721 	}
19722 
19723 	/*
19724 	 * We can't loop forever as each pass verifies at least one new context,
19725 	 * and there are only two contexts per global subprog.
19726 	 */
19727 	if (new_cnt)
19728 		goto again;
19729 
19730 	return 0;
19731 }
19732 
19733 static int do_check_main(struct bpf_verifier_env *env)
19734 {
19735 	int ret;
19736 
19737 	env->insn_idx = 0;
19738 	ret = do_check_common(env, 0, env->prog->sleepable);
19739 	if (!ret)
19740 		env->prog->aux->stack_depth = env->subprog_info[0].stack_depth;
19741 	return ret;
19742 }
19743 
19744 static void print_verification_stats(struct bpf_verifier_env *env)
19745 {
19746 	/* Skip over hidden subprogs which are not verified. */
19747 	int i, subprog_cnt = env->subprog_cnt - env->hidden_subprog_cnt;
19748 
19749 	if (env->log.level & BPF_LOG_STATS) {
19750 		verbose(env, "verification time %lld usec\n",
19751 			div_u64(env->verification_time, 1000));
19752 		verbose(env, "stack depth max %d\n", env->max_stack_depth);
19753 		for (i = 0; i < subprog_cnt; i++) {
19754 			const char *name = env->subprog_info[i].name;
19755 			const char *kind;
19756 
19757 			if (!name || !name[0])
19758 				name = "<unknown>";
19759 			kind = i == 0 ? "main" :
19760 			       bpf_subprog_is_global(env, i) ? "global" : "static";
19761 			verbose(env, "subprog %d (%s) %s insns_self %d insns_total %d stack %d\n",
19762 				i, name, kind, env->subprog_info[i].insns_self,
19763 				env->subprog_info[i].insns_total,
19764 				env->subprog_info[i].stack_depth);
19765 		}
19766 	}
19767 	verbose(env, "processed %d insns (limit %d) max_states_per_insn %d "
19768 		"total_states %d peak_states %d mark_read %d\n",
19769 		env->insn_processed, BPF_COMPLEXITY_LIMIT_INSNS,
19770 		env->max_states_per_insn, env->total_states,
19771 		env->peak_states, env->longest_mark_read_walk);
19772 }
19773 
19774 int bpf_prog_ctx_arg_info_init(struct bpf_prog *prog,
19775 			       const struct bpf_ctx_arg_aux *info, u32 cnt)
19776 {
19777 	prog->aux->ctx_arg_info = kmemdup_array(info, cnt, sizeof(*info), GFP_KERNEL_ACCOUNT);
19778 	prog->aux->ctx_arg_info_size = cnt;
19779 
19780 	return prog->aux->ctx_arg_info ? 0 : -ENOMEM;
19781 }
19782 
19783 static int check_struct_ops_btf_id(struct bpf_verifier_env *env)
19784 {
19785 	const struct btf_type *t, *func_proto;
19786 	const struct bpf_struct_ops_desc *st_ops_desc;
19787 	const struct bpf_struct_ops_arg_info *arg_info;
19788 	const struct bpf_struct_ops *st_ops;
19789 	const struct btf_member *member;
19790 	struct bpf_prog *prog = env->prog;
19791 	bool has_refcounted_arg = false;
19792 	u32 btf_id, member_idx, member_off;
19793 	struct btf *btf;
19794 	const char *mname;
19795 	int i, err;
19796 
19797 	if (!prog->gpl_compatible) {
19798 		verbose(env, "struct ops programs must have a GPL compatible license\n");
19799 		return -EINVAL;
19800 	}
19801 
19802 	if (!prog->aux->attach_btf_id)
19803 		return -ENOTSUPP;
19804 
19805 	btf = prog->aux->attach_btf;
19806 	if (btf_is_module(btf)) {
19807 		/* Make sure st_ops is valid through the lifetime of env */
19808 		env->attach_btf_mod = btf_try_get_module(btf);
19809 		if (!env->attach_btf_mod) {
19810 			verbose(env, "struct_ops module %s is not found\n",
19811 				btf_get_name(btf));
19812 			return -ENOTSUPP;
19813 		}
19814 	}
19815 
19816 	btf_id = prog->aux->attach_btf_id;
19817 	st_ops_desc = bpf_struct_ops_find(btf, btf_id);
19818 	if (!st_ops_desc) {
19819 		verbose(env, "attach_btf_id %u is not a supported struct\n",
19820 			btf_id);
19821 		return -ENOTSUPP;
19822 	}
19823 	st_ops = st_ops_desc->st_ops;
19824 
19825 	t = st_ops_desc->type;
19826 	member_idx = prog->expected_attach_type;
19827 	if (member_idx >= btf_type_vlen(t)) {
19828 		verbose(env, "attach to invalid member idx %u of struct %s\n",
19829 			member_idx, st_ops->name);
19830 		return -EINVAL;
19831 	}
19832 
19833 	member = &btf_type_member(t)[member_idx];
19834 	mname = btf_name_by_offset(btf, member->name_off);
19835 	func_proto = btf_type_resolve_func_ptr(btf, member->type,
19836 					       NULL);
19837 	if (!func_proto) {
19838 		verbose(env, "attach to invalid member %s(@idx %u) of struct %s\n",
19839 			mname, member_idx, st_ops->name);
19840 		return -EINVAL;
19841 	}
19842 
19843 	member_off = __btf_member_bit_offset(t, member) / 8;
19844 	err = bpf_struct_ops_supported(st_ops, member_off);
19845 	if (err) {
19846 		verbose(env, "attach to unsupported member %s of struct %s\n",
19847 			mname, st_ops->name);
19848 		return err;
19849 	}
19850 
19851 	if (st_ops->check_member) {
19852 		err = st_ops->check_member(t, member, prog);
19853 
19854 		if (err) {
19855 			verbose(env, "attach to unsupported member %s of struct %s\n",
19856 				mname, st_ops->name);
19857 			return err;
19858 		}
19859 	}
19860 
19861 	if (prog->aux->priv_stack_requested && !bpf_jit_supports_private_stack()) {
19862 		verbose(env, "Private stack not supported by jit\n");
19863 		return -EACCES;
19864 	}
19865 
19866 	arg_info = &st_ops_desc->arg_info[member_idx];
19867 	for (i = 0; i < arg_info->cnt; i++) {
19868 		const struct bpf_ctx_arg_aux *info = &arg_info->info[i];
19869 
19870 		if (info->refcounted)
19871 			has_refcounted_arg = true;
19872 		if (base_type(info->reg_type) == PTR_TO_ARENA) {
19873 			if (!bpf_jit_supports_arena_args()) {
19874 				verbose(env, "JIT does not support arena arguments\n");
19875 				return -ENOTSUPP;
19876 			}
19877 			if (!prog->aux->arena) {
19878 				verbose(env,
19879 					"arena argument of %s requires a program with an associated arena\n",
19880 					mname);
19881 				return -EINVAL;
19882 			}
19883 		}
19884 	}
19885 
19886 	/* Tail call is not allowed for programs with refcounted arguments since we
19887 	 * cannot guarantee that valid refcounted kptrs will be passed to the callee.
19888 	 */
19889 	for (i = 0; i < env->subprog_cnt; i++) {
19890 		if (has_refcounted_arg && env->subprog_info[i].has_tail_call) {
19891 			verbose(env, "program with __ref argument cannot tail call\n");
19892 			return -EINVAL;
19893 		}
19894 	}
19895 
19896 	prog->aux->st_ops = st_ops;
19897 	prog->aux->attach_st_ops_member_off = member_off;
19898 
19899 	prog->aux->attach_func_proto = func_proto;
19900 	prog->aux->attach_func_name = mname;
19901 	env->ops = st_ops->verifier_ops;
19902 
19903 	return bpf_prog_ctx_arg_info_init(prog, arg_info->info, arg_info->cnt);
19904 }
19905 #define SECURITY_PREFIX "security_"
19906 
19907 #ifdef CONFIG_FUNCTION_ERROR_INJECTION
19908 
19909 /* list of non-sleepable functions that are otherwise on
19910  * ALLOW_ERROR_INJECTION list
19911  */
19912 BTF_SET_START(btf_non_sleepable_error_inject)
19913 /* Three functions below can be called from sleepable and non-sleepable context.
19914  * Assume non-sleepable from bpf safety point of view.
19915  */
19916 BTF_ID(func, __filemap_add_folio)
19917 #ifdef CONFIG_FAIL_PAGE_ALLOC
19918 BTF_ID(func, should_fail_alloc_page)
19919 #endif
19920 #ifdef CONFIG_FAILSLAB
19921 BTF_ID(func, should_failslab)
19922 #endif
19923 BTF_SET_END(btf_non_sleepable_error_inject)
19924 
19925 static int check_non_sleepable_error_inject(u32 btf_id)
19926 {
19927 	return btf_id_set_contains(&btf_non_sleepable_error_inject, btf_id);
19928 }
19929 
19930 static int check_attach_sleepable(u32 btf_id, unsigned long addr, const char *func_name)
19931 {
19932 	/* fentry/fexit/fmod_ret progs can be sleepable if they are
19933 	 * attached to ALLOW_ERROR_INJECTION and are not in denylist.
19934 	 */
19935 	if (!check_non_sleepable_error_inject(btf_id) &&
19936 	    within_error_injection_list(addr))
19937 		return 0;
19938 
19939 	return -EINVAL;
19940 }
19941 
19942 static int check_attach_modify_return(unsigned long addr, const char *func_name)
19943 {
19944 	if (within_error_injection_list(addr) ||
19945 	    !strncmp(SECURITY_PREFIX, func_name, sizeof(SECURITY_PREFIX) - 1))
19946 		return 0;
19947 
19948 	return -EINVAL;
19949 }
19950 
19951 #else
19952 
19953 /* Unfortunately, the arch-specific prefixes are hard-coded in arch syscall code
19954  * so we need to hard-code them, too. Ftrace has arch_syscall_match_sym_name()
19955  * but that just compares two concrete function names.
19956  */
19957 static bool has_arch_syscall_prefix(const char *func_name)
19958 {
19959 #if defined(__x86_64__)
19960 	return !strncmp(func_name, "__x64_", 6);
19961 #elif defined(__i386__)
19962 	return !strncmp(func_name, "__ia32_", 7);
19963 #elif defined(__s390x__)
19964 	return !strncmp(func_name, "__s390x_", 8);
19965 #elif defined(__aarch64__)
19966 	return !strncmp(func_name, "__arm64_", 8);
19967 #elif defined(__riscv)
19968 	return !strncmp(func_name, "__riscv_", 8);
19969 #elif defined(__powerpc__) || defined(__powerpc64__)
19970 	return !strncmp(func_name, "sys_", 4);
19971 #elif defined(__loongarch__)
19972 	return !strncmp(func_name, "sys_", 4);
19973 #else
19974 	return false;
19975 #endif
19976 }
19977 
19978 /* Without error injection, allow sleepable and fmod_ret progs on syscalls. */
19979 
19980 static int check_attach_sleepable(u32 btf_id, unsigned long addr, const char *func_name)
19981 {
19982 	if (has_arch_syscall_prefix(func_name))
19983 		return 0;
19984 
19985 	return -EINVAL;
19986 }
19987 
19988 static int check_attach_modify_return(unsigned long addr, const char *func_name)
19989 {
19990 	if (has_arch_syscall_prefix(func_name) ||
19991 	    !strncmp(SECURITY_PREFIX, func_name, sizeof(SECURITY_PREFIX) - 1))
19992 		return 0;
19993 
19994 	return -EINVAL;
19995 }
19996 
19997 #endif /* CONFIG_FUNCTION_ERROR_INJECTION */
19998 
19999 static bool is_tracing_multi_id(const struct bpf_prog *prog, u32 btf_id)
20000 {
20001 	return is_tracing_multi(prog->expected_attach_type) && bpf_multi_func_btf_id[0] == btf_id;
20002 }
20003 
20004 static int btf_id_allow_sleepable(u32 btf_id, unsigned long addr, const struct bpf_prog *prog,
20005 				  const struct btf *btf)
20006 {
20007 	const struct btf_type *t;
20008 	const char *tname;
20009 
20010 	if (!btf_is_kernel(btf))
20011 		return -EINVAL;
20012 
20013 	switch (prog->type) {
20014 	case BPF_PROG_TYPE_TRACING:
20015 		t = btf_type_by_id(btf, btf_id);
20016 		if (!t)
20017 			return -EINVAL;
20018 		tname = btf_name_by_offset(btf, t->name_off);
20019 		if (!tname)
20020 			return -EINVAL;
20021 
20022 		/*
20023 		 * *.multi sleepable programs will pass initial sleepable check,
20024 		 * the actual attached btf ids are checked later during the link
20025 		 * attachment.
20026 		 */
20027 		if (is_tracing_multi_id(prog, btf_id))
20028 			return 0;
20029 		if (!check_attach_sleepable(btf_id, addr, tname))
20030 			return 0;
20031 		/*
20032 		 * fentry/fexit/fmod_ret progs can also be sleepable if they are
20033 		 * in the fmodret id set with the KF_SLEEPABLE flag.
20034 		 */
20035 		else {
20036 			u32 *flags = btf_kfunc_is_modify_return(btf, btf_id, prog);
20037 
20038 			if (flags && (*flags & KF_SLEEPABLE))
20039 				return 0;
20040 		}
20041 		break;
20042 	case BPF_PROG_TYPE_LSM:
20043 		/*
20044 		 * LSM progs check that they are attached to bpf_lsm_*() funcs.
20045 		 * Only some of them are sleepable.
20046 		 */
20047 		if (bpf_lsm_is_sleepable_hook(btf_id))
20048 			return 0;
20049 		break;
20050 	default:
20051 		break;
20052 	}
20053 	return -EINVAL;
20054 }
20055 
20056 /*
20057  * Resolve the prototype describing a trace target's real ABI. A
20058  * KF_IMPLICIT_ARGS kfunc has its injected args stripped from the public
20059  * prototype, so use the _impl prototype; other targets use their own.
20060  */
20061 static const struct btf_type *
20062 btf_attach_func_proto(struct bpf_verifier_log *log, struct btf *btf, u32 func_id)
20063 {
20064 	const struct btf_type *func;
20065 	struct module *mod = NULL;
20066 	const char *name;
20067 	int implicit;
20068 
20069 	func = btf_type_by_id(btf, func_id);
20070 	if (!func || !btf_type_is_func(func))
20071 		return NULL;
20072 	name = btf_name_by_offset(btf, func->name_off);
20073 
20074 	/*
20075 	 * btf_kfunc_check_flag() reads kfunc_set_tab, which for a module is
20076 	 * stable only once it is live; hold a module ref across the read to
20077 	 * exclude a concurrent module load.
20078 	 */
20079 	if (btf_is_module(btf)) {
20080 		mod = btf_try_get_module(btf);
20081 		if (!mod)
20082 			return NULL;
20083 	}
20084 	implicit = btf_kfunc_check_flag(btf, func_id, KF_IMPLICIT_ARGS);
20085 	module_put(mod);
20086 
20087 	if (implicit == -EINVAL) {
20088 		bpf_log(log, "kfunc %s has inconsistent KF_IMPLICIT_ARGS\n", name);
20089 		return NULL;
20090 	}
20091 	if (implicit > 0)
20092 		return find_kfunc_impl_proto(log, btf, name);
20093 
20094 	return btf_type_by_id(btf, func->type);
20095 }
20096 
20097 static bool attach_uses_trampoline_retval(enum bpf_attach_type type)
20098 {
20099 	switch (type) {
20100 	case BPF_MODIFY_RETURN:
20101 	case BPF_TRACE_FEXIT:
20102 	case BPF_TRACE_FEXIT_MULTI:
20103 	case BPF_TRACE_FSESSION:
20104 	case BPF_TRACE_FSESSION_MULTI:
20105 		return true;
20106 	default:
20107 		return false;
20108 	}
20109 }
20110 
20111 int bpf_check_attach_target(struct bpf_verifier_log *log,
20112 			    const struct bpf_prog *prog,
20113 			    const struct bpf_prog *tgt_prog,
20114 			    u32 btf_id,
20115 			    struct bpf_attach_target_info *tgt_info)
20116 {
20117 	bool prog_extension = prog->type == BPF_PROG_TYPE_EXT;
20118 	bool prog_tracing = prog->type == BPF_PROG_TYPE_TRACING;
20119 	char trace_symbol[KSYM_SYMBOL_LEN];
20120 	const char prefix[] = "btf_trace_";
20121 	struct bpf_raw_event_map *btp;
20122 	int ret = 0, subprog = -1, i;
20123 	const struct btf_type *t;
20124 	bool conservative = true;
20125 	const char *tname, *fname;
20126 	struct btf *btf;
20127 	long addr = 0;
20128 	struct module *mod = NULL;
20129 
20130 	if (!btf_id) {
20131 		bpf_log(log, "Tracing programs must provide btf_id\n");
20132 		return -EINVAL;
20133 	}
20134 	btf = tgt_prog ? tgt_prog->aux->btf : prog->aux->attach_btf;
20135 	if (!btf) {
20136 		bpf_log(log,
20137 			"Tracing program can only be attached to another program annotated with BTF\n");
20138 		return -EINVAL;
20139 	}
20140 	t = btf_type_by_id(btf, btf_id);
20141 	if (!t) {
20142 		bpf_log(log, "attach_btf_id %u is invalid\n", btf_id);
20143 		return -EINVAL;
20144 	}
20145 	tname = btf_name_by_offset(btf, t->name_off);
20146 	if (!tname) {
20147 		bpf_log(log, "attach_btf_id %u doesn't have a name\n", btf_id);
20148 		return -EINVAL;
20149 	}
20150 	if (tgt_prog) {
20151 		struct bpf_prog_aux *aux = tgt_prog->aux;
20152 		bool tgt_changes_pkt_data;
20153 		bool tgt_might_sleep;
20154 
20155 		if (bpf_prog_is_dev_bound(prog->aux) &&
20156 		    !bpf_prog_dev_bound_match(prog, tgt_prog)) {
20157 			bpf_log(log, "Target program bound device mismatch");
20158 			return -EINVAL;
20159 		}
20160 
20161 		for (i = 0; i < aux->func_info_cnt; i++)
20162 			if (aux->func_info[i].type_id == btf_id) {
20163 				subprog = i;
20164 				break;
20165 			}
20166 		if (subprog == -1) {
20167 			bpf_log(log, "Subprog %s doesn't exist\n", tname);
20168 			return -EINVAL;
20169 		}
20170 		/*
20171 		 * A struct_ops indirect trampoline converts arena arguments
20172 		 * before invoking its program. A tracing or extension program
20173 		 * attached to the main program would see the converted offset as a
20174 		 * regular BTF pointer.
20175 		 */
20176 		if (subprog == 0 && bpf_prog_has_arena_ctx_arg(tgt_prog)) {
20177 			bpf_log(log, "Cannot attach to a target with arena context arguments\n");
20178 			return -EOPNOTSUPP;
20179 		}
20180 		if (aux->func && aux->func[subprog]->aux->exception_cb) {
20181 			bpf_log(log,
20182 				"%s programs cannot attach to exception callback\n",
20183 				prog_extension ? "Extension" : "Tracing");
20184 			return -EINVAL;
20185 		}
20186 		conservative = aux->func_info_aux[subprog].unreliable;
20187 		if (prog_extension) {
20188 			if (conservative) {
20189 				bpf_log(log,
20190 					"Cannot replace static functions\n");
20191 				return -EINVAL;
20192 			}
20193 			if (!prog->jit_requested) {
20194 				bpf_log(log,
20195 					"Extension programs should be JITed\n");
20196 				return -EINVAL;
20197 			}
20198 			tgt_changes_pkt_data = aux->func
20199 					       ? aux->func[subprog]->aux->changes_pkt_data
20200 					       : aux->changes_pkt_data;
20201 			if (prog->aux->changes_pkt_data && !tgt_changes_pkt_data) {
20202 				bpf_log(log,
20203 					"Extension program changes packet data, while original does not\n");
20204 				return -EINVAL;
20205 			}
20206 
20207 			tgt_might_sleep = aux->func
20208 					  ? aux->func[subprog]->aux->might_sleep
20209 					  : aux->might_sleep;
20210 			if (prog->aux->might_sleep && !tgt_might_sleep) {
20211 				bpf_log(log,
20212 					"Extension program may sleep, while original does not\n");
20213 				return -EINVAL;
20214 			}
20215 		}
20216 		if (!tgt_prog->jited) {
20217 			bpf_log(log, "Can attach to only JITed progs\n");
20218 			return -EINVAL;
20219 		}
20220 		if (prog_tracing) {
20221 			if (aux->attach_tracing_prog) {
20222 				/*
20223 				 * Target program is an fentry/fexit which is already attached
20224 				 * to another tracing program. More levels of nesting
20225 				 * attachment are not allowed.
20226 				 */
20227 				bpf_log(log, "Cannot nest tracing program attach more than once\n");
20228 				return -EINVAL;
20229 			}
20230 		} else if (tgt_prog->type == prog->type) {
20231 			/*
20232 			 * To avoid potential call chain cycles, prevent attaching of a
20233 			 * program extension to another extension. It's ok to attach
20234 			 * fentry/fexit to extension program.
20235 			 */
20236 			bpf_log(log, "Cannot recursively attach\n");
20237 			return -EINVAL;
20238 		}
20239 		if (tgt_prog->type == BPF_PROG_TYPE_TRACING &&
20240 		    prog_extension &&
20241 		    (tgt_prog->expected_attach_type == BPF_TRACE_FENTRY ||
20242 		     tgt_prog->expected_attach_type == BPF_TRACE_FEXIT ||
20243 		     tgt_prog->expected_attach_type == BPF_TRACE_FENTRY_MULTI ||
20244 		     tgt_prog->expected_attach_type == BPF_TRACE_FEXIT_MULTI ||
20245 		     tgt_prog->expected_attach_type == BPF_TRACE_FSESSION ||
20246 		     tgt_prog->expected_attach_type == BPF_TRACE_FSESSION_MULTI)) {
20247 			/* Program extensions can extend all program types
20248 			 * except fentry/fexit. The reason is the following.
20249 			 * The fentry/fexit programs are used for performance
20250 			 * analysis, stats and can be attached to any program
20251 			 * type. When extension program is replacing XDP function
20252 			 * it is necessary to allow performance analysis of all
20253 			 * functions. Both original XDP program and its program
20254 			 * extension. Hence attaching fentry/fexit to
20255 			 * BPF_PROG_TYPE_EXT is allowed. If extending of
20256 			 * fentry/fexit was allowed it would be possible to create
20257 			 * long call chain fentry->extension->fentry->extension
20258 			 * beyond reasonable stack size. Hence extending fentry
20259 			 * is not allowed.
20260 			 */
20261 			bpf_log(log, "Cannot extend fentry/fexit/fsession\n");
20262 			return -EINVAL;
20263 		}
20264 	} else {
20265 		if (prog_extension) {
20266 			bpf_log(log, "Cannot replace kernel functions\n");
20267 			return -EINVAL;
20268 		}
20269 	}
20270 
20271 	switch (prog->expected_attach_type) {
20272 	case BPF_TRACE_RAW_TP:
20273 		if (tgt_prog) {
20274 			bpf_log(log,
20275 				"Only FENTRY/FEXIT/FSESSION progs are attachable to another BPF prog\n");
20276 			return -EINVAL;
20277 		}
20278 		if (!btf_type_is_typedef(t)) {
20279 			bpf_log(log, "attach_btf_id %u is not a typedef\n",
20280 				btf_id);
20281 			return -EINVAL;
20282 		}
20283 		if (strncmp(prefix, tname, sizeof(prefix) - 1)) {
20284 			bpf_log(log, "attach_btf_id %u points to wrong type name %s\n",
20285 				btf_id, tname);
20286 			return -EINVAL;
20287 		}
20288 		tname += sizeof(prefix) - 1;
20289 
20290 		/* The func_proto of "btf_trace_##tname" is generated from typedef without argument
20291 		 * names. Thus using bpf_raw_event_map to get argument names.
20292 		 */
20293 		btp = bpf_get_raw_tracepoint(tname);
20294 		if (!btp)
20295 			return -EINVAL;
20296 		if (prog->sleepable && !tracepoint_is_faultable(btp->tp)) {
20297 			bpf_log(log, "Sleepable program cannot attach to non-faultable tracepoint %s\n",
20298 				tname);
20299 			bpf_put_raw_tracepoint(btp);
20300 			return -EINVAL;
20301 		}
20302 		fname = kallsyms_lookup((unsigned long)btp->bpf_func, NULL, NULL, NULL,
20303 					trace_symbol);
20304 		bpf_put_raw_tracepoint(btp);
20305 
20306 		if (fname)
20307 			ret = btf_find_by_name_kind(btf, fname, BTF_KIND_FUNC);
20308 
20309 		if (!fname || ret < 0) {
20310 			bpf_log(log, "Cannot find btf of tracepoint template, fall back to %s%s.\n",
20311 				prefix, tname);
20312 			t = btf_type_by_id(btf, t->type);
20313 			if (!btf_type_is_ptr(t))
20314 				/* should never happen in valid vmlinux build */
20315 				return -EINVAL;
20316 		} else {
20317 			t = btf_type_by_id(btf, ret);
20318 			if (!btf_type_is_func(t))
20319 				/* should never happen in valid vmlinux build */
20320 				return -EINVAL;
20321 		}
20322 
20323 		t = btf_type_by_id(btf, t->type);
20324 		if (!btf_type_is_func_proto(t))
20325 			/* should never happen in valid vmlinux build */
20326 			return -EINVAL;
20327 
20328 		break;
20329 	case BPF_TRACE_ITER:
20330 		if (!btf_type_is_func(t)) {
20331 			bpf_log(log, "attach_btf_id %u is not a function\n",
20332 				btf_id);
20333 			return -EINVAL;
20334 		}
20335 		t = btf_type_by_id(btf, t->type);
20336 		if (!btf_type_is_func_proto(t))
20337 			return -EINVAL;
20338 		ret = btf_distill_func_proto(log, btf, t, tname, &tgt_info->fmodel);
20339 		if (ret)
20340 			return ret;
20341 		break;
20342 	default:
20343 		if (!prog_extension)
20344 			return -EINVAL;
20345 		fallthrough;
20346 	case BPF_MODIFY_RETURN:
20347 	case BPF_LSM_MAC:
20348 	case BPF_LSM_CGROUP:
20349 	case BPF_TRACE_FENTRY:
20350 	case BPF_TRACE_FEXIT:
20351 	case BPF_TRACE_FSESSION:
20352 	case BPF_TRACE_FSESSION_MULTI:
20353 	case BPF_TRACE_FENTRY_MULTI:
20354 	case BPF_TRACE_FEXIT_MULTI:
20355 		if ((prog->expected_attach_type == BPF_TRACE_FSESSION ||
20356 		    prog->expected_attach_type == BPF_TRACE_FSESSION_MULTI) &&
20357 		    !bpf_jit_supports_fsession()) {
20358 			bpf_log(log, "JIT does not support fsession\n");
20359 			return -EOPNOTSUPP;
20360 		}
20361 		if (!btf_type_is_func(t)) {
20362 			bpf_log(log, "attach_btf_id %u is not a function\n",
20363 				btf_id);
20364 			return -EINVAL;
20365 		}
20366 		if (prog_extension &&
20367 		    btf_check_type_match(log, prog, btf, t))
20368 			return -EINVAL;
20369 		t = btf_attach_func_proto(log, btf, btf_id);
20370 		if (!t || !btf_type_is_func_proto(t))
20371 			return -EINVAL;
20372 
20373 		if ((prog->aux->saved_dst_prog_type || prog->aux->saved_dst_attach_type) &&
20374 		    (!tgt_prog || prog->aux->saved_dst_prog_type != tgt_prog->type ||
20375 		     prog->aux->saved_dst_attach_type != tgt_prog->expected_attach_type))
20376 			return -EINVAL;
20377 
20378 		if (tgt_prog && conservative)
20379 			t = NULL;
20380 
20381 		ret = btf_distill_func_proto(log, btf, t, tname, &tgt_info->fmodel);
20382 		if (ret < 0)
20383 			return ret;
20384 
20385 		if (tgt_info->fmodel.ret_size > 8 &&
20386 		    attach_uses_trampoline_retval(prog->expected_attach_type)) {
20387 			bpf_log(log,
20388 				"Attach to function %s with a >8 byte return value is not supported for this attach type\n",
20389 				tname);
20390 			return -EOPNOTSUPP;
20391 		}
20392 
20393 		/*
20394 		 * *.multi programs don't need an address during program
20395 		 * verification, we just take the module ref if needed.
20396 		 */
20397 		if (is_tracing_multi_id(prog, btf_id)) {
20398 			if (btf_is_module(btf)) {
20399 				mod = btf_try_get_module(btf);
20400 				if (!mod)
20401 					return -ENOENT;
20402 			}
20403 			addr = 0;
20404 		} else if (tgt_prog) {
20405 			if (subprog == 0)
20406 				addr = (long) tgt_prog->bpf_func;
20407 			else
20408 				addr = (long) tgt_prog->aux->func[subprog]->bpf_func;
20409 		} else {
20410 			if (btf_is_module(btf)) {
20411 				mod = btf_try_get_module(btf);
20412 				if (mod)
20413 					addr = find_kallsyms_symbol_value(mod, tname);
20414 				else
20415 					addr = 0;
20416 			} else {
20417 				addr = kallsyms_lookup_name(tname);
20418 			}
20419 			if (!addr) {
20420 				module_put(mod);
20421 				bpf_log(log,
20422 					"The address of function %s cannot be found\n",
20423 					tname);
20424 				return -ENOENT;
20425 			}
20426 		}
20427 
20428 		if (prog->sleepable) {
20429 			ret = btf_id_allow_sleepable(btf_id, addr, prog, btf);
20430 			if (ret) {
20431 				module_put(mod);
20432 				bpf_log(log, "%s is not sleepable\n", tname);
20433 				return ret;
20434 			}
20435 		} else if (prog->expected_attach_type == BPF_MODIFY_RETURN) {
20436 			if (tgt_prog) {
20437 				module_put(mod);
20438 				bpf_log(log, "can't modify return codes of BPF programs\n");
20439 				return -EINVAL;
20440 			}
20441 			ret = -EINVAL;
20442 			if (btf_kfunc_is_modify_return(btf, btf_id, prog) ||
20443 			    !check_attach_modify_return(addr, tname))
20444 				ret = 0;
20445 			if (ret) {
20446 				module_put(mod);
20447 				bpf_log(log, "%s() is not modifiable\n", tname);
20448 				return ret;
20449 			}
20450 		}
20451 
20452 		break;
20453 	}
20454 	tgt_info->tgt_addr = addr;
20455 	tgt_info->tgt_name = tname;
20456 	tgt_info->tgt_type = t;
20457 	tgt_info->tgt_mod = mod;
20458 	return 0;
20459 }
20460 
20461 BTF_SET_START(btf_id_deny)
20462 BTF_ID_UNUSED
20463 #ifdef CONFIG_SMP
20464 BTF_ID(func, ___migrate_enable)
20465 BTF_ID(func, migrate_disable)
20466 BTF_ID(func, migrate_enable)
20467 #endif
20468 #if !defined CONFIG_PREEMPT_RCU && !defined CONFIG_TINY_RCU
20469 BTF_ID(func, rcu_read_unlock_strict)
20470 #endif
20471 #if defined(CONFIG_DEBUG_PREEMPT) || defined(CONFIG_TRACE_PREEMPT_TOGGLE)
20472 BTF_ID(func, preempt_count_add)
20473 BTF_ID(func, preempt_count_sub)
20474 #endif
20475 #ifdef CONFIG_PREEMPT_RCU
20476 BTF_ID(func, __rcu_read_lock)
20477 BTF_ID(func, __rcu_read_unlock)
20478 #endif
20479 BTF_SET_END(btf_id_deny)
20480 
20481 /* fexit and fmod_ret can't be used to attach to __noreturn functions.
20482  * Currently, we must manually list all __noreturn functions here. Once a more
20483  * robust solution is implemented, this workaround can be removed.
20484  */
20485 BTF_SET_START(noreturn_deny)
20486 #ifdef CONFIG_IA32_EMULATION
20487 BTF_ID(func, __ia32_sys_exit)
20488 BTF_ID(func, __ia32_sys_exit_group)
20489 #endif
20490 #ifdef CONFIG_KUNIT
20491 BTF_ID(func, __kunit_abort)
20492 BTF_ID(func, kunit_try_catch_throw)
20493 #endif
20494 #ifdef CONFIG_MODULES
20495 BTF_ID(func, __module_put_and_kthread_exit)
20496 #endif
20497 #ifdef CONFIG_X86_64
20498 BTF_ID(func, __x64_sys_exit)
20499 BTF_ID(func, __x64_sys_exit_group)
20500 #endif
20501 BTF_ID(func, do_exit)
20502 BTF_ID(func, do_group_exit)
20503 BTF_ID(func, kthread_complete_and_exit)
20504 BTF_ID(func, make_task_dead)
20505 BTF_SET_END(noreturn_deny)
20506 
20507 static bool can_be_sleepable(struct bpf_prog *prog)
20508 {
20509 	if (prog->type == BPF_PROG_TYPE_TRACING) {
20510 		switch (prog->expected_attach_type) {
20511 		case BPF_TRACE_FENTRY:
20512 		case BPF_TRACE_FEXIT:
20513 		case BPF_MODIFY_RETURN:
20514 		case BPF_TRACE_ITER:
20515 		case BPF_TRACE_FSESSION:
20516 		case BPF_TRACE_RAW_TP:
20517 		case BPF_TRACE_FENTRY_MULTI:
20518 		case BPF_TRACE_FEXIT_MULTI:
20519 		case BPF_TRACE_FSESSION_MULTI:
20520 			return true;
20521 		default:
20522 			return false;
20523 		}
20524 	}
20525 	if (prog->type == BPF_PROG_TYPE_LSM)
20526 		return prog->expected_attach_type != BPF_LSM_CGROUP;
20527 
20528 	return prog->type == BPF_PROG_TYPE_KPROBE /* only for uprobes */ ||
20529 	       prog->type == BPF_PROG_TYPE_STRUCT_OPS ||
20530 	       prog->type == BPF_PROG_TYPE_RAW_TRACEPOINT ||
20531 	       prog->type == BPF_PROG_TYPE_TRACEPOINT;
20532 }
20533 
20534 static int check_attach_btf_id(struct bpf_verifier_env *env)
20535 {
20536 	struct bpf_prog *prog = env->prog;
20537 	struct bpf_prog *tgt_prog = prog->aux->dst_prog;
20538 	struct bpf_attach_target_info tgt_info = {};
20539 	u32 btf_id = prog->aux->attach_btf_id;
20540 	struct bpf_trampoline *tr;
20541 	int ret;
20542 	u64 key;
20543 
20544 	if (prog->type == BPF_PROG_TYPE_SYSCALL) {
20545 		if (prog->sleepable)
20546 			/* attach_btf_id checked to be zero already */
20547 			return 0;
20548 		verbose(env, "Syscall programs can only be sleepable\n");
20549 		return -EINVAL;
20550 	}
20551 
20552 	if (prog->sleepable && !can_be_sleepable(prog)) {
20553 		verbose(env, "Program of this type cannot be sleepable\n");
20554 		return -EINVAL;
20555 	}
20556 
20557 	if (prog->type == BPF_PROG_TYPE_STRUCT_OPS)
20558 		return check_struct_ops_btf_id(env);
20559 
20560 	if (prog->type != BPF_PROG_TYPE_TRACING &&
20561 	    prog->type != BPF_PROG_TYPE_LSM &&
20562 	    prog->type != BPF_PROG_TYPE_EXT)
20563 		return 0;
20564 
20565 	ret = bpf_check_attach_target(&env->log, prog, tgt_prog, btf_id, &tgt_info);
20566 	if (ret)
20567 		return ret;
20568 
20569 	if (tgt_prog && prog->type == BPF_PROG_TYPE_EXT) {
20570 		/* to make freplace equivalent to their targets, they need to
20571 		 * inherit env->ops and expected_attach_type for the rest of the
20572 		 * verification
20573 		 */
20574 		env->ops = bpf_verifier_ops[tgt_prog->type];
20575 		prog->expected_attach_type = tgt_prog->expected_attach_type;
20576 	}
20577 
20578 	/* store info about the attachment target that will be used later */
20579 	prog->aux->attach_func_proto = tgt_info.tgt_type;
20580 	prog->aux->attach_func_name = tgt_info.tgt_name;
20581 	prog->aux->mod = tgt_info.tgt_mod;
20582 
20583 	if (tgt_prog) {
20584 		prog->aux->saved_dst_prog_type = tgt_prog->type;
20585 		prog->aux->saved_dst_attach_type = tgt_prog->expected_attach_type;
20586 	}
20587 
20588 	if (prog->expected_attach_type == BPF_TRACE_RAW_TP) {
20589 		prog->aux->attach_btf_trace = true;
20590 		return 0;
20591 	} else if (prog->expected_attach_type == BPF_TRACE_ITER) {
20592 		return bpf_iter_prog_supported(prog);
20593 	}
20594 
20595 	if (prog->type == BPF_PROG_TYPE_LSM) {
20596 		ret = bpf_lsm_verify_prog(&env->log, prog);
20597 		if (ret < 0)
20598 			return ret;
20599 	} else if (prog->type == BPF_PROG_TYPE_TRACING &&
20600 		   btf_id_set_contains(&btf_id_deny, btf_id)) {
20601 		verbose(env, "Attaching tracing programs to function '%s' is rejected.\n",
20602 			tgt_info.tgt_name);
20603 		return -EINVAL;
20604 	} else if ((prog->expected_attach_type == BPF_TRACE_FEXIT ||
20605 		   prog->expected_attach_type == BPF_TRACE_FSESSION ||
20606 		   prog->expected_attach_type == BPF_TRACE_FSESSION_MULTI ||
20607 		   prog->expected_attach_type == BPF_MODIFY_RETURN) &&
20608 		   btf_id_set_contains(&noreturn_deny, btf_id)) {
20609 		verbose(env, "Attaching fexit/fsession/fmod_ret to __noreturn function '%s' is rejected.\n",
20610 			tgt_info.tgt_name);
20611 		return -EINVAL;
20612 	}
20613 
20614 	/*
20615 	 * We don't get trampoline for tracing_multi programs at this point,
20616 	 * it's done when tracing_multi link is created.
20617 	 */
20618 	if (prog->type == BPF_PROG_TYPE_TRACING &&
20619 	    is_tracing_multi(prog->expected_attach_type))
20620 		return 0;
20621 
20622 	key = bpf_trampoline_compute_key(tgt_prog, prog->aux->attach_btf, btf_id);
20623 	tr = bpf_trampoline_get(key, &tgt_info);
20624 	if (!tr)
20625 		return -ENOMEM;
20626 
20627 	if (tgt_prog && tgt_prog->aux->tail_call_reachable)
20628 		bpf_trampoline_set_flags(tr, BPF_TRAMP_F_TAIL_CALL_CTX);
20629 
20630 	prog->aux->dst_trampoline = tr;
20631 	return 0;
20632 }
20633 
20634 int bpf_check_attach_btf_id_multi(struct btf *btf, struct bpf_prog *prog, u32 btf_id,
20635 				  struct bpf_attach_target_info *tgt_info)
20636 {
20637 	const struct btf_type *t;
20638 	unsigned long addr;
20639 	const char *tname;
20640 	int err;
20641 
20642 	if (!btf_id || !btf)
20643 		return -EINVAL;
20644 
20645 	/* Check noreturn attachment. */
20646 	if ((prog->expected_attach_type == BPF_TRACE_FEXIT_MULTI ||
20647 	     prog->expected_attach_type == BPF_TRACE_FSESSION_MULTI) &&
20648 	     btf_id_set_contains(&noreturn_deny, btf_id))
20649 		return -EINVAL;
20650 	/* Check denied attachment. */
20651 	if (btf_id_set_contains(&btf_id_deny, btf_id))
20652 		return -EINVAL;
20653 
20654 	/* Check and get function target data. */
20655 	t = btf_type_by_id(btf, btf_id);
20656 	if (!t)
20657 		return -EINVAL;
20658 	tname = btf_name_by_offset(btf, t->name_off);
20659 	if (!tname)
20660 		return -EINVAL;
20661 	t = btf_attach_func_proto(NULL, btf, btf_id);
20662 	if (!t || !btf_type_is_func_proto(t))
20663 		return -EINVAL;
20664 	err = btf_distill_func_proto(NULL, btf, t, tname, &tgt_info->fmodel);
20665 	if (err < 0)
20666 		return err;
20667 	if (tgt_info->fmodel.ret_size > 8 &&
20668 	    attach_uses_trampoline_retval(prog->expected_attach_type))
20669 		return -EOPNOTSUPP;
20670 	if (btf_is_module(btf)) {
20671 		/* The bpf program already holds reference to module. */
20672 		if (WARN_ON_ONCE(!prog->aux->mod))
20673 			return -EINVAL;
20674 		addr = find_kallsyms_symbol_value(prog->aux->mod, tname);
20675 	} else {
20676 		addr = kallsyms_lookup_name(tname);
20677 	}
20678 	if (!addr || !ftrace_location(addr))
20679 		return -ENOENT;
20680 
20681 	/* Check sleepable program attachment. */
20682 	if (prog->sleepable) {
20683 		err = btf_id_allow_sleepable(btf_id, addr, prog, btf);
20684 		if (err)
20685 			return err;
20686 	}
20687 	tgt_info->tgt_addr = addr;
20688 	return 0;
20689 }
20690 
20691 struct btf *bpf_get_btf_vmlinux(void)
20692 {
20693 	/* Pairs with the smp_store_release() on the parse path below. */
20694 	struct btf *btf = smp_load_acquire(&btf_vmlinux);
20695 
20696 	if (!btf && IS_ENABLED(CONFIG_DEBUG_INFO_BTF)) {
20697 		mutex_lock(&btf_vmlinux_lock);
20698 		btf = btf_vmlinux;
20699 		if (!btf) {
20700 			btf = btf_parse_vmlinux();
20701 			/*
20702 			 * Order the parsed BTF contents and the globals the
20703 			 * parse populated (e.g. bpf_ctx_convert.t) before
20704 			 * the pointer publication. Pairs with the acquire
20705 			 * on the lockless fast path above.
20706 			 */
20707 			smp_store_release(&btf_vmlinux, btf);
20708 		}
20709 		mutex_unlock(&btf_vmlinux_lock);
20710 	}
20711 	return btf;
20712 }
20713 
20714 /*
20715  * The add_fd_from_fd_array() is executed only if fd_array_cnt is non-zero. In
20716  * this case expect that every file descriptor in the array is either a map or
20717  * a BTF. Everything else is considered to be trash.
20718  */
20719 static int add_fd_from_fd_array(struct bpf_verifier_env *env, u32 idx, int fd)
20720 {
20721 	struct bpf_map *map;
20722 	struct btf *btf;
20723 	CLASS(fd, f)(fd);
20724 	int err;
20725 
20726 	map = __bpf_map_get(f);
20727 	if (!IS_ERR(map)) {
20728 		err = __add_used_map(env, map);
20729 		if (err < 0)
20730 			return err;
20731 		fd_slot_set_map(&env->fd_array[idx], map);
20732 		return 0;
20733 	}
20734 
20735 	btf = __btf_get_by_fd(f);
20736 	if (!IS_ERR(btf)) {
20737 		btf_get(btf);
20738 		err = __add_used_btf(env, btf);
20739 		if (err < 0)
20740 			return err;
20741 		fd_slot_set_btf(&env->fd_array[idx], btf);
20742 		return 0;
20743 	}
20744 
20745 	verbose(env, "fd %d is not pointing to valid bpf_map or btf\n", fd);
20746 	return PTR_ERR(map);
20747 }
20748 
20749 /*
20750  * A continuous fd_array is resolved into an in-memory cache with one slot
20751  * per entry. The bound here is deliberately generous and not derived from
20752  * the per-program object limits: Duplicate entries /are/ permitted, and
20753  * the number of distinct maps and BTFs a program can bind is enforced when
20754  * each entry is resolved by __add_used_map() and __add_used_btf().
20755  */
20756 #define MAX_FD_ARRAY_CNT 4096
20757 
20758 static int process_fd_array_continuous(struct bpf_verifier_env *env,
20759 				       bpfptr_t fd_array, u32 cnt)
20760 {
20761 	int fd, ret;
20762 	u32 i;
20763 
20764 	if (cnt > MAX_FD_ARRAY_CNT) {
20765 		verbose(env, "fd_array has too many entries (%u, max %u)\n",
20766 			cnt, MAX_FD_ARRAY_CNT);
20767 		return -E2BIG;
20768 	}
20769 
20770 	env->fd_array = kvzalloc_objs(*env->fd_array, cnt, GFP_KERNEL_ACCOUNT);
20771 	if (!env->fd_array)
20772 		return -ENOMEM;
20773 	env->fd_array_cnt = cnt;
20774 	for (i = 0; i < cnt; i++) {
20775 		if (copy_from_bpfptr_offset(&fd, fd_array,
20776 					    (size_t)i * sizeof(fd), sizeof(fd)))
20777 			return -EFAULT;
20778 		ret = add_fd_from_fd_array(env, i, fd);
20779 		if (ret)
20780 			return ret;
20781 	}
20782 	return 0;
20783 }
20784 
20785 static int process_fd_array(struct bpf_verifier_env *env,
20786 			    union bpf_attr *attr, bpfptr_t uattr)
20787 {
20788 	bpfptr_t fd_array = make_bpfptr(attr->fd_array, uattr.is_kernel);
20789 
20790 	if (bpfptr_is_null(fd_array)) {
20791 		if (attr->fd_array_cnt) {
20792 			verbose(env, "fd_array_cnt %u without fd_array is invalid\n",
20793 				attr->fd_array_cnt);
20794 			return -EINVAL;
20795 		}
20796 		return 0;
20797 	}
20798 	/*
20799 	 * New API: the caller passes fd_array_cnt and a continuous array that
20800 	 * is resolved and bound up front. Legacy API (no fd_array_cnt): keep
20801 	 * the caller's array and resolve entries on the spot at each reference.
20802 	 */
20803 	if (attr->fd_array_cnt)
20804 		return process_fd_array_continuous(env, fd_array,
20805 						   attr->fd_array_cnt);
20806 	env->fd_array_raw = fd_array;
20807 	return 0;
20808 }
20809 
20810 /* replace a generic kfunc with a specialized version if necessary */
20811 static int specialize_kfunc(struct bpf_verifier_env *env, struct bpf_kfunc_desc *desc, int insn_idx)
20812 {
20813 	struct bpf_prog *prog = env->prog;
20814 	bool seen_direct_write;
20815 	void *xdp_kfunc;
20816 	bool is_rdonly;
20817 	u32 func_id = desc->func_id;
20818 	u16 offset = desc->offset;
20819 	unsigned long addr = desc->addr;
20820 
20821 	if (offset) /* return if module BTF is used */
20822 		return 0;
20823 
20824 	if (bpf_dev_bound_kfunc_id(func_id)) {
20825 		xdp_kfunc = bpf_dev_bound_resolve_kfunc(prog, func_id);
20826 		if (xdp_kfunc)
20827 			addr = (unsigned long)xdp_kfunc;
20828 		/* fallback to default kfunc when not supported by netdev */
20829 	} else if (func_id == special_kfunc_list[KF_bpf_dynptr_from_skb]) {
20830 		seen_direct_write = env->seen_direct_write;
20831 		is_rdonly = !may_access_direct_pkt_data(env, NULL, BPF_WRITE);
20832 
20833 		if (is_rdonly)
20834 			addr = (unsigned long)bpf_dynptr_from_skb_rdonly;
20835 
20836 		/* restore env->seen_direct_write to its original value, since
20837 		 * may_access_direct_pkt_data mutates it
20838 		 */
20839 		env->seen_direct_write = seen_direct_write;
20840 	} else if (func_id == special_kfunc_list[KF_bpf_set_dentry_xattr]) {
20841 		if (bpf_lsm_has_d_inode_locked(prog))
20842 			addr = (unsigned long)bpf_set_dentry_xattr_locked;
20843 	} else if (func_id == special_kfunc_list[KF_bpf_remove_dentry_xattr]) {
20844 		if (bpf_lsm_has_d_inode_locked(prog))
20845 			addr = (unsigned long)bpf_remove_dentry_xattr_locked;
20846 	} else if (func_id == special_kfunc_list[KF_bpf_dynptr_from_file]) {
20847 		if (!env->insn_aux_data[insn_idx].non_sleepable)
20848 			addr = (unsigned long)bpf_dynptr_from_file_sleepable;
20849 	} else if (func_id == special_kfunc_list[KF_bpf_arena_alloc_pages]) {
20850 		if (env->insn_aux_data[insn_idx].non_sleepable)
20851 			addr = (unsigned long)bpf_arena_alloc_pages_non_sleepable;
20852 	} else if (func_id == special_kfunc_list[KF_bpf_arena_free_pages]) {
20853 		if (env->insn_aux_data[insn_idx].non_sleepable)
20854 			addr = (unsigned long)bpf_arena_free_pages_non_sleepable;
20855 	}
20856 	desc->addr = addr;
20857 	return 0;
20858 }
20859 
20860 static void __fixup_collection_insert_kfunc(struct bpf_insn_aux_data *insn_aux,
20861 					    u16 struct_meta_reg,
20862 					    u16 node_offset_reg,
20863 					    struct bpf_insn *insn,
20864 					    struct bpf_insn *insn_buf,
20865 					    int *cnt)
20866 {
20867 	struct btf_struct_meta *kptr_struct_meta = insn_aux->kptr_struct_meta;
20868 	struct bpf_insn addr[2] = { BPF_LD_IMM64(struct_meta_reg, (long)kptr_struct_meta) };
20869 
20870 	insn_buf[0] = addr[0];
20871 	insn_buf[1] = addr[1];
20872 	insn_buf[2] = BPF_MOV64_IMM(node_offset_reg, insn_aux->insert_off);
20873 	insn_buf[3] = *insn;
20874 	*cnt = 4;
20875 }
20876 
20877 int bpf_fixup_kfunc_call(struct bpf_verifier_env *env, struct bpf_insn *insn,
20878 		     struct bpf_insn *insn_buf, int insn_idx, int *cnt)
20879 {
20880 	struct bpf_kfunc_desc *desc;
20881 	int err;
20882 
20883 	if (!insn->imm) {
20884 		verbose(env, "invalid kernel function call not eliminated in verifier pass\n");
20885 		return -EINVAL;
20886 	}
20887 
20888 	*cnt = 0;
20889 
20890 	/* insn->imm has the btf func_id. Replace it with an offset relative to
20891 	 * __bpf_call_base, unless the JIT needs to call functions that are
20892 	 * further than 32 bits away (bpf_jit_supports_far_kfunc_call()).
20893 	 */
20894 	desc = find_kfunc_desc(env->prog, insn->imm, insn->off);
20895 	if (!desc) {
20896 		verifier_bug(env, "kernel function descriptor not found for func_id %u",
20897 			     insn->imm);
20898 		return -EFAULT;
20899 	}
20900 
20901 	err = specialize_kfunc(env, desc, insn_idx);
20902 	if (err)
20903 		return err;
20904 
20905 	if (!bpf_jit_supports_far_kfunc_call())
20906 		insn->imm = BPF_CALL_IMM(desc->addr);
20907 
20908 	if (is_bpf_obj_new_kfunc(desc->func_id) || is_bpf_percpu_obj_new_kfunc(desc->func_id)) {
20909 		struct btf_struct_meta *kptr_struct_meta = env->insn_aux_data[insn_idx].kptr_struct_meta;
20910 		struct bpf_insn addr[2] = { BPF_LD_IMM64(BPF_REG_2, (long)kptr_struct_meta) };
20911 		u64 obj_new_size = env->insn_aux_data[insn_idx].obj_new_size;
20912 
20913 		if (is_bpf_percpu_obj_new_kfunc(desc->func_id) && kptr_struct_meta) {
20914 			verifier_bug(env, "NULL kptr_struct_meta expected at insn_idx %d",
20915 				     insn_idx);
20916 			return -EFAULT;
20917 		}
20918 
20919 		insn_buf[0] = BPF_MOV64_IMM(BPF_REG_1, obj_new_size);
20920 		insn_buf[1] = addr[0];
20921 		insn_buf[2] = addr[1];
20922 		insn_buf[3] = *insn;
20923 		*cnt = 4;
20924 	} else if (is_bpf_obj_drop_kfunc(desc->func_id) ||
20925 		   is_bpf_percpu_obj_drop_kfunc(desc->func_id) ||
20926 		   is_bpf_refcount_acquire_kfunc(desc->func_id)) {
20927 		struct btf_struct_meta *kptr_struct_meta = env->insn_aux_data[insn_idx].kptr_struct_meta;
20928 		struct bpf_insn addr[2] = { BPF_LD_IMM64(BPF_REG_2, (long)kptr_struct_meta) };
20929 
20930 		if (is_bpf_percpu_obj_drop_kfunc(desc->func_id) && kptr_struct_meta) {
20931 			verifier_bug(env, "NULL kptr_struct_meta expected at insn_idx %d",
20932 				     insn_idx);
20933 			return -EFAULT;
20934 		}
20935 
20936 		if (is_bpf_refcount_acquire_kfunc(desc->func_id) && !kptr_struct_meta) {
20937 			verifier_bug(env, "kptr_struct_meta expected at insn_idx %d",
20938 				     insn_idx);
20939 			return -EFAULT;
20940 		}
20941 
20942 		insn_buf[0] = addr[0];
20943 		insn_buf[1] = addr[1];
20944 		insn_buf[2] = *insn;
20945 		*cnt = 3;
20946 	} else if (is_bpf_list_push_kfunc(desc->func_id) ||
20947 		   is_bpf_rbtree_add_kfunc(desc->func_id)) {
20948 		struct btf_struct_meta *kptr_struct_meta = env->insn_aux_data[insn_idx].kptr_struct_meta;
20949 		int struct_meta_reg = BPF_REG_3;
20950 		int node_offset_reg = BPF_REG_4;
20951 
20952 		/* list_add/rbtree_add have an extra arg (prev/less),
20953 		 * so args-to-fixup are in diff regs.
20954 		 */
20955 		if (desc->func_id == special_kfunc_list[KF_bpf_list_add] ||
20956 		    is_bpf_rbtree_add_kfunc(desc->func_id)) {
20957 			struct_meta_reg = BPF_REG_4;
20958 			node_offset_reg = BPF_REG_5;
20959 		}
20960 
20961 		if (!kptr_struct_meta) {
20962 			verifier_bug(env, "kptr_struct_meta expected at insn_idx %d",
20963 				     insn_idx);
20964 			return -EFAULT;
20965 		}
20966 
20967 		__fixup_collection_insert_kfunc(&env->insn_aux_data[insn_idx], struct_meta_reg,
20968 						node_offset_reg, insn, insn_buf, cnt);
20969 	} else if (desc->func_id == special_kfunc_list[KF_bpf_cast_to_kern_ctx] ||
20970 		   desc->func_id == special_kfunc_list[KF_bpf_rdonly_cast]) {
20971 		insn_buf[0] = BPF_MOV64_REG(BPF_REG_0, BPF_REG_1);
20972 		*cnt = 1;
20973 	} else if (desc->func_id == special_kfunc_list[KF_bpf_session_is_return] &&
20974 		   (env->prog->expected_attach_type == BPF_TRACE_FSESSION ||
20975 		    env->prog->expected_attach_type == BPF_TRACE_FSESSION_MULTI)) {
20976 
20977 		/*
20978 		 * inline the bpf_session_is_return() for fsession:
20979 		 *   bool bpf_session_is_return(void *ctx)
20980 		 *   {
20981 		 *       return (((u64 *)ctx)[-1] >> BPF_TRAMP_IS_RETURN_SHIFT) & 1;
20982 		 *   }
20983 		 */
20984 		insn_buf[0] = BPF_LDX_MEM(BPF_DW, BPF_REG_0, BPF_REG_1, -8);
20985 		insn_buf[1] = BPF_ALU64_IMM(BPF_RSH, BPF_REG_0, BPF_TRAMP_IS_RETURN_SHIFT);
20986 		insn_buf[2] = BPF_ALU64_IMM(BPF_AND, BPF_REG_0, 1);
20987 		*cnt = 3;
20988 	} else if (desc->func_id == special_kfunc_list[KF_bpf_session_cookie] &&
20989 		   (env->prog->expected_attach_type == BPF_TRACE_FSESSION ||
20990 		    env->prog->expected_attach_type == BPF_TRACE_FSESSION_MULTI)) {
20991 		/*
20992 		 * inline bpf_session_cookie() for fsession:
20993 		 *   __u64 *bpf_session_cookie(void *ctx)
20994 		 *   {
20995 		 *       u64 off = (((u64 *)ctx)[-1] >> BPF_TRAMP_COOKIE_INDEX_SHIFT) & 0xFF;
20996 		 *       return &((u64 *)ctx)[-off];
20997 		 *   }
20998 		 */
20999 		insn_buf[0] = BPF_LDX_MEM(BPF_DW, BPF_REG_0, BPF_REG_1, -8);
21000 		insn_buf[1] = BPF_ALU64_IMM(BPF_RSH, BPF_REG_0, BPF_TRAMP_COOKIE_INDEX_SHIFT);
21001 		insn_buf[2] = BPF_ALU64_IMM(BPF_AND, BPF_REG_0, 0xFF);
21002 		insn_buf[3] = BPF_ALU64_IMM(BPF_LSH, BPF_REG_0, 3);
21003 		insn_buf[4] = BPF_ALU64_REG(BPF_SUB, BPF_REG_0, BPF_REG_1);
21004 		insn_buf[5] = BPF_ALU64_IMM(BPF_NEG, BPF_REG_0, 0);
21005 		*cnt = 6;
21006 	} else if (desc->func_id == special_kfunc_list[KF_bpf_iter_num_new]) {
21007 		/* inline bpf_iter_num_new(&it, start, end); R1=&it, R2=start, R3=end */
21008 		int i = 0;
21009 
21010 		/* if (start > end) goto einval; */
21011 		insn_buf[i++] = BPF_JMP32_REG(BPF_JSGT, BPF_REG_2, BPF_REG_3, 8);
21012 		/* r0 = (u32)end - (u32)start; if (r0 > BPF_MAX_LOOPS) goto e2big; */
21013 		insn_buf[i++] = BPF_MOV32_REG(BPF_REG_0, BPF_REG_3);
21014 		insn_buf[i++] = BPF_ALU32_REG(BPF_SUB, BPF_REG_0, BPF_REG_2);
21015 		insn_buf[i++] = BPF_JMP_IMM(BPF_JGT, BPF_REG_0, BPF_MAX_LOOPS, 8);
21016 		/* s->cur = start - 1; s->end = end; return 0; */
21017 		insn_buf[i++] = BPF_ALU32_IMM(BPF_ADD, BPF_REG_2, -1);
21018 		insn_buf[i++] = BPF_STX_MEM(BPF_W, BPF_REG_1, BPF_REG_2, 0);
21019 		insn_buf[i++] = BPF_STX_MEM(BPF_W, BPF_REG_1, BPF_REG_3, 4);
21020 		insn_buf[i++] = BPF_MOV64_IMM(BPF_REG_0, 0);
21021 		insn_buf[i++] = BPF_JMP_A(5);
21022 		/* einval: s->cur = s->end = 0; return -EINVAL; */
21023 		insn_buf[i++] = BPF_ST_MEM(BPF_DW, BPF_REG_1, 0, 0);
21024 		insn_buf[i++] = BPF_MOV64_IMM(BPF_REG_0, -EINVAL);
21025 		insn_buf[i++] = BPF_JMP_A(2);
21026 		/* e2big: s->cur = s->end = 0; return -E2BIG; */
21027 		insn_buf[i++] = BPF_ST_MEM(BPF_DW, BPF_REG_1, 0, 0);
21028 		insn_buf[i++] = BPF_MOV64_IMM(BPF_REG_0, -E2BIG);
21029 		*cnt = i;
21030 	} else if (desc->func_id == special_kfunc_list[KF_bpf_iter_num_next]) {
21031 		/* inline bpf_iter_num_next(&it); R1=&it, returns &s->cur or NULL */
21032 		int i = 0;
21033 
21034 		/* r0 = s->cur + 1; if ((s32)r0 >= s->end) goto done; */
21035 		insn_buf[i++] = BPF_LDX_MEM(BPF_W, BPF_REG_0, BPF_REG_1, 0);
21036 		insn_buf[i++] = BPF_ALU32_IMM(BPF_ADD, BPF_REG_0, 1);
21037 		insn_buf[i++] = BPF_LDX_MEM(BPF_W, BPF_REG_2, BPF_REG_1, 4);
21038 		insn_buf[i++] = BPF_JMP32_REG(BPF_JSGE, BPF_REG_0, BPF_REG_2, 3);
21039 		/* s->cur = r0; return &s->cur; */
21040 		insn_buf[i++] = BPF_STX_MEM(BPF_W, BPF_REG_1, BPF_REG_0, 0);
21041 		insn_buf[i++] = BPF_MOV64_REG(BPF_REG_0, BPF_REG_1);
21042 		insn_buf[i++] = BPF_JMP_A(2);
21043 		/* done: s->cur = s->end = 0; return NULL; */
21044 		insn_buf[i++] = BPF_ST_MEM(BPF_DW, BPF_REG_1, 0, 0);
21045 		insn_buf[i++] = BPF_MOV64_IMM(BPF_REG_0, 0);
21046 		*cnt = i;
21047 	} else if (desc->func_id == special_kfunc_list[KF_bpf_iter_num_destroy]) {
21048 		/* bpf_iter_num_destroy() is a no-op; emit a nop to drop the call */
21049 		insn_buf[0] = BPF_JMP_A(0);
21050 		*cnt = 1;
21051 	}
21052 
21053 	if (env->insn_aux_data[insn_idx].arg_prog) {
21054 		u32 regno = env->insn_aux_data[insn_idx].arg_prog;
21055 		struct bpf_insn ld_addrs[2] = { BPF_LD_IMM64(regno, (long)env->prog->aux) };
21056 		int idx = *cnt;
21057 
21058 		insn_buf[idx++] = ld_addrs[0];
21059 		insn_buf[idx++] = ld_addrs[1];
21060 		insn_buf[idx++] = *insn;
21061 		*cnt = idx;
21062 	}
21063 	return 0;
21064 }
21065 
21066 static enum bpf_sig_keyring bpf_classify_keyring(s32 keyring_id)
21067 {
21068 	switch (keyring_id) {
21069 	case 0:
21070 		return BPF_SIG_KEYRING_BUILTIN;
21071 	case (s32)(unsigned long)VERIFY_USE_SECONDARY_KEYRING:
21072 		return BPF_SIG_KEYRING_SECONDARY;
21073 	case (s32)(unsigned long)VERIFY_USE_PLATFORM_KEYRING:
21074 		return BPF_SIG_KEYRING_PLATFORM;
21075 	default:
21076 		return BPF_SIG_KEYRING_USER;
21077 	}
21078 }
21079 
21080 /*
21081  * Verify the PKCS#7 signature of a loaded program. Called from bpf_check()
21082  * once the program's metadata maps have been resolved into used_maps, so
21083  * the exact maps folded into the signature are the ones the program binds.
21084  *
21085  * The signature covers the instructions followed by the frozen contents of
21086  * each map, in @maps order: insns || map_0 || map_1 || [...]. On success the
21087  * verdict and keyring info are recorded on prog->aux.
21088  */
21089 static int bpf_prog_verify_signature(struct bpf_verifier_env *env,
21090 				     union bpf_attr *attr, bool is_kernel)
21091 {
21092 	bpfptr_t usig = make_bpfptr(attr->signature, is_kernel);
21093 	struct bpf_dynptr_kern sig_ptr, data_ptr;
21094 	struct bpf_prog *prog = env->prog;
21095 	struct bpf_map **maps = env->used_maps;
21096 	struct bpf_key *key = NULL;
21097 	void *sig, *data = NULL;
21098 	u32 map_cnt = env->used_map_cnt;
21099 	u32 i, off, insns_sz;
21100 	u64 data_sz;
21101 	int err = 0;
21102 
21103 	/*
21104 	 * Don't attempt to use kmalloc_large or vmalloc for signatures.
21105 	 * Practical signature for BPF program should be below this limit.
21106 	 */
21107 	if (!attr->signature_size ||
21108 	    attr->signature_size > KMALLOC_MAX_CACHE_SIZE)
21109 		return -EINVAL;
21110 	if (system_keyring_id_check(attr->keyring_id) == 0)
21111 		key = bpf_lookup_system_key(attr->keyring_id);
21112 	else
21113 		key = bpf_lookup_user_key(attr->keyring_id, 0);
21114 	if (!key) {
21115 		verbose(env, "cannot resolve signing keyring with keyring_id %d\n",
21116 			attr->keyring_id);
21117 		return -EINVAL;
21118 	}
21119 
21120 	sig = kvmemdup_bpfptr(usig, attr->signature_size);
21121 	if (IS_ERR(sig)) {
21122 		bpf_key_put(key);
21123 		return PTR_ERR(sig);
21124 	}
21125 
21126 	insns_sz = prog->len * sizeof(struct bpf_insn);
21127 	data_sz = insns_sz;
21128 	for (i = 0; i < map_cnt; i++) {
21129 		struct bpf_map *map = maps[i];
21130 
21131 		if (map->map_type != BPF_MAP_TYPE_ARRAY ||
21132 		    !map->ops->map_direct_value_addr) {
21133 			verbose(env, "signed program metadata map '%s' must be an array\n",
21134 				map->name);
21135 			err = -EINVAL;
21136 			goto out;
21137 		}
21138 		if (!READ_ONCE(map->frozen)) {
21139 			verbose(env, "signed program metadata map '%s' must be frozen\n",
21140 				map->name);
21141 			err = -EPERM;
21142 			goto out;
21143 		}
21144 		if (bpf_map_write_active(map)) {
21145 			verbose(env, "signed program metadata map '%s' has active writers\n",
21146 				map->name);
21147 			err = -EBUSY;
21148 			goto out;
21149 		}
21150 		if (!map->excl_prog_sha) {
21151 			verbose(env, "signed program metadata map '%s' must be exclusive\n",
21152 				map->name);
21153 			err = -EPERM;
21154 			goto out;
21155 		}
21156 		data_sz += map->value_size;
21157 	}
21158 	if (bpf_dynptr_check_size(data_sz)) {
21159 		verbose(env, "signed payload too large: %llu bytes\n", data_sz);
21160 		err = -E2BIG;
21161 		goto out;
21162 	}
21163 	data = kvmalloc(data_sz, GFP_KERNEL_ACCOUNT | __GFP_ZERO);
21164 	if (!data) {
21165 		err = -ENOMEM;
21166 		goto out;
21167 	}
21168 	memcpy(data, prog->insnsi, insns_sz);
21169 	off = insns_sz;
21170 	for (i = 0; i < map_cnt; i++) {
21171 		struct bpf_map *map = maps[i];
21172 		u64 addr;
21173 
21174 		err = map->ops->map_direct_value_addr(map, &addr, 0);
21175 		if (err) {
21176 			verbose(env, "failed to read signed metadata map '%s': %d\n",
21177 				map->name, err);
21178 			goto out;
21179 		}
21180 		memcpy(data + off, (void *)(unsigned long)addr,
21181 		       map->value_size);
21182 		off += map->value_size;
21183 	}
21184 
21185 	bpf_dynptr_init(&data_ptr, data, BPF_DYNPTR_TYPE_LOCAL, 0, data_sz);
21186 	bpf_dynptr_init(&sig_ptr, sig, BPF_DYNPTR_TYPE_LOCAL, 0,
21187 			attr->signature_size);
21188 
21189 	err = bpf_verify_pkcs7_signature((struct bpf_dynptr *)&data_ptr,
21190 					 (struct bpf_dynptr *)&sig_ptr, key);
21191 	if (err) {
21192 		verbose(env, "signature verification failed: %d\n", err);
21193 	} else {
21194 		verbose(env, "signature verification passed\n");
21195 		prog->aux->sig.keyring_serial = bpf_key_serial(key);
21196 		prog->aux->sig.keyring_type = bpf_classify_keyring(attr->keyring_id);
21197 		prog->aux->sig.verdict = BPF_SIG_VERIFIED;
21198 	}
21199 out:
21200 	kvfree(data);
21201 	bpf_key_put(key);
21202 	kvfree(sig);
21203 	return err;
21204 }
21205 
21206 int bpf_check(struct bpf_prog **prog, union bpf_attr *attr, bpfptr_t uattr,
21207 	      struct bpf_log_attr *attr_log)
21208 {
21209 	u64 start_time = ktime_get_ns();
21210 	struct bpf_verifier_env *env;
21211 	int i, len, ret = -EINVAL, err;
21212 	bool is_priv;
21213 
21214 	BTF_TYPE_EMIT(enum bpf_features);
21215 
21216 	/* no program is valid */
21217 	if (ARRAY_SIZE(bpf_verifier_ops) == 0)
21218 		return -EINVAL;
21219 
21220 	/* 'struct bpf_verifier_env' can be global, but since it's not small,
21221 	 * allocate/free it every time bpf_check() is called
21222 	 */
21223 	env = kvzalloc_obj(struct bpf_verifier_env, GFP_KERNEL_ACCOUNT);
21224 	if (!env)
21225 		return -ENOMEM;
21226 
21227 	env->bt.env = env;
21228 	env->prog = *prog;
21229 	env->ops = bpf_verifier_ops[env->prog->type];
21230 
21231 	env->allow_ptr_leaks = bpf_allow_ptr_leaks(env->prog->aux->token);
21232 	env->allow_uninit_stack = bpf_allow_uninit_stack(env->prog->aux->token);
21233 	env->bypass_spec_v1 = bpf_bypass_spec_v1(env->prog->aux->token);
21234 	env->bypass_spec_v4 = bpf_bypass_spec_v4(env->prog->aux->token);
21235 	env->bpf_capable = is_priv = bpf_token_capable(env->prog->aux->token, CAP_BPF);
21236 	env->signature = attr->signature;
21237 
21238 	/* user could have requested verbose verifier output
21239 	 * and supplied buffer to store the verification trace
21240 	 */
21241 	ret = bpf_vlog_init(&env->log, attr_log->level, attr_log->ubuf, attr_log->size);
21242 	if (ret)
21243 		goto err_free_env;
21244 	ret = bpf_diag_init(env);
21245 	if (ret)
21246 		goto err_prep;
21247 	if (env->prog->insnsi[env->prog->len - 1].code == (BPF_LD | BPF_IMM | BPF_DW)) {
21248 		verbose(env, "invalid bpf_ld_imm64 insn\n");
21249 		ret = -EINVAL;
21250 		goto err_prep;
21251 	}
21252 	if (env->signature) {
21253 		ret = bpf_prog_calc_tag(env->prog);
21254 		if (ret < 0)
21255 			goto err_prep;
21256 	}
21257 
21258 	ret = process_fd_array(env, attr, uattr);
21259 	if (ret)
21260 		goto err_prep;
21261 
21262 	if (env->signature) {
21263 		ret = bpf_prog_verify_signature(env, attr, uattr.is_kernel);
21264 		if (ret)
21265 			goto err_prep;
21266 	}
21267 
21268 	ret = security_bpf_prog_load(env->prog, attr, env->prog->aux->token,
21269 				     uattr.is_kernel);
21270 	if (ret)
21271 		goto err_prep;
21272 
21273 	bpf_get_btf_vmlinux();
21274 
21275 	/* Serialize verification of unprivileged programs. */
21276 	if (!is_priv)
21277 		mutex_lock(&bpf_verifier_lock);
21278 
21279 	len = env->insn_aux_data_len = env->prog->len;
21280 	env->insn_aux_data =
21281 		__vmalloc(array_size(sizeof(struct bpf_insn_aux_data), len),
21282 			  GFP_KERNEL_ACCOUNT | __GFP_ZERO);
21283 	ret = -ENOMEM;
21284 	if (!env->insn_aux_data)
21285 		goto skip_full_check;
21286 	for (i = 0; i < len; i++)
21287 		env->insn_aux_data[i].orig_idx = i;
21288 	env->succ = bpf_iarray_realloc(NULL, 2);
21289 	if (!env->succ)
21290 		goto skip_full_check;
21291 
21292 	mark_verifier_state_clean(env);
21293 
21294 	if (IS_ERR(btf_vmlinux)) {
21295 		/* Either gcc or pahole or kernel are broken. */
21296 		verbose(env, "in-kernel BTF is malformed\n");
21297 		ret = PTR_ERR(btf_vmlinux);
21298 		goto skip_full_check;
21299 	}
21300 
21301 	env->strict_alignment = !!(attr->prog_flags & BPF_F_STRICT_ALIGNMENT);
21302 	if (!IS_ENABLED(CONFIG_HAVE_EFFICIENT_UNALIGNED_ACCESS))
21303 		env->strict_alignment = true;
21304 	if (attr->prog_flags & BPF_F_ANY_ALIGNMENT)
21305 		env->strict_alignment = false;
21306 
21307 	if (is_priv)
21308 		env->test_state_freq = attr->prog_flags & BPF_F_TEST_STATE_FREQ;
21309 	env->test_reg_invariants = attr->prog_flags & BPF_F_TEST_REG_INVARIANTS;
21310 
21311 	env->explored_states = kvzalloc_objs(struct list_head,
21312 					     state_htab_size(env),
21313 					     GFP_KERNEL_ACCOUNT);
21314 	ret = -ENOMEM;
21315 	if (!env->explored_states)
21316 		goto skip_full_check;
21317 
21318 	for (i = 0; i < state_htab_size(env); i++)
21319 		INIT_LIST_HEAD(&env->explored_states[i]);
21320 	INIT_LIST_HEAD(&env->free_list);
21321 
21322 	/* Prepare BTF and func_info needed to discover all subprograms. */
21323 	ret = bpf_prepare_btf_info(env, attr, uattr);
21324 	if (ret < 0)
21325 		goto skip_full_check;
21326 
21327 	/* Apply CO-RE before validating the program's instruction layout. */
21328 	ret = bpf_check_core_relo(env, attr, uattr);
21329 	if (ret < 0)
21330 		goto skip_full_check;
21331 
21332 	/* Discover all subprograms before validating their layout and BTF. */
21333 	ret = add_subprogs(env);
21334 	if (ret < 0)
21335 		goto skip_full_check;
21336 
21337 	ret = check_subprogs(env);
21338 	if (ret < 0)
21339 		goto skip_full_check;
21340 
21341 	/* Validate BTF against the complete subprogram layout. */
21342 	ret = bpf_check_btf_info(env, attr, uattr);
21343 	if (ret < 0)
21344 		goto skip_full_check;
21345 
21346 	/* Validate instructions and resolve the program's referenced resources. */
21347 	ret = check_and_resolve_insns(env);
21348 	if (ret < 0)
21349 		goto skip_full_check;
21350 
21351 	/* Build kfunc prototypes after resolving program resources. */
21352 	ret = add_kfuncs(env);
21353 	if (ret < 0)
21354 		goto skip_full_check;
21355 
21356 	if (bpf_prog_is_offloaded(env->prog->aux)) {
21357 		ret = bpf_prog_offload_verifier_prep(env->prog);
21358 		if (ret)
21359 			goto skip_full_check;
21360 	}
21361 
21362 	ret = bpf_check_cfg(env);
21363 	if (ret < 0)
21364 		goto skip_full_check;
21365 
21366 	ret = bpf_compute_postorder(env);
21367 	if (ret < 0)
21368 		goto skip_full_check;
21369 
21370 	ret = bpf_stack_liveness_init(env);
21371 	if (ret)
21372 		goto skip_full_check;
21373 
21374 	ret = check_attach_btf_id(env);
21375 	if (ret)
21376 		goto skip_full_check;
21377 
21378 	ret = bpf_compute_const_regs(env);
21379 	if (ret < 0)
21380 		goto skip_full_check;
21381 
21382 	ret = bpf_prune_dead_branches(env);
21383 	if (ret < 0)
21384 		goto skip_full_check;
21385 
21386 	ret = sort_subprogs_topo(env);
21387 	if (ret < 0)
21388 		goto skip_full_check;
21389 
21390 	ret = bpf_compute_scc(env);
21391 	if (ret < 0)
21392 		goto skip_full_check;
21393 
21394 	ret = bpf_compute_live_registers(env);
21395 	if (ret < 0)
21396 		goto skip_full_check;
21397 
21398 	ret = mark_fastcall_patterns(env);
21399 	if (ret < 0)
21400 		goto skip_full_check;
21401 
21402 	ret = do_check_main(env);
21403 	ret = ret ?: do_check_subprogs(env);
21404 
21405 	if (ret == 0 && bpf_prog_is_offloaded(env->prog->aux))
21406 		ret = bpf_prog_offload_finalize(env);
21407 
21408 skip_full_check:
21409 	kvfree(env->explored_states);
21410 
21411 	/* might decrease stack depth, keep it before passes that
21412 	 * allocate additional slots.
21413 	 */
21414 	if (ret == 0)
21415 		ret = bpf_remove_fastcall_spills_fills(env);
21416 
21417 	if (ret == 0)
21418 		ret = check_max_stack_depth(env);
21419 
21420 	/* instruction rewrites happen after this point */
21421 	if (ret == 0)
21422 		ret = bpf_optimize_bpf_loop(env);
21423 
21424 	if (is_priv) {
21425 		if (ret == 0)
21426 			bpf_opt_hard_wire_dead_code_branches(env);
21427 		if (ret == 0)
21428 			ret = bpf_opt_remove_dead_code(env);
21429 		if (ret == 0)
21430 			ret = bpf_opt_remove_nops(env);
21431 	} else {
21432 		if (ret == 0)
21433 			sanitize_dead_code(env);
21434 	}
21435 
21436 	if (ret == 0)
21437 		/* program is valid, convert *(u32*)(ctx + off) accesses */
21438 		ret = bpf_convert_ctx_accesses(env);
21439 
21440 	if (ret == 0)
21441 		ret = bpf_do_misc_fixups(env);
21442 
21443 	/* do 32-bit optimization after insn patching has done so those patched
21444 	 * insns could be handled correctly.
21445 	 */
21446 	if (ret == 0 && !bpf_prog_is_offloaded(env->prog->aux)) {
21447 		ret = bpf_opt_subreg_zext_lo32_rnd_hi32(env, attr);
21448 		env->prog->aux->verifier_zext = bpf_jit_needs_zext() ? !ret
21449 								     : false;
21450 	}
21451 
21452 	if (ret == 0)
21453 		ret = bpf_fixup_call_args(env);
21454 
21455 	env->verification_time = ktime_get_ns() - start_time;
21456 	print_verification_stats(env);
21457 	env->prog->aux->verified_insns = env->insn_processed;
21458 
21459 	/* preserve original error even if log finalization is successful */
21460 	err = bpf_log_attr_finalize(attr_log, &env->log);
21461 	if (err)
21462 		ret = err;
21463 
21464 	if (ret)
21465 		goto err_release_maps;
21466 
21467 	if (env->used_map_cnt) {
21468 		/* if program passed verifier, update used_maps in bpf_prog_info */
21469 		env->prog->aux->used_maps = kmalloc_objs(env->used_maps[0],
21470 							 env->used_map_cnt,
21471 							 GFP_KERNEL_ACCOUNT);
21472 
21473 		if (!env->prog->aux->used_maps) {
21474 			ret = -ENOMEM;
21475 			goto err_release_maps;
21476 		}
21477 
21478 		memcpy(env->prog->aux->used_maps, env->used_maps,
21479 		       sizeof(env->used_maps[0]) * env->used_map_cnt);
21480 		env->prog->aux->used_map_cnt = env->used_map_cnt;
21481 	}
21482 	if (env->used_btf_cnt) {
21483 		/* if program passed verifier, update used_btfs in bpf_prog_aux */
21484 		env->prog->aux->used_btfs = kmalloc_objs(env->used_btfs[0],
21485 							 env->used_btf_cnt,
21486 							 GFP_KERNEL_ACCOUNT);
21487 		if (!env->prog->aux->used_btfs) {
21488 			ret = -ENOMEM;
21489 			goto err_release_maps;
21490 		}
21491 
21492 		memcpy(env->prog->aux->used_btfs, env->used_btfs,
21493 		       sizeof(env->used_btfs[0]) * env->used_btf_cnt);
21494 		env->prog->aux->used_btf_cnt = env->used_btf_cnt;
21495 	}
21496 	if (env->used_map_cnt || env->used_btf_cnt) {
21497 		/* program is valid. Convert pseudo bpf_ld_imm64 into generic
21498 		 * bpf_ld_imm64 instructions
21499 		 */
21500 		convert_pseudo_ld_imm64(env);
21501 	}
21502 
21503 	adjust_btf_func(env);
21504 
21505 	/* extension progs temporarily inherit the attach_type of their targets
21506 	   for verification purposes, so set it back to zero before returning
21507 	 */
21508 	if (env->prog->type == BPF_PROG_TYPE_EXT)
21509 		env->prog->expected_attach_type = 0;
21510 
21511 	env->prog = __bpf_prog_select_runtime(env, env->prog, &ret);
21512 
21513 err_release_maps:
21514 	if (ret)
21515 		release_insn_arrays(env);
21516 	if (!env->prog->aux->used_maps)
21517 		/* if we didn't copy map pointers into bpf_prog_info, release
21518 		 * them now. Otherwise free_used_maps() will release them.
21519 		 */
21520 		release_maps(env);
21521 	if (!env->prog->aux->used_btfs)
21522 		release_btfs(env);
21523 
21524 	*prog = env->prog;
21525 
21526 	module_put(env->attach_btf_mod);
21527 	if (!is_priv)
21528 		mutex_unlock(&bpf_verifier_lock);
21529 	goto err_free_env;
21530 err_prep:
21531 	err = bpf_log_attr_finalize(attr_log, &env->log);
21532 	if (err)
21533 		ret = err;
21534 	release_insn_arrays(env);
21535 	release_maps(env);
21536 	release_btfs(env);
21537 err_free_env:
21538 	if (env->insn_aux_data)
21539 		bpf_clear_insn_aux_data(env, 0, env->insn_aux_data_len);
21540 	vfree(env->insn_aux_data);
21541 	kvfree(env->fd_array);
21542 	bpf_stack_liveness_free(env);
21543 	kvfree(env->cfg.insn_postorder);
21544 	kvfree(env->scc_info);
21545 	kvfree(env->succ);
21546 	kvfree(env->gotox_tmp_buf);
21547 	bpf_diag_free(env);
21548 	kvfree(env);
21549 	return ret;
21550 }
21551