xref: /linux/kernel/bpf/verifier.c (revision 5a8cd539ac19f7a68e68e1d25ef9ca2ff55b8500)
1 // SPDX-License-Identifier: GPL-2.0-only
2 /* Copyright (c) 2011-2014 PLUMgrid, http://plumgrid.com
3  * Copyright (c) 2016 Facebook
4  * Copyright (c) 2018 Covalent IO, Inc. http://covalent.io
5  */
6 #include <uapi/linux/btf.h>
7 #include <linux/bpf-cgroup.h>
8 #include <linux/kernel.h>
9 #include <linux/types.h>
10 #include <linux/slab.h>
11 #include <linux/bpf.h>
12 #include <linux/btf.h>
13 #include <linux/bpf_verifier.h>
14 #include <linux/filter.h>
15 #include <net/netlink.h>
16 #include <linux/file.h>
17 #include <linux/vmalloc.h>
18 #include <linux/stringify.h>
19 #include <linux/bsearch.h>
20 #include <linux/sort.h>
21 #include <linux/perf_event.h>
22 #include <linux/ctype.h>
23 #include <linux/error-injection.h>
24 #include <linux/bpf_lsm.h>
25 #include <linux/security.h>
26 #include <linux/verification.h>
27 #include <linux/btf_ids.h>
28 #include <linux/poison.h>
29 #include <linux/module.h>
30 #include <linux/cpumask.h>
31 #include <linux/cnum.h>
32 #include <linux/bpf_mem_alloc.h>
33 #include <net/xdp.h>
34 #include <linux/trace_events.h>
35 #include <linux/kallsyms.h>
36 
37 #include "diagnostics.h"
38 #include "disasm.h"
39 
40 static const struct bpf_verifier_ops * const bpf_verifier_ops[] = {
41 #define BPF_PROG_TYPE(_id, _name, prog_ctx_type, kern_ctx_type) \
42 	[_id] = & _name ## _verifier_ops,
43 #define BPF_MAP_TYPE(_id, _ops)
44 #define BPF_LINK_TYPE(_id, _name)
45 #include <linux/bpf_types.h>
46 #undef BPF_PROG_TYPE
47 #undef BPF_MAP_TYPE
48 #undef BPF_LINK_TYPE
49 };
50 
51 enum bpf_features {
52 	BPF_FEAT_RDONLY_CAST_TO_VOID = 0,
53 	BPF_FEAT_STREAMS	     = 1,
54 	__MAX_BPF_FEAT,
55 };
56 
57 struct bpf_mem_alloc bpf_global_percpu_ma;
58 static bool bpf_global_percpu_ma_set;
59 
60 /* bpf_check() is a static code analyzer that walks eBPF program
61  * instruction by instruction and updates register/stack state.
62  * All paths of conditional branches are analyzed until 'bpf_exit' insn.
63  *
64  * The first pass is depth-first-search to check that the program is a DAG.
65  * It rejects the following programs:
66  * - larger than BPF_MAXINSNS insns
67  * - if loop is present (detected via back-edge)
68  * - unreachable insns exist (shouldn't be a forest. program = one function)
69  * - out of bounds or malformed jumps
70  * The second pass is all possible path descent from the 1st insn.
71  * Since it's analyzing all paths through the program, the length of the
72  * analysis is limited to 64k insn, which may be hit even if total number of
73  * insn is less then 4K, but there are too many branches that change stack/regs.
74  * Number of 'branches to be analyzed' is limited to 1k
75  *
76  * On entry to each instruction, each register has a type, and the instruction
77  * changes the types of the registers depending on instruction semantics.
78  * If instruction is BPF_MOV64_REG(BPF_REG_1, BPF_REG_5), then type of R5 is
79  * copied to R1.
80  *
81  * All registers are 64-bit.
82  * R0 - return register
83  * R1-R5 argument passing registers
84  * R6-R9 callee saved registers
85  * R10 - frame pointer read-only
86  *
87  * At the start of BPF program the register R1 contains a pointer to bpf_context
88  * and has type PTR_TO_CTX.
89  *
90  * Verifier tracks arithmetic operations on pointers in case:
91  *    BPF_MOV64_REG(BPF_REG_1, BPF_REG_10),
92  *    BPF_ALU64_IMM(BPF_ADD, BPF_REG_1, -20),
93  * 1st insn copies R10 (which has FRAME_PTR) type into R1
94  * and 2nd arithmetic instruction is pattern matched to recognize
95  * that it wants to construct a pointer to some element within stack.
96  * So after 2nd insn, the register R1 has type PTR_TO_STACK
97  * (and -20 constant is saved for further stack bounds checking).
98  * Meaning that this reg is a pointer to stack plus known immediate constant.
99  *
100  * Most of the time the registers have SCALAR_VALUE type, which
101  * means the register has some value, but it's not a valid pointer.
102  * (like pointer plus pointer becomes SCALAR_VALUE type)
103  *
104  * When verifier sees load or store instructions the type of base register
105  * can be: PTR_TO_MAP_VALUE, PTR_TO_CTX, PTR_TO_STACK, PTR_TO_SOCKET. These are
106  * four pointer types recognized by check_mem_access() function.
107  *
108  * PTR_TO_MAP_VALUE means that this register is pointing to 'map element value'
109  * and the range of [ptr, ptr + map's value_size) is accessible.
110  *
111  * registers used to pass values to function calls are checked against
112  * function argument constraints.
113  *
114  * ARG_PTR_TO_MAP_KEY is one of such argument constraints.
115  * It means that the register type passed to this function must be
116  * PTR_TO_STACK and it will be used inside the function as
117  * 'pointer to map element key'
118  *
119  * For example the argument constraints for bpf_map_lookup_elem():
120  *   .ret_type = RET_PTR_TO_MAP_VALUE_OR_NULL,
121  *   .arg1_type = ARG_CONST_MAP_PTR,
122  *   .arg2_type = ARG_PTR_TO_MAP_KEY,
123  *
124  * ret_type says that this function returns 'pointer to map elem value or null'
125  * function expects 1st argument to be a const pointer to 'struct bpf_map' and
126  * 2nd argument should be a pointer to stack, which will be used inside
127  * the helper function as a pointer to map element key.
128  *
129  * On the kernel side the helper function looks like:
130  * u64 bpf_map_lookup_elem(u64 r1, u64 r2, u64 r3, u64 r4, u64 r5)
131  * {
132  *    struct bpf_map *map = (struct bpf_map *) (unsigned long) r1;
133  *    void *key = (void *) (unsigned long) r2;
134  *    void *value;
135  *
136  *    here kernel can access 'key' and 'map' pointers safely, knowing that
137  *    [key, key + map->key_size) bytes are valid and were initialized on
138  *    the stack of eBPF program.
139  * }
140  *
141  * Corresponding eBPF program may look like:
142  *    BPF_MOV64_REG(BPF_REG_2, BPF_REG_10),  // after this insn R2 type is FRAME_PTR
143  *    BPF_ALU64_IMM(BPF_ADD, BPF_REG_2, -4), // after this insn R2 type is PTR_TO_STACK
144  *    BPF_LD_MAP_FD(BPF_REG_1, map_fd),      // after this insn R1 type is CONST_PTR_TO_MAP
145  *    BPF_RAW_INSN(BPF_JMP | BPF_CALL, 0, 0, 0, BPF_FUNC_map_lookup_elem),
146  * here verifier looks at prototype of map_lookup_elem() and sees:
147  * .arg1_type == ARG_CONST_MAP_PTR and R1->type == CONST_PTR_TO_MAP, which is ok,
148  * Now verifier knows that this map has key of R1->map_ptr->key_size bytes
149  *
150  * Then .arg2_type == ARG_PTR_TO_MAP_KEY and R2->type == PTR_TO_STACK, ok so far,
151  * Now verifier checks that [R2, R2 + map's key_size) are within stack limits
152  * and were initialized prior to this call.
153  * If it's ok, then verifier allows this BPF_CALL insn and looks at
154  * .ret_type which is RET_PTR_TO_MAP_VALUE_OR_NULL, so it sets
155  * R0->type = PTR_TO_MAP_VALUE_OR_NULL which means bpf_map_lookup_elem() function
156  * returns either pointer to map value or NULL.
157  *
158  * When type PTR_TO_MAP_VALUE_OR_NULL passes through 'if (reg != 0) goto +off'
159  * insn, the register holding that pointer in the true branch changes state to
160  * PTR_TO_MAP_VALUE and the same register changes state to CONST_IMM in the false
161  * branch. See check_cond_jmp_op().
162  *
163  * After the call R0 is set to return type of the function and registers R1-R5
164  * are set to NOT_INIT to indicate that they are no longer readable.
165  *
166  * The following reference types represent a potential reference to a kernel
167  * resource which, after first being allocated, must be checked and freed by
168  * the BPF program:
169  * - PTR_TO_SOCKET_OR_NULL, PTR_TO_SOCKET
170  *
171  * When the verifier sees a helper call return a reference type, it allocates a
172  * pointer id for the reference and stores it in the current function state.
173  * Similar to the way that PTR_TO_MAP_VALUE_OR_NULL is converted into
174  * PTR_TO_MAP_VALUE, PTR_TO_SOCKET_OR_NULL becomes PTR_TO_SOCKET when the type
175  * passes through a NULL-check conditional. For the branch wherein the state is
176  * changed to CONST_IMM, the verifier releases the reference.
177  *
178  * For each helper function that allocates a reference, such as
179  * bpf_sk_lookup_tcp(), there is a corresponding release function, such as
180  * bpf_sk_release(). When a reference type passes into the release function,
181  * the verifier also releases the reference. If any unchecked or unreleased
182  * reference remains at the end of the program, the verifier rejects it.
183  */
184 
185 /* verifier_state + insn_idx are pushed to stack when branch is encountered */
186 struct bpf_verifier_stack_elem {
187 	/* verifier state is 'st'
188 	 * before processing instruction 'insn_idx'
189 	 * and after processing instruction 'prev_insn_idx'
190 	 */
191 	struct bpf_verifier_state st;
192 	int insn_idx;
193 	int prev_insn_idx;
194 	struct bpf_verifier_stack_elem *next;
195 	/* length of verifier log at the time this state was pushed on stack */
196 	u32 log_pos;
197 	u64 diag_log_pos;
198 };
199 
200 #define BPF_COMPLEXITY_LIMIT_JMP_SEQ	8192
201 #define BPF_COMPLEXITY_LIMIT_STATES	64
202 
203 #define BPF_GLOBAL_PERCPU_MA_MAX_SIZE  512
204 
205 #define BPF_PRIV_STACK_MIN_SIZE		64
206 
207 static int acquire_reference(struct bpf_verifier_env *env, int insn_idx, int parent_id);
208 static int __release_reference_nomark(struct bpf_verifier_state *state, int id);
209 static int release_reference_nomark(struct bpf_verifier_env *env, int id);
210 static int release_reference(struct bpf_verifier_env *env, int id);
211 static void invalidate_non_owning_refs(struct bpf_verifier_env *env);
212 static void invalidate_rcu_protected_refs(struct bpf_verifier_env *env);
213 static bool in_rbtree_lock_required_cb(struct bpf_verifier_env *env);
214 static bool is_tracing_prog_type(enum bpf_prog_type type);
215 static int ref_set_non_owning(struct bpf_verifier_env *env,
216 			      struct bpf_reg_state *reg);
217 static bool is_trusted_reg(struct bpf_verifier_env *env, const struct bpf_reg_state *reg);
218 static inline bool in_sleepable_context(struct bpf_verifier_env *env);
219 static const char *non_sleepable_context_description(struct bpf_verifier_env *env);
220 static void scalar32_min_max_add(struct bpf_reg_state *dst_reg, struct bpf_reg_state *src_reg);
221 static void scalar_min_max_add(struct bpf_reg_state *dst_reg, struct bpf_reg_state *src_reg);
222 
bpf_map_ptr_store(struct bpf_insn_aux_data * aux,struct bpf_map * map,bool unpriv,bool poison)223 static void bpf_map_ptr_store(struct bpf_insn_aux_data *aux,
224 			      struct bpf_map *map,
225 			      bool unpriv, bool poison)
226 {
227 	unpriv |= bpf_map_ptr_unpriv(aux);
228 	aux->map_ptr_state.unpriv = unpriv;
229 	aux->map_ptr_state.poison = poison;
230 	aux->map_ptr_state.map_ptr = map;
231 }
232 
bpf_map_key_store(struct bpf_insn_aux_data * aux,u64 state)233 static void bpf_map_key_store(struct bpf_insn_aux_data *aux, u64 state)
234 {
235 	bool poisoned = bpf_map_key_poisoned(aux);
236 
237 	aux->map_key_state = state | BPF_MAP_KEY_SEEN |
238 			     (poisoned ? BPF_MAP_KEY_POISON : 0ULL);
239 }
240 
update_ref_obj(struct ref_obj_desc * ref_obj,struct bpf_reg_state * reg)241 static void update_ref_obj(struct ref_obj_desc *ref_obj, struct bpf_reg_state *reg)
242 {
243 	ref_obj->id = reg->id;
244 	ref_obj->parent_id = reg->parent_id;
245 	ref_obj->cnt++;
246 }
247 
validate_ref_obj(struct bpf_verifier_env * env,struct ref_obj_desc * ref_obj)248 static int validate_ref_obj(struct bpf_verifier_env *env, struct ref_obj_desc *ref_obj)
249 {
250 	if (ref_obj->cnt > 1) {
251 		verifier_bug(env, "function expects only one referenced object but got %d\n",
252 			     ref_obj->cnt);
253 		return -EFAULT;
254 	}
255 
256 	return 0;
257 }
258 
259 struct bpf_kfunc_meta {
260 	struct btf *btf;
261 	const struct btf_type *proto;
262 	const char *name;
263 	const u32 *flags;
264 	s32 id;
265 };
266 
267 struct btf *btf_vmlinux;
268 
269 typedef struct argno {
270 	int argno;
271 } argno_t;
272 
argno_from_reg(u32 regno)273 static argno_t argno_from_reg(u32 regno)
274 {
275 	return (argno_t){ .argno = regno };
276 }
277 
argno_from_arg(u32 arg)278 static argno_t argno_from_arg(u32 arg)
279 {
280 	return (argno_t){ .argno = -arg };
281 }
282 
reg_from_argno(argno_t a)283 static int reg_from_argno(argno_t a)
284 {
285 	if (a.argno >= 0)
286 		return a.argno;
287 	if (a.argno >= -MAX_BPF_FUNC_REG_ARGS)
288 		return -a.argno;
289 	return -1;
290 }
291 
arg_from_argno(argno_t a)292 static int arg_from_argno(argno_t a)
293 {
294 	if (a.argno < 0)
295 		return -a.argno;
296 	return -1;
297 }
298 
arg_idx_from_argno(argno_t a)299 static int arg_idx_from_argno(argno_t a)
300 {
301 	return arg_from_argno(a) - 1;
302 }
303 
btf_type_name(const struct btf * btf,u32 id)304 static const char *btf_type_name(const struct btf *btf, u32 id)
305 {
306 	return btf_name_by_offset(btf, btf_type_by_id(btf, id)->name_off);
307 }
308 
309 static DEFINE_MUTEX(bpf_verifier_lock);
310 static DEFINE_MUTEX(btf_vmlinux_lock);
311 static DEFINE_MUTEX(bpf_percpu_ma_lock);
312 
verbose(void * private_data,const char * fmt,...)313 __printf(2, 3) static void verbose(void *private_data, const char *fmt, ...)
314 {
315 	struct bpf_verifier_env *env = private_data;
316 	va_list args;
317 
318 	if (!bpf_verifier_log_needed(&env->log))
319 		return;
320 
321 	va_start(args, fmt);
322 	bpf_verifier_vlog(&env->log, fmt, args);
323 	va_end(args);
324 }
325 
verbose_invalid_scalar(struct bpf_verifier_env * env,struct bpf_reg_state * reg,struct bpf_retval_range range,const char * ctx,const char * reg_name)326 static void verbose_invalid_scalar(struct bpf_verifier_env *env,
327 				   struct bpf_reg_state *reg,
328 				   struct bpf_retval_range range, const char *ctx,
329 				   const char *reg_name)
330 {
331 	bool unknown = true;
332 
333 	verbose(env, "%s the register %s has", ctx, reg_name);
334 	if (reg_smin(reg) > S64_MIN) {
335 		verbose(env, " smin=%lld", reg_smin(reg));
336 		unknown = false;
337 	}
338 	if (reg_smax(reg) < S64_MAX) {
339 		verbose(env, " smax=%lld", reg_smax(reg));
340 		unknown = false;
341 	}
342 	if (unknown)
343 		verbose(env, " unknown scalar value");
344 	verbose(env, " should have been in [%d, %d]\n", range.minval, range.maxval);
345 }
346 
reg_not_null(struct bpf_verifier_env * env,const struct bpf_reg_state * reg)347 static bool reg_not_null(struct bpf_verifier_env *env, const struct bpf_reg_state *reg)
348 {
349 	enum bpf_reg_type type;
350 
351 	type = reg->type;
352 	if (type_may_be_null(type))
353 		return false;
354 
355 	type = base_type(type);
356 	return type == PTR_TO_SOCKET ||
357 		type == PTR_TO_TCP_SOCK ||
358 		type == PTR_TO_MAP_VALUE ||
359 		type == PTR_TO_MAP_KEY ||
360 		type == PTR_TO_SOCK_COMMON ||
361 		(type == PTR_TO_BTF_ID && is_trusted_reg(env, reg)) ||
362 		(type == PTR_TO_MEM && !(reg->type & PTR_UNTRUSTED)) ||
363 		type == CONST_PTR_TO_MAP;
364 }
365 
reg_btf_record(const struct bpf_reg_state * reg)366 static struct btf_record *reg_btf_record(const struct bpf_reg_state *reg)
367 {
368 	struct btf_record *rec = NULL;
369 	struct btf_struct_meta *meta;
370 
371 	if (reg->type == PTR_TO_MAP_VALUE) {
372 		rec = reg->map_ptr->record;
373 	} else if (type_is_ptr_alloc_obj(reg->type)) {
374 		meta = btf_find_struct_meta(reg->btf, reg->btf_id);
375 		if (meta)
376 			rec = meta->record;
377 	}
378 	return rec;
379 }
380 
bpf_subprog_is_global(const struct bpf_verifier_env * env,int subprog)381 bool bpf_subprog_is_global(const struct bpf_verifier_env *env, int subprog)
382 {
383 	struct bpf_func_info_aux *aux = env->prog->aux->func_info_aux;
384 
385 	return aux && aux[subprog].linkage == BTF_FUNC_GLOBAL;
386 }
387 
subprog_returns_void(struct bpf_verifier_env * env,int subprog)388 static bool subprog_returns_void(struct bpf_verifier_env *env, int subprog)
389 {
390 	const struct btf_type *type, *func, *func_proto;
391 	const struct btf *btf = env->prog->aux->btf;
392 	u32 btf_id;
393 
394 	btf_id = env->prog->aux->func_info[subprog].type_id;
395 
396 	func = btf_type_by_id(btf, btf_id);
397 	if (verifier_bug_if(!func, env, "btf_id %u not found", btf_id))
398 		return false;
399 
400 	func_proto = btf_type_by_id(btf, func->type);
401 	if (!func_proto)
402 		return false;
403 
404 	type = btf_type_skip_modifiers(btf, func_proto->type, NULL);
405 	if (!type)
406 		return false;
407 
408 	return btf_type_is_void(type);
409 }
410 
bpf_subprog_name(const struct bpf_verifier_env * env,int subprog)411 const char *bpf_subprog_name(const struct bpf_verifier_env *env, int subprog)
412 {
413 	struct bpf_func_info *info;
414 
415 	if (!env->prog->aux->func_info)
416 		return "";
417 
418 	info = &env->prog->aux->func_info[subprog];
419 	return btf_type_name(env->prog->aux->btf, info->type_id);
420 }
421 
bpf_mark_subprog_exc_cb(struct bpf_verifier_env * env,int subprog)422 void bpf_mark_subprog_exc_cb(struct bpf_verifier_env *env, int subprog)
423 {
424 	struct bpf_subprog_info *info = subprog_info(env, subprog);
425 
426 	info->is_cb = true;
427 	info->is_async_cb = true;
428 	info->is_exception_cb = true;
429 }
430 
subprog_is_exc_cb(struct bpf_verifier_env * env,int subprog)431 static bool subprog_is_exc_cb(struct bpf_verifier_env *env, int subprog)
432 {
433 	return subprog_info(env, subprog)->is_exception_cb;
434 }
435 
reg_may_point_to_spin_lock(const struct bpf_reg_state * reg)436 static bool reg_may_point_to_spin_lock(const struct bpf_reg_state *reg)
437 {
438 	return btf_record_has_field(reg_btf_record(reg), BPF_SPIN_LOCK | BPF_RES_SPIN_LOCK);
439 }
440 
type_is_rdonly_mem(u32 type)441 static bool type_is_rdonly_mem(u32 type)
442 {
443 	return type & MEM_RDONLY;
444 }
445 
is_acquire_function(enum bpf_func_id func_id,const struct bpf_map * map)446 static bool is_acquire_function(enum bpf_func_id func_id,
447 				const struct bpf_map *map)
448 {
449 	enum bpf_map_type map_type = map ? map->map_type : BPF_MAP_TYPE_UNSPEC;
450 
451 	if (func_id == BPF_FUNC_sk_lookup_tcp ||
452 	    func_id == BPF_FUNC_sk_lookup_udp ||
453 	    func_id == BPF_FUNC_skc_lookup_tcp ||
454 	    func_id == BPF_FUNC_ringbuf_reserve ||
455 	    func_id == BPF_FUNC_kptr_xchg)
456 		return true;
457 
458 	if (func_id == BPF_FUNC_map_lookup_elem &&
459 	    (map_type == BPF_MAP_TYPE_SOCKMAP ||
460 	     map_type == BPF_MAP_TYPE_SOCKHASH))
461 		return true;
462 
463 	return false;
464 }
465 
is_ptr_cast_function(enum bpf_func_id func_id)466 static bool is_ptr_cast_function(enum bpf_func_id func_id)
467 {
468 	return func_id == BPF_FUNC_tcp_sock ||
469 		func_id == BPF_FUNC_sk_fullsock ||
470 		func_id == BPF_FUNC_skc_to_tcp_sock ||
471 		func_id == BPF_FUNC_skc_to_tcp6_sock ||
472 		func_id == BPF_FUNC_skc_to_udp6_sock ||
473 		func_id == BPF_FUNC_skc_to_mptcp_sock ||
474 		func_id == BPF_FUNC_skc_to_tcp_timewait_sock ||
475 		func_id == BPF_FUNC_skc_to_tcp_request_sock;
476 }
477 
478 static bool is_sync_callback_calling_kfunc(u32 btf_id);
479 static bool is_async_callback_calling_kfunc(u32 btf_id);
480 static bool is_callback_calling_kfunc(u32 btf_id);
481 
482 static bool is_bpf_wq_set_callback_kfunc(u32 btf_id);
483 static bool is_task_work_add_kfunc(u32 func_id);
484 
is_sync_callback_calling_function(enum bpf_func_id func_id)485 static bool is_sync_callback_calling_function(enum bpf_func_id func_id)
486 {
487 	return func_id == BPF_FUNC_for_each_map_elem ||
488 	       func_id == BPF_FUNC_find_vma ||
489 	       func_id == BPF_FUNC_loop ||
490 	       func_id == BPF_FUNC_user_ringbuf_drain;
491 }
492 
is_async_callback_calling_function(enum bpf_func_id func_id)493 static bool is_async_callback_calling_function(enum bpf_func_id func_id)
494 {
495 	return func_id == BPF_FUNC_timer_set_callback;
496 }
497 
is_callback_calling_function(enum bpf_func_id func_id)498 static bool is_callback_calling_function(enum bpf_func_id func_id)
499 {
500 	return is_sync_callback_calling_function(func_id) ||
501 	       is_async_callback_calling_function(func_id);
502 }
503 
bpf_is_sync_callback_calling_insn(struct bpf_insn * insn)504 bool bpf_is_sync_callback_calling_insn(struct bpf_insn *insn)
505 {
506 	return (bpf_helper_call(insn) && is_sync_callback_calling_function(insn->imm)) ||
507 	       (bpf_pseudo_kfunc_call(insn) && is_sync_callback_calling_kfunc(insn->imm));
508 }
509 
bpf_is_async_callback_calling_insn(struct bpf_insn * insn)510 bool bpf_is_async_callback_calling_insn(struct bpf_insn *insn)
511 {
512 	return (bpf_helper_call(insn) && is_async_callback_calling_function(insn->imm)) ||
513 	       (bpf_pseudo_kfunc_call(insn) && is_async_callback_calling_kfunc(insn->imm));
514 }
515 
is_async_cb_sleepable(struct bpf_verifier_env * env,struct bpf_insn * insn)516 static bool is_async_cb_sleepable(struct bpf_verifier_env *env, struct bpf_insn *insn)
517 {
518 	/* bpf_timer callbacks are never sleepable. */
519 	if (bpf_helper_call(insn) && insn->imm == BPF_FUNC_timer_set_callback)
520 		return false;
521 
522 	/* bpf_wq and bpf_task_work callbacks are always sleepable. */
523 	if (bpf_pseudo_kfunc_call(insn) && insn->off == 0 &&
524 	    (is_bpf_wq_set_callback_kfunc(insn->imm) || is_task_work_add_kfunc(insn->imm)))
525 		return true;
526 
527 	verifier_bug(env, "unhandled async callback in is_async_cb_sleepable");
528 	return false;
529 }
530 
bpf_is_may_goto_insn(struct bpf_insn * insn)531 bool bpf_is_may_goto_insn(struct bpf_insn *insn)
532 {
533 	return insn->code == (BPF_JMP | BPF_JCOND) && insn->src_reg == BPF_MAY_GOTO;
534 }
535 
is_spi_bounds_valid(struct bpf_func_state * state,int spi,int nr_slots)536 static bool is_spi_bounds_valid(struct bpf_func_state *state, int spi, int nr_slots)
537 {
538        int allocated_slots = state->allocated_stack / BPF_REG_SIZE;
539 
540        /* We need to check that slots between [spi - nr_slots + 1, spi] are
541 	* within [0, allocated_stack).
542 	*
543 	* Please note that the spi grows downwards. For example, a dynptr
544 	* takes the size of two stack slots; the first slot will be at
545 	* spi and the second slot will be at spi - 1.
546 	*/
547        return spi - nr_slots + 1 >= 0 && spi < allocated_slots;
548 }
549 
stack_slot_obj_get_spi(struct bpf_verifier_env * env,struct bpf_reg_state * reg,const char * obj_kind,int nr_slots)550 static int stack_slot_obj_get_spi(struct bpf_verifier_env *env, struct bpf_reg_state *reg,
551 			          const char *obj_kind, int nr_slots)
552 {
553 	int off, spi;
554 
555 	if (!tnum_is_const(reg->var_off)) {
556 		verbose(env, "%s has to be at a constant offset\n", obj_kind);
557 		return -EINVAL;
558 	}
559 
560 	off = reg->var_off.value;
561 	if (off % BPF_REG_SIZE) {
562 		verbose(env, "cannot pass in %s at an offset=%d\n", obj_kind, off);
563 		return -EINVAL;
564 	}
565 
566 	spi = bpf_get_spi(off);
567 	if (spi + 1 < nr_slots) {
568 		verbose(env, "cannot pass in %s at an offset=%d\n", obj_kind, off);
569 		return -EINVAL;
570 	}
571 
572 	if (!is_spi_bounds_valid(bpf_func(env, reg), spi, nr_slots))
573 		return -ERANGE;
574 	return spi;
575 }
576 
dynptr_get_spi(struct bpf_verifier_env * env,struct bpf_reg_state * reg)577 static int dynptr_get_spi(struct bpf_verifier_env *env, struct bpf_reg_state *reg)
578 {
579 	return stack_slot_obj_get_spi(env, reg, "dynptr", BPF_DYNPTR_NR_SLOTS);
580 }
581 
iter_get_spi(struct bpf_verifier_env * env,struct bpf_reg_state * reg,int nr_slots)582 static int iter_get_spi(struct bpf_verifier_env *env, struct bpf_reg_state *reg, int nr_slots)
583 {
584 	return stack_slot_obj_get_spi(env, reg, "iter", nr_slots);
585 }
586 
irq_flag_get_spi(struct bpf_verifier_env * env,struct bpf_reg_state * reg)587 static int irq_flag_get_spi(struct bpf_verifier_env *env, struct bpf_reg_state *reg)
588 {
589 	return stack_slot_obj_get_spi(env, reg, "irq_flag", 1);
590 }
591 
arg_to_dynptr_type(enum bpf_arg_type arg_type)592 static enum bpf_dynptr_type arg_to_dynptr_type(enum bpf_arg_type arg_type)
593 {
594 	switch (arg_type & DYNPTR_TYPE_FLAG_MASK) {
595 	case DYNPTR_TYPE_LOCAL:
596 		return BPF_DYNPTR_TYPE_LOCAL;
597 	case DYNPTR_TYPE_RINGBUF:
598 		return BPF_DYNPTR_TYPE_RINGBUF;
599 	case DYNPTR_TYPE_SKB:
600 		return BPF_DYNPTR_TYPE_SKB;
601 	case DYNPTR_TYPE_XDP:
602 		return BPF_DYNPTR_TYPE_XDP;
603 	case DYNPTR_TYPE_SKB_META:
604 		return BPF_DYNPTR_TYPE_SKB_META;
605 	case DYNPTR_TYPE_FILE:
606 		return BPF_DYNPTR_TYPE_FILE;
607 	default:
608 		return BPF_DYNPTR_TYPE_INVALID;
609 	}
610 }
611 
get_dynptr_type_flag(enum bpf_dynptr_type type)612 static enum bpf_type_flag get_dynptr_type_flag(enum bpf_dynptr_type type)
613 {
614 	switch (type) {
615 	case BPF_DYNPTR_TYPE_LOCAL:
616 		return DYNPTR_TYPE_LOCAL;
617 	case BPF_DYNPTR_TYPE_RINGBUF:
618 		return DYNPTR_TYPE_RINGBUF;
619 	case BPF_DYNPTR_TYPE_SKB:
620 		return DYNPTR_TYPE_SKB;
621 	case BPF_DYNPTR_TYPE_XDP:
622 		return DYNPTR_TYPE_XDP;
623 	case BPF_DYNPTR_TYPE_SKB_META:
624 		return DYNPTR_TYPE_SKB_META;
625 	case BPF_DYNPTR_TYPE_FILE:
626 		return DYNPTR_TYPE_FILE;
627 	default:
628 		return 0;
629 	}
630 }
631 
dynptr_type_referenced(enum bpf_dynptr_type type)632 static bool dynptr_type_referenced(enum bpf_dynptr_type type)
633 {
634 	return type == BPF_DYNPTR_TYPE_RINGBUF || type == BPF_DYNPTR_TYPE_FILE;
635 }
636 
637 static void __mark_dynptr_reg(struct bpf_reg_state *reg,
638 			      enum bpf_dynptr_type type,
639 			      bool first_slot, int id, int parent_id);
640 
mark_dynptr_stack_regs(struct bpf_verifier_env * env,struct bpf_reg_state * sreg1,struct bpf_reg_state * sreg2,enum bpf_dynptr_type type,int parent_id)641 static void mark_dynptr_stack_regs(struct bpf_verifier_env *env,
642 				   struct bpf_reg_state *sreg1,
643 				   struct bpf_reg_state *sreg2,
644 				   enum bpf_dynptr_type type, int parent_id)
645 {
646 	int id = ++env->id_gen;
647 
648 	__mark_dynptr_reg(sreg1, type, true, id, parent_id);
649 	__mark_dynptr_reg(sreg2, type, false, id, parent_id);
650 }
651 
mark_dynptr_cb_reg(struct bpf_verifier_env * env,struct bpf_reg_state * reg,enum bpf_dynptr_type type)652 static void mark_dynptr_cb_reg(struct bpf_verifier_env *env,
653 			       struct bpf_reg_state *reg,
654 			       enum bpf_dynptr_type type)
655 {
656 	__mark_dynptr_reg(reg, type, true, ++env->id_gen, 0);
657 }
658 
659 static int destroy_if_dynptr_stack_slot(struct bpf_verifier_env *env,
660 				        struct bpf_func_state *state, int spi);
661 
mark_stack_slots_dynptr(struct bpf_verifier_env * env,struct bpf_reg_state * reg,enum bpf_arg_type arg_type,int insn_idx,struct ref_obj_desc * ref_obj,struct bpf_dynptr_desc * dynptr)662 static int mark_stack_slots_dynptr(struct bpf_verifier_env *env, struct bpf_reg_state *reg,
663 				   enum bpf_arg_type arg_type, int insn_idx,
664 				   struct ref_obj_desc *ref_obj, struct bpf_dynptr_desc *dynptr)
665 {
666 	struct bpf_func_state *state = bpf_func(env, reg);
667 	int spi, i, err, parent_id = 0;
668 	enum bpf_dynptr_type type;
669 
670 	spi = dynptr_get_spi(env, reg);
671 	if (spi < 0)
672 		return spi;
673 
674 	/* We cannot assume both spi and spi - 1 belong to the same dynptr,
675 	 * hence we need to call destroy_if_dynptr_stack_slot twice for both,
676 	 * to ensure that for the following example:
677 	 *	[d1][d1][d2][d2]
678 	 * spi    3   2   1   0
679 	 * So marking spi = 2 should lead to destruction of both d1 and d2. In
680 	 * case they do belong to same dynptr, second call won't see slot_type
681 	 * as STACK_DYNPTR and will simply skip destruction.
682 	 */
683 	err = destroy_if_dynptr_stack_slot(env, state, spi);
684 	if (err)
685 		return err;
686 	err = destroy_if_dynptr_stack_slot(env, state, spi - 1);
687 	if (err)
688 		return err;
689 
690 	for (i = 0; i < BPF_REG_SIZE; i++) {
691 		state->stack[spi].slot_type[i] = STACK_DYNPTR;
692 		state->stack[spi - 1].slot_type[i] = STACK_DYNPTR;
693 	}
694 
695 	type = arg_to_dynptr_type(arg_type);
696 	if (type == BPF_DYNPTR_TYPE_INVALID)
697 		return -EINVAL;
698 
699 	if (dynptr->type == BPF_DYNPTR_TYPE_INVALID) { /* dynptr constructors */
700 		err = validate_ref_obj(env, ref_obj);
701 		if (err)
702 			return err;
703 
704 		/* Track parent's id if the parent is a referenced object */
705 		parent_id = ref_obj->id;
706 
707 		if (dynptr_type_referenced(type)) {
708 			int id;
709 
710 			/*
711 			 * Create an intermediate reference that tracks the referenced
712 			 * object for the referenced dynptr. Freeing a referenced dynptr
713 			 * through helpers/kfuncs will invalidate all clones.
714 			 */
715 			id = acquire_reference(env, insn_idx, parent_id);
716 			if (id < 0)
717 				return id;
718 
719 			parent_id = id;
720 		}
721 	} else { /* bpf_dynptr_clone() */
722 		parent_id = dynptr->parent_id;
723 	}
724 
725 	mark_dynptr_stack_regs(env, &state->stack[spi].spilled_ptr,
726 			       &state->stack[spi - 1].spilled_ptr, type, parent_id);
727 
728 	return 0;
729 }
730 
invalidate_dynptr(struct bpf_verifier_env * env,struct bpf_stack_state * stack)731 static void invalidate_dynptr(struct bpf_verifier_env *env, struct bpf_stack_state *stack)
732 {
733 	int i;
734 
735 	for (i = 0; i < BPF_REG_SIZE; i++) {
736 		stack[0].slot_type[i] = STACK_INVALID;
737 		stack[1].slot_type[i] = STACK_INVALID;
738 	}
739 
740 	bpf_mark_reg_not_init(env, &stack[0].spilled_ptr);
741 	bpf_mark_reg_not_init(env, &stack[1].spilled_ptr);
742 }
743 
unmark_stack_slots_dynptr(struct bpf_verifier_env * env,struct bpf_reg_state * reg)744 static int unmark_stack_slots_dynptr(struct bpf_verifier_env *env, struct bpf_reg_state *reg)
745 {
746 	struct bpf_func_state *state = bpf_func(env, reg);
747 	int spi;
748 
749 	spi = dynptr_get_spi(env, reg);
750 	if (spi < 0)
751 		return spi;
752 
753 	/*
754 	 * For referenced dynptr, release the parent ref which cascades to
755 	 * all clones and derived slices. For non-referenced dynptr, only
756 	 * the dynptr and slices derived from it will be invalidated.
757 	 */
758 	reg = &state->stack[spi].spilled_ptr;
759 	return release_reference(env, dynptr_type_referenced(reg->dynptr.type)
760 				      ? reg->parent_id
761 				      : reg->id);
762 }
763 
764 static void __mark_reg_unknown(const struct bpf_verifier_env *env,
765 			       struct bpf_reg_state *reg);
766 
mark_reg_invalid(const struct bpf_verifier_env * env,struct bpf_reg_state * reg)767 static void mark_reg_invalid(const struct bpf_verifier_env *env, struct bpf_reg_state *reg)
768 {
769 	if (!env->allow_ptr_leaks)
770 		bpf_mark_reg_not_init(env, reg);
771 	else
772 		__mark_reg_unknown(env, reg);
773 }
774 
dynptr_ref_cnt(struct bpf_verifier_env * env,int v_parent_id)775 static int dynptr_ref_cnt(struct bpf_verifier_env *env, int v_parent_id)
776 {
777 	struct bpf_stack_state *stack;
778 	struct bpf_func_state *state;
779 	struct bpf_reg_state *reg;
780 	int ref_cnt = 0;
781 
782 	bpf_for_each_reg_in_vstate_mask(env->cur_state, state, reg, stack, 1 << STACK_DYNPTR, ({
783 		if (!stack || stack->slot_type[0] != STACK_DYNPTR)
784 			continue;
785 		if (!stack->spilled_ptr.dynptr.first_slot)
786 			continue;
787 		if (stack->spilled_ptr.parent_id == v_parent_id)
788 			ref_cnt++;
789 	}));
790 
791 	return ref_cnt;
792 }
793 
destroy_if_dynptr_stack_slot(struct bpf_verifier_env * env,struct bpf_func_state * state,int spi)794 static int destroy_if_dynptr_stack_slot(struct bpf_verifier_env *env,
795 				        struct bpf_func_state *state, int spi)
796 {
797 	int err = 0;
798 
799 	/* We always ensure that STACK_DYNPTR is never set partially,
800 	 * hence just checking for slot_type[0] is enough. This is
801 	 * different for STACK_SPILL, where it may be only set for
802 	 * 1 byte, so code has to use is_spilled_reg.
803 	 */
804 	if (state->stack[spi].slot_type[0] != STACK_DYNPTR)
805 		return 0;
806 
807 	/* Reposition spi to first slot */
808 	if (!state->stack[spi].spilled_ptr.dynptr.first_slot)
809 		spi = spi + 1;
810 
811 	/*
812 	 * A referenced dynptr can be overwritten only if there is at
813 	 * least one other dynptr sharing the same virtual ref parent,
814 	 * ensuring the reference can still be properly released.
815 	 */
816 	if (dynptr_type_referenced(state->stack[spi].spilled_ptr.dynptr.type) &&
817 	    dynptr_ref_cnt(env, state->stack[spi].spilled_ptr.parent_id) <= 1) {
818 		verbose(env, "cannot overwrite referenced dynptr\n");
819 		bpf_diag_res(
820 			env, env->insn_idx, "referenced dynptr overwrite",
821 			"This stack slot contains a dynptr that owns or protects a referenced resource. Overwriting the last dynptr for that resource would lose the verifier-tracked release path.",
822 			"Release or clone the dynptr so another live dynptr still tracks the referenced resource before overwriting this stack slot.");
823 		return -EINVAL;
824 	}
825 
826 	/* Invalidate the dynptr and any derived slices */
827 	err = release_reference(env, state->stack[spi].spilled_ptr.id);
828 	if (!err) {
829 		mark_stack_slot_scratched(env, spi);
830 		mark_stack_slot_scratched(env, spi - 1);
831 	}
832 
833 	return err;
834 }
835 
is_dynptr_reg_valid_uninit(struct bpf_verifier_env * env,struct bpf_reg_state * reg)836 static bool is_dynptr_reg_valid_uninit(struct bpf_verifier_env *env, struct bpf_reg_state *reg)
837 {
838 	int spi;
839 
840 	if (reg->type == CONST_PTR_TO_DYNPTR)
841 		return false;
842 
843 	spi = dynptr_get_spi(env, reg);
844 
845 	/* -ERANGE (i.e. spi not falling into allocated stack slots) isn't an
846 	 * error because this just means the stack state hasn't been updated yet.
847 	 * We will do check_mem_access to check and update stack bounds later.
848 	 */
849 	if (spi < 0 && spi != -ERANGE)
850 		return false;
851 
852 	/* We don't need to check if the stack slots are marked by previous
853 	 * dynptr initializations because we allow overwriting existing unreferenced
854 	 * STACK_DYNPTR slots, see mark_stack_slots_dynptr which calls
855 	 * destroy_if_dynptr_stack_slot to ensure dynptr objects at the slots we are
856 	 * touching are completely destructed before we reinitialize them for a new
857 	 * one. For referenced ones, destroy_if_dynptr_stack_slot returns an error early
858 	 * instead of delaying it until the end where the user will get "Unreleased
859 	 * reference" error.
860 	 */
861 	return true;
862 }
863 
is_dynptr_reg_valid_init(struct bpf_verifier_env * env,struct bpf_reg_state * reg)864 static bool is_dynptr_reg_valid_init(struct bpf_verifier_env *env, struct bpf_reg_state *reg)
865 {
866 	struct bpf_func_state *state = bpf_func(env, reg);
867 	int i, spi;
868 
869 	/* This already represents first slot of initialized bpf_dynptr.
870 	 *
871 	 * CONST_PTR_TO_DYNPTR already has fixed and var_off as 0 due to
872 	 * check_func_arg_reg_off's logic, so we don't need to check its
873 	 * offset and alignment.
874 	 */
875 	if (reg->type == CONST_PTR_TO_DYNPTR)
876 		return true;
877 
878 	spi = dynptr_get_spi(env, reg);
879 	if (spi < 0)
880 		return false;
881 	if (!state->stack[spi].spilled_ptr.dynptr.first_slot)
882 		return false;
883 
884 	for (i = 0; i < BPF_REG_SIZE; i++) {
885 		if (state->stack[spi].slot_type[i] != STACK_DYNPTR ||
886 		    state->stack[spi - 1].slot_type[i] != STACK_DYNPTR)
887 			return false;
888 	}
889 
890 	return true;
891 }
892 
dynptr_reg_type(struct bpf_verifier_env * env,struct bpf_reg_state * reg)893 static enum bpf_dynptr_type dynptr_reg_type(struct bpf_verifier_env *env, struct bpf_reg_state *reg)
894 {
895 	struct bpf_func_state *state;
896 	int spi;
897 
898 	if (reg->type == CONST_PTR_TO_DYNPTR)
899 		return reg->dynptr.type;
900 
901 	spi = dynptr_get_spi(env, reg);
902 	if (spi < 0)
903 		return BPF_DYNPTR_TYPE_INVALID;
904 	state = bpf_func(env, reg);
905 	return state->stack[spi].spilled_ptr.dynptr.type;
906 }
907 
is_dynptr_type_expected(struct bpf_verifier_env * env,struct bpf_reg_state * reg,enum bpf_arg_type arg_type)908 static bool is_dynptr_type_expected(struct bpf_verifier_env *env, struct bpf_reg_state *reg,
909 				    enum bpf_arg_type arg_type)
910 {
911 	/* ARG_PTR_TO_DYNPTR takes any type of dynptr */
912 	if (arg_type == ARG_PTR_TO_DYNPTR)
913 		return true;
914 
915 	return dynptr_reg_type(env, reg) == arg_to_dynptr_type(arg_type);
916 }
917 
918 static void __mark_reg_known_zero(struct bpf_reg_state *reg);
919 
920 static bool in_rcu_cs(struct bpf_verifier_env *env);
921 
922 static bool is_kfunc_rcu_protected(struct bpf_call_arg_meta *meta);
923 
mark_stack_slots_iter(struct bpf_verifier_env * env,struct bpf_call_arg_meta * meta,struct bpf_reg_state * reg,int insn_idx,struct btf * btf,u32 btf_id,int nr_slots)924 static int mark_stack_slots_iter(struct bpf_verifier_env *env,
925 				 struct bpf_call_arg_meta *meta,
926 				 struct bpf_reg_state *reg, int insn_idx,
927 				 struct btf *btf, u32 btf_id, int nr_slots)
928 {
929 	struct bpf_func_state *state = bpf_func(env, reg);
930 	int spi, i, j, id;
931 
932 	spi = iter_get_spi(env, reg, nr_slots);
933 	if (spi < 0)
934 		return spi;
935 
936 	id = acquire_reference(env, insn_idx, 0);
937 	if (id < 0)
938 		return id;
939 
940 	for (i = 0; i < nr_slots; i++) {
941 		struct bpf_stack_state *slot = &state->stack[spi - i];
942 		struct bpf_reg_state *st = &slot->spilled_ptr;
943 
944 		__mark_reg_known_zero(st);
945 		st->type = PTR_TO_STACK; /* we don't have dedicated reg type */
946 		if (is_kfunc_rcu_protected(meta)) {
947 			if (in_rcu_cs(env))
948 				st->type |= MEM_RCU;
949 			else
950 				st->type |= PTR_UNTRUSTED;
951 		}
952 		st->id = i == 0 ? id : 0;
953 		st->iter.btf = btf;
954 		st->iter.btf_id = btf_id;
955 		st->iter.state = BPF_ITER_STATE_ACTIVE;
956 		st->iter.depth = 0;
957 
958 		for (j = 0; j < BPF_REG_SIZE; j++)
959 			slot->slot_type[j] = STACK_ITER;
960 
961 		mark_stack_slot_scratched(env, spi - i);
962 	}
963 
964 	return 0;
965 }
966 
unmark_stack_slots_iter(struct bpf_verifier_env * env,struct bpf_reg_state * reg,int nr_slots)967 static int unmark_stack_slots_iter(struct bpf_verifier_env *env,
968 				   struct bpf_reg_state *reg, int nr_slots)
969 {
970 	struct bpf_func_state *state = bpf_func(env, reg);
971 	int spi, i, j;
972 
973 	spi = iter_get_spi(env, reg, nr_slots);
974 	if (spi < 0)
975 		return spi;
976 
977 	for (i = 0; i < nr_slots; i++) {
978 		struct bpf_stack_state *slot = &state->stack[spi - i];
979 		struct bpf_reg_state *st = &slot->spilled_ptr;
980 
981 		if (i == 0)
982 			WARN_ON_ONCE(release_reference(env, st->id));
983 
984 		bpf_mark_reg_not_init(env, st);
985 
986 		for (j = 0; j < BPF_REG_SIZE; j++)
987 			slot->slot_type[j] = STACK_INVALID;
988 
989 		mark_stack_slot_scratched(env, spi - i);
990 	}
991 
992 	return 0;
993 }
994 
is_iter_reg_valid_uninit(struct bpf_verifier_env * env,struct bpf_reg_state * reg,int nr_slots)995 static bool is_iter_reg_valid_uninit(struct bpf_verifier_env *env,
996 				     struct bpf_reg_state *reg, int nr_slots)
997 {
998 	struct bpf_func_state *state = bpf_func(env, reg);
999 	int spi, i, j;
1000 
1001 	/* For -ERANGE (i.e. spi not falling into allocated stack slots), we
1002 	 * will do check_mem_access to check and update stack bounds later, so
1003 	 * return true for that case.
1004 	 */
1005 	spi = iter_get_spi(env, reg, nr_slots);
1006 	if (spi == -ERANGE)
1007 		return true;
1008 	if (spi < 0)
1009 		return false;
1010 
1011 	for (i = 0; i < nr_slots; i++) {
1012 		struct bpf_stack_state *slot = &state->stack[spi - i];
1013 
1014 		for (j = 0; j < BPF_REG_SIZE; j++)
1015 			if (slot->slot_type[j] == STACK_ITER)
1016 				return false;
1017 	}
1018 
1019 	return true;
1020 }
1021 
is_iter_reg_valid_init(struct bpf_verifier_env * env,struct bpf_reg_state * reg,struct btf * btf,u32 btf_id,int nr_slots)1022 static int is_iter_reg_valid_init(struct bpf_verifier_env *env, struct bpf_reg_state *reg,
1023 				   struct btf *btf, u32 btf_id, int nr_slots)
1024 {
1025 	struct bpf_func_state *state = bpf_func(env, reg);
1026 	int spi, i, j;
1027 
1028 	spi = iter_get_spi(env, reg, nr_slots);
1029 	if (spi < 0)
1030 		return -EINVAL;
1031 
1032 	for (i = 0; i < nr_slots; i++) {
1033 		struct bpf_stack_state *slot = &state->stack[spi - i];
1034 		struct bpf_reg_state *st = &slot->spilled_ptr;
1035 
1036 		if (st->type & PTR_UNTRUSTED)
1037 			return -EPROTO;
1038 		/* only main (first) slot has id set */
1039 		if (i == 0 && !st->id)
1040 			return -EINVAL;
1041 		if (i != 0 && st->id)
1042 			return -EINVAL;
1043 		if (st->iter.btf != btf || st->iter.btf_id != btf_id)
1044 			return -EINVAL;
1045 
1046 		for (j = 0; j < BPF_REG_SIZE; j++)
1047 			if (slot->slot_type[j] != STACK_ITER)
1048 				return -EINVAL;
1049 	}
1050 
1051 	return 0;
1052 }
1053 
1054 static int acquire_irq_state(struct bpf_verifier_env *env, int insn_idx);
1055 static int release_irq_state(struct bpf_verifier_env *env, int id);
1056 
mark_stack_slot_irq_flag(struct bpf_verifier_env * env,struct bpf_call_arg_meta * meta,struct bpf_reg_state * reg,int insn_idx,int kfunc_class)1057 static int mark_stack_slot_irq_flag(struct bpf_verifier_env *env,
1058 				     struct bpf_call_arg_meta *meta,
1059 				     struct bpf_reg_state *reg, int insn_idx,
1060 				     int kfunc_class)
1061 {
1062 	struct bpf_func_state *state = bpf_func(env, reg);
1063 	struct bpf_stack_state *slot;
1064 	struct bpf_reg_state *st;
1065 	int spi, i, id;
1066 
1067 	spi = irq_flag_get_spi(env, reg);
1068 	if (spi < 0)
1069 		return spi;
1070 
1071 	id = acquire_irq_state(env, insn_idx);
1072 	if (id < 0)
1073 		return id;
1074 
1075 	slot = &state->stack[spi];
1076 	st = &slot->spilled_ptr;
1077 
1078 	__mark_reg_known_zero(st);
1079 	st->type = PTR_TO_STACK; /* we don't have dedicated reg type */
1080 	st->id = id;
1081 	st->irq.kfunc_class = kfunc_class;
1082 
1083 	for (i = 0; i < BPF_REG_SIZE; i++)
1084 		slot->slot_type[i] = STACK_IRQ_FLAG;
1085 
1086 	mark_stack_slot_scratched(env, spi);
1087 	return 0;
1088 }
1089 
unmark_stack_slot_irq_flag(struct bpf_verifier_env * env,struct bpf_reg_state * reg,int kfunc_class)1090 static int unmark_stack_slot_irq_flag(struct bpf_verifier_env *env, struct bpf_reg_state *reg,
1091 				      int kfunc_class)
1092 {
1093 	struct bpf_func_state *state = bpf_func(env, reg);
1094 	struct bpf_stack_state *slot;
1095 	struct bpf_reg_state *st;
1096 	int spi, i, err;
1097 
1098 	spi = irq_flag_get_spi(env, reg);
1099 	if (spi < 0)
1100 		return spi;
1101 
1102 	slot = &state->stack[spi];
1103 	st = &slot->spilled_ptr;
1104 
1105 	if (st->irq.kfunc_class != kfunc_class) {
1106 		const char *flag_kfunc = st->irq.kfunc_class == IRQ_NATIVE_KFUNC ? "native" : "lock";
1107 		const char *used_kfunc = kfunc_class == IRQ_NATIVE_KFUNC ? "native" : "lock";
1108 		const char *reason;
1109 
1110 		verbose(env, "irq flag acquired by %s kfuncs cannot be restored with %s kfuncs\n",
1111 			flag_kfunc, used_kfunc);
1112 		reason = bpf_diag_fmt(env,
1113 				      "This IRQ flag was saved by %s IRQ kfuncs, but the restore call "
1114 			"belongs to the %s IRQ kfunc family. Save and restore operations "
1115 			"must use the same family.",
1116 			flag_kfunc, used_kfunc);
1117 		bpf_diag_irq(env, env->insn_idx, "IRQ flag restore mismatch", reason,
1118 			     "Restore the flag with the matching IRQ restore kfunc for the save "
1119 			     "operation that created it.",
1120 			     bpf_diag_irq_depth(env->cur_state));
1121 		return -EINVAL;
1122 	}
1123 
1124 	err = release_irq_state(env, st->id);
1125 	WARN_ON_ONCE(err && err != -EACCES);
1126 	if (err) {
1127 		int insn_idx = 0;
1128 
1129 		for (int i = 0; i < env->cur_state->acquired_refs; i++) {
1130 			if (env->cur_state->refs[i].id == env->cur_state->active_irq_id) {
1131 				insn_idx = env->cur_state->refs[i].insn_idx;
1132 				break;
1133 			}
1134 		}
1135 
1136 		verbose(env, "cannot restore irq state out of order, expected id=%d acquired at insn_idx=%d\n",
1137 			env->cur_state->active_irq_id, insn_idx);
1138 		bpf_diag_irq(env, env->insn_idx, "IRQ flag restore out of order",
1139 			     "IRQ-disabled regions must be restored in last-in, first-out order, "
1140 			     "but this restore does not match the currently active IRQ flag.",
1141 			     "Restore nested IRQ flags in the reverse order they were saved.",
1142 			     bpf_diag_irq_depth(env->cur_state));
1143 		return err;
1144 	}
1145 
1146 	bpf_mark_reg_not_init(env, st);
1147 
1148 	for (i = 0; i < BPF_REG_SIZE; i++)
1149 		slot->slot_type[i] = STACK_INVALID;
1150 
1151 	mark_stack_slot_scratched(env, spi);
1152 	return 0;
1153 }
1154 
is_irq_flag_reg_valid_uninit(struct bpf_verifier_env * env,struct bpf_reg_state * reg)1155 static bool is_irq_flag_reg_valid_uninit(struct bpf_verifier_env *env, struct bpf_reg_state *reg)
1156 {
1157 	struct bpf_func_state *state = bpf_func(env, reg);
1158 	struct bpf_stack_state *slot;
1159 	int spi, i;
1160 
1161 	/* For -ERANGE (i.e. spi not falling into allocated stack slots), we
1162 	 * will do check_mem_access to check and update stack bounds later, so
1163 	 * return true for that case.
1164 	 */
1165 	spi = irq_flag_get_spi(env, reg);
1166 	if (spi == -ERANGE)
1167 		return true;
1168 	if (spi < 0)
1169 		return false;
1170 
1171 	slot = &state->stack[spi];
1172 
1173 	for (i = 0; i < BPF_REG_SIZE; i++)
1174 		if (slot->slot_type[i] == STACK_IRQ_FLAG)
1175 			return false;
1176 	return true;
1177 }
1178 
is_irq_flag_reg_valid_init(struct bpf_verifier_env * env,struct bpf_reg_state * reg)1179 static int is_irq_flag_reg_valid_init(struct bpf_verifier_env *env, struct bpf_reg_state *reg)
1180 {
1181 	struct bpf_func_state *state = bpf_func(env, reg);
1182 	struct bpf_stack_state *slot;
1183 	struct bpf_reg_state *st;
1184 	int spi, i;
1185 
1186 	spi = irq_flag_get_spi(env, reg);
1187 	if (spi < 0)
1188 		return -EINVAL;
1189 
1190 	slot = &state->stack[spi];
1191 	st = &slot->spilled_ptr;
1192 
1193 	if (!st->id)
1194 		return -EINVAL;
1195 
1196 	for (i = 0; i < BPF_REG_SIZE; i++)
1197 		if (slot->slot_type[i] != STACK_IRQ_FLAG)
1198 			return -EINVAL;
1199 	return 0;
1200 }
1201 
1202 /* Check if given stack slot is "special":
1203  *   - spilled register state (STACK_SPILL);
1204  *   - dynptr state (STACK_DYNPTR);
1205  *   - iter state (STACK_ITER).
1206  *   - irq flag state (STACK_IRQ_FLAG)
1207  */
is_stack_slot_special(const struct bpf_stack_state * stack)1208 static bool is_stack_slot_special(const struct bpf_stack_state *stack)
1209 {
1210 	enum bpf_stack_slot_type type = stack->slot_type[BPF_REG_SIZE - 1];
1211 
1212 	switch (type) {
1213 	case STACK_SPILL:
1214 	case STACK_DYNPTR:
1215 	case STACK_ITER:
1216 	case STACK_IRQ_FLAG:
1217 		return true;
1218 	case STACK_INVALID:
1219 	case STACK_POISON:
1220 	case STACK_MISC:
1221 	case STACK_ZERO:
1222 		return false;
1223 	default:
1224 		WARN_ONCE(1, "unknown stack slot type %d\n", type);
1225 		return true;
1226 	}
1227 }
1228 
1229 /* The reg state of a pointer or a bounded scalar was saved when
1230  * it was spilled to the stack.
1231  */
1232 
1233 /*
1234  * Mark stack slot as STACK_MISC, unless it is already:
1235  * - STACK_INVALID, in which case they are equivalent.
1236  * - STACK_ZERO, in which case we preserve more precise STACK_ZERO.
1237  * - STACK_POISON, which truly forbids access to the slot.
1238  * Regardless of allow_ptr_leaks setting (i.e., privileged or unprivileged
1239  * mode), we won't promote STACK_INVALID to STACK_MISC. In privileged case it is
1240  * unnecessary as both are considered equivalent when loading data and pruning,
1241  * in case of unprivileged mode it will be incorrect to allow reads of invalid
1242  * slots.
1243  */
mark_stack_slot_misc(struct bpf_verifier_env * env,u8 * stype)1244 static void mark_stack_slot_misc(struct bpf_verifier_env *env, u8 *stype)
1245 {
1246 	if (*stype == STACK_ZERO)
1247 		return;
1248 	if (*stype == STACK_INVALID || *stype == STACK_POISON)
1249 		return;
1250 	*stype = STACK_MISC;
1251 }
1252 
scrub_spilled_slot(u8 * stype)1253 static void scrub_spilled_slot(u8 *stype)
1254 {
1255 	if (*stype != STACK_INVALID && *stype != STACK_POISON)
1256 		*stype = STACK_MISC;
1257 }
1258 
1259 /* copy array src of length n * size bytes to dst. dst is reallocated if it's too
1260  * small to hold src. This is different from krealloc since we don't want to preserve
1261  * the contents of dst.
1262  *
1263  * Leaves dst untouched if src is NULL or length is zero. Returns NULL if memory could
1264  * not be allocated.
1265  */
copy_array(void * dst,const void * src,size_t n,size_t size,gfp_t flags)1266 static void *copy_array(void *dst, const void *src, size_t n, size_t size, gfp_t flags)
1267 {
1268 	size_t alloc_bytes;
1269 	void *orig = dst;
1270 	size_t bytes;
1271 
1272 	if (ZERO_OR_NULL_PTR(src))
1273 		goto out;
1274 
1275 	if (unlikely(check_mul_overflow(n, size, &bytes)))
1276 		return NULL;
1277 
1278 	alloc_bytes = max(ksize(orig), kmalloc_size_roundup(bytes));
1279 	dst = krealloc(orig, alloc_bytes, flags);
1280 	if (!dst) {
1281 		kfree(orig);
1282 		return NULL;
1283 	}
1284 
1285 	memcpy(dst, src, bytes);
1286 out:
1287 	return dst ? dst : ZERO_SIZE_PTR;
1288 }
1289 
1290 /* resize an array from old_n items to new_n items. the array is reallocated if it's too
1291  * small to hold new_n items. new items are zeroed out if the array grows.
1292  *
1293  * Contrary to krealloc_array, does not free arr if new_n is zero.
1294  */
realloc_array(void * arr,size_t old_n,size_t new_n,size_t size)1295 static void *realloc_array(void *arr, size_t old_n, size_t new_n, size_t size)
1296 {
1297 	size_t alloc_size;
1298 	void *new_arr;
1299 
1300 	if (!new_n || old_n == new_n)
1301 		goto out;
1302 
1303 	alloc_size = kmalloc_size_roundup(size_mul(new_n, size));
1304 	new_arr = krealloc(arr, alloc_size, GFP_KERNEL_ACCOUNT);
1305 	if (!new_arr) {
1306 		kfree(arr);
1307 		return NULL;
1308 	}
1309 	arr = new_arr;
1310 
1311 	if (new_n > old_n)
1312 		memset(arr + old_n * size, 0, (new_n - old_n) * size);
1313 
1314 out:
1315 	return arr ? arr : ZERO_SIZE_PTR;
1316 }
1317 
copy_reference_state(struct bpf_verifier_state * dst,const struct bpf_verifier_state * src)1318 static int copy_reference_state(struct bpf_verifier_state *dst, const struct bpf_verifier_state *src)
1319 {
1320 	dst->refs = copy_array(dst->refs, src->refs, src->acquired_refs,
1321 			       sizeof(struct bpf_reference_state), GFP_KERNEL_ACCOUNT);
1322 	if (!dst->refs)
1323 		return -ENOMEM;
1324 
1325 	dst->acquired_refs = src->acquired_refs;
1326 	dst->active_locks = src->active_locks;
1327 	dst->active_preempt_locks = src->active_preempt_locks;
1328 	dst->active_rcu_locks = src->active_rcu_locks;
1329 	dst->active_irq_id = src->active_irq_id;
1330 	dst->active_lock_id = src->active_lock_id;
1331 	dst->active_lock_ptr = src->active_lock_ptr;
1332 	return 0;
1333 }
1334 
copy_stack_state(struct bpf_func_state * dst,const struct bpf_func_state * src)1335 static int copy_stack_state(struct bpf_func_state *dst, const struct bpf_func_state *src)
1336 {
1337 	size_t n = src->allocated_stack / BPF_REG_SIZE;
1338 
1339 	dst->stack = copy_array(dst->stack, src->stack, n, sizeof(struct bpf_stack_state),
1340 				GFP_KERNEL_ACCOUNT);
1341 	if (!dst->stack)
1342 		return -ENOMEM;
1343 
1344 	dst->allocated_stack = src->allocated_stack;
1345 
1346 	/* copy stack args state */
1347 	n = src->out_stack_arg_cnt;
1348 	if (n) {
1349 		dst->stack_arg_regs = copy_array(dst->stack_arg_regs, src->stack_arg_regs, n,
1350 						 sizeof(struct bpf_reg_state),
1351 						 GFP_KERNEL_ACCOUNT);
1352 		if (!dst->stack_arg_regs)
1353 			return -ENOMEM;
1354 	}
1355 
1356 	dst->out_stack_arg_cnt = src->out_stack_arg_cnt;
1357 	return 0;
1358 }
1359 
resize_reference_state(struct bpf_verifier_state * state,size_t n)1360 static int resize_reference_state(struct bpf_verifier_state *state, size_t n)
1361 {
1362 	state->refs = realloc_array(state->refs, state->acquired_refs, n,
1363 				    sizeof(struct bpf_reference_state));
1364 	if (!state->refs)
1365 		return -ENOMEM;
1366 
1367 	state->acquired_refs = n;
1368 	return 0;
1369 }
1370 
1371 /* Possibly update state->allocated_stack to be at least size bytes. Also
1372  * possibly update the function's high-water mark in its bpf_subprog_info.
1373  */
grow_stack_state(struct bpf_verifier_env * env,struct bpf_func_state * state,int size)1374 static int grow_stack_state(struct bpf_verifier_env *env, struct bpf_func_state *state, int size)
1375 {
1376 	size_t old_n = state->allocated_stack / BPF_REG_SIZE, n;
1377 
1378 	/* The stack size is always a multiple of BPF_REG_SIZE. */
1379 	size = round_up(size, BPF_REG_SIZE);
1380 	n = size / BPF_REG_SIZE;
1381 
1382 	if (old_n >= n)
1383 		return 0;
1384 
1385 	state->stack = realloc_array(state->stack, old_n, n, sizeof(struct bpf_stack_state));
1386 	if (!state->stack)
1387 		return -ENOMEM;
1388 
1389 	state->allocated_stack = size;
1390 
1391 	/* update known max for given subprogram */
1392 	if (env->subprog_info[state->subprogno].stack_depth < size)
1393 		env->subprog_info[state->subprogno].stack_depth = size;
1394 
1395 	return 0;
1396 }
1397 
grow_stack_arg_slots(struct bpf_verifier_env * env,struct bpf_func_state * state,int cnt)1398 static int grow_stack_arg_slots(struct bpf_verifier_env *env,
1399 				struct bpf_func_state *state, int cnt)
1400 {
1401 	size_t old_n = state->out_stack_arg_cnt;
1402 
1403 	if (old_n >= cnt)
1404 		return 0;
1405 
1406 	state->stack_arg_regs = realloc_array(state->stack_arg_regs, old_n, cnt,
1407 					      sizeof(struct bpf_reg_state));
1408 	if (!state->stack_arg_regs)
1409 		return -ENOMEM;
1410 
1411 	state->out_stack_arg_cnt = cnt;
1412 	return 0;
1413 }
1414 
1415 /* Acquire a pointer id from the env and update the state->refs to include
1416  * this new pointer reference.
1417  * On success, returns a valid pointer id to associate with the register
1418  * On failure, returns a negative errno.
1419  */
acquire_reference_state(struct bpf_verifier_env * env,int insn_idx)1420 static struct bpf_reference_state *acquire_reference_state(struct bpf_verifier_env *env, int insn_idx)
1421 {
1422 	struct bpf_verifier_state *state = env->cur_state;
1423 	int new_ofs = state->acquired_refs;
1424 	int err;
1425 
1426 	err = resize_reference_state(state, state->acquired_refs + 1);
1427 	if (err)
1428 		return NULL;
1429 	state->refs[new_ofs].insn_idx = insn_idx;
1430 
1431 	return &state->refs[new_ofs];
1432 }
1433 
acquire_reference(struct bpf_verifier_env * env,int insn_idx,int parent_id)1434 static int acquire_reference(struct bpf_verifier_env *env, int insn_idx, int parent_id)
1435 {
1436 	struct bpf_reference_state *s;
1437 
1438 	s = acquire_reference_state(env, insn_idx);
1439 	if (!s)
1440 		return -ENOMEM;
1441 	s->type = REF_TYPE_PTR;
1442 	s->id = ++env->id_gen;
1443 	s->parent_id = parent_id;
1444 	bpf_diag_record_ref_acquire(env, insn_idx, s->id);
1445 	return s->id;
1446 }
1447 
acquire_lock_state(struct bpf_verifier_env * env,int insn_idx,enum ref_state_type type,int id,void * ptr)1448 static int acquire_lock_state(struct bpf_verifier_env *env, int insn_idx, enum ref_state_type type,
1449 			      int id, void *ptr)
1450 {
1451 	struct bpf_verifier_state *state = env->cur_state;
1452 	struct bpf_reference_state *s;
1453 
1454 	s = acquire_reference_state(env, insn_idx);
1455 	if (!s)
1456 		return -ENOMEM;
1457 	s->type = type;
1458 	s->id = id;
1459 	s->ptr = ptr;
1460 
1461 	state->active_locks++;
1462 	state->active_lock_id = id;
1463 	state->active_lock_ptr = ptr;
1464 	bpf_diag_record_context(env, insn_idx, BPF_DIAG_CONTEXT_LOCK, true,
1465 				state->active_locks);
1466 	return 0;
1467 }
1468 
acquire_irq_state(struct bpf_verifier_env * env,int insn_idx)1469 static int acquire_irq_state(struct bpf_verifier_env *env, int insn_idx)
1470 {
1471 	struct bpf_verifier_state *state = env->cur_state;
1472 	struct bpf_reference_state *s;
1473 
1474 	s = acquire_reference_state(env, insn_idx);
1475 	if (!s)
1476 		return -ENOMEM;
1477 	s->type = REF_TYPE_IRQ;
1478 	s->id = ++env->id_gen;
1479 
1480 	state->active_irq_id = s->id;
1481 	bpf_diag_record_context(env, insn_idx, BPF_DIAG_CONTEXT_IRQ, true,
1482 				bpf_diag_irq_depth(state));
1483 	return s->id;
1484 }
1485 
release_reference_state(struct bpf_verifier_state * state,int idx)1486 static void release_reference_state(struct bpf_verifier_state *state, int idx)
1487 {
1488 	int last_idx;
1489 	size_t rem;
1490 
1491 	/* IRQ state requires the relative ordering of elements remaining the
1492 	 * same, since it relies on the refs array to behave as a stack, so that
1493 	 * it can detect out-of-order IRQ restore. Hence use memmove to shift
1494 	 * the array instead of swapping the final element into the deleted idx.
1495 	 */
1496 	last_idx = state->acquired_refs - 1;
1497 	rem = state->acquired_refs - idx - 1;
1498 	if (last_idx && idx != last_idx)
1499 		memmove(&state->refs[idx], &state->refs[idx + 1], sizeof(*state->refs) * rem);
1500 	memset(&state->refs[last_idx], 0, sizeof(*state->refs));
1501 	state->acquired_refs--;
1502 	return;
1503 }
1504 
find_reference_state(struct bpf_verifier_state * state,int id)1505 static bool find_reference_state(struct bpf_verifier_state *state, int id)
1506 {
1507 	int i;
1508 
1509 	for (i = 0; i < state->acquired_refs; i++) {
1510 		if (state->refs[i].type != REF_TYPE_PTR)
1511 			continue;
1512 		if (state->refs[i].id == id)
1513 			return true;
1514 	}
1515 
1516 	return false;
1517 }
1518 
reg_is_referenced(struct bpf_verifier_env * env,const struct bpf_reg_state * reg)1519 static bool reg_is_referenced(struct bpf_verifier_env *env, const struct bpf_reg_state *reg)
1520 {
1521 	return find_reference_state(env->cur_state, reg->id);
1522 }
1523 
release_lock_state(struct bpf_verifier_env * env,int type,int id,void * ptr)1524 static int release_lock_state(struct bpf_verifier_env *env, int type, int id, void *ptr)
1525 {
1526 	struct bpf_verifier_state *state = env->cur_state;
1527 	void *prev_ptr = NULL;
1528 	u32 prev_id = 0;
1529 	int i;
1530 
1531 	for (i = 0; i < state->acquired_refs; i++) {
1532 		if (state->refs[i].type == type && state->refs[i].id == id &&
1533 		    state->refs[i].ptr == ptr) {
1534 			release_reference_state(state, i);
1535 			state->active_locks--;
1536 			/* Reassign active lock (id, ptr). */
1537 			state->active_lock_id = prev_id;
1538 			state->active_lock_ptr = prev_ptr;
1539 			bpf_diag_record_context(env, env->insn_idx, BPF_DIAG_CONTEXT_LOCK,
1540 						false, state->active_locks);
1541 			return 0;
1542 		}
1543 		if (state->refs[i].type & REF_TYPE_LOCK_MASK) {
1544 			prev_id = state->refs[i].id;
1545 			prev_ptr = state->refs[i].ptr;
1546 		}
1547 	}
1548 	return -EINVAL;
1549 }
1550 
release_irq_state(struct bpf_verifier_env * env,int id)1551 static int release_irq_state(struct bpf_verifier_env *env, int id)
1552 {
1553 	struct bpf_verifier_state *state = env->cur_state;
1554 	u32 prev_id = 0;
1555 	int i;
1556 
1557 	if (id != state->active_irq_id)
1558 		return -EACCES;
1559 
1560 	for (i = 0; i < state->acquired_refs; i++) {
1561 		if (state->refs[i].type != REF_TYPE_IRQ)
1562 			continue;
1563 		if (state->refs[i].id == id) {
1564 			release_reference_state(state, i);
1565 			state->active_irq_id = prev_id;
1566 			bpf_diag_record_context(env, env->insn_idx, BPF_DIAG_CONTEXT_IRQ,
1567 						false, bpf_diag_irq_depth(state));
1568 			return 0;
1569 		} else {
1570 			prev_id = state->refs[i].id;
1571 		}
1572 	}
1573 	return -EINVAL;
1574 }
1575 
find_lock_state(struct bpf_verifier_state * state,enum ref_state_type type,int id,void * ptr)1576 static struct bpf_reference_state *find_lock_state(struct bpf_verifier_state *state, enum ref_state_type type,
1577 						   int id, void *ptr)
1578 {
1579 	int i;
1580 
1581 	for (i = 0; i < state->acquired_refs; i++) {
1582 		struct bpf_reference_state *s = &state->refs[i];
1583 
1584 		if (!(s->type & type))
1585 			continue;
1586 
1587 		if (s->id == id && s->ptr == ptr)
1588 			return s;
1589 	}
1590 	return NULL;
1591 }
1592 
free_func_state(struct bpf_func_state * state)1593 static void free_func_state(struct bpf_func_state *state)
1594 {
1595 	if (!state)
1596 		return;
1597 	kfree(state->stack_arg_regs);
1598 	kfree(state->stack);
1599 	kfree(state);
1600 }
1601 
bpf_clear_jmp_history(struct bpf_verifier_state * state)1602 void bpf_clear_jmp_history(struct bpf_verifier_state *state)
1603 {
1604 	kfree(state->jmp_history);
1605 	state->jmp_history = NULL;
1606 	state->jmp_history_cnt = 0;
1607 }
1608 
bpf_free_verifier_state(struct bpf_verifier_state * state,bool free_self)1609 void bpf_free_verifier_state(struct bpf_verifier_state *state,
1610 			    bool free_self)
1611 {
1612 	int i;
1613 
1614 	for (i = 0; i <= state->curframe; i++) {
1615 		free_func_state(state->frame[i]);
1616 		state->frame[i] = NULL;
1617 	}
1618 	kfree(state->refs);
1619 	bpf_clear_jmp_history(state);
1620 	if (free_self)
1621 		kfree(state);
1622 }
1623 
1624 /* copy verifier state from src to dst growing dst stack space
1625  * when necessary to accommodate larger src stack
1626  */
copy_func_state(struct bpf_func_state * dst,const struct bpf_func_state * src)1627 static int copy_func_state(struct bpf_func_state *dst,
1628 			   const struct bpf_func_state *src)
1629 {
1630 	memcpy(dst, src, offsetof(struct bpf_func_state, stack));
1631 	/* Instruction accounting is path-local, not part of verifier state. */
1632 	dst->insns_subtotal = 0;
1633 	return copy_stack_state(dst, src);
1634 }
1635 
bpf_copy_verifier_state(struct bpf_verifier_state * dst_state,const struct bpf_verifier_state * src)1636 int bpf_copy_verifier_state(struct bpf_verifier_state *dst_state,
1637 			   const struct bpf_verifier_state *src)
1638 {
1639 	struct bpf_func_state *dst;
1640 	int i, err;
1641 
1642 	dst_state->jmp_history = copy_array(dst_state->jmp_history, src->jmp_history,
1643 					  src->jmp_history_cnt, sizeof(*dst_state->jmp_history),
1644 					  GFP_KERNEL_ACCOUNT);
1645 	if (!dst_state->jmp_history)
1646 		return -ENOMEM;
1647 	dst_state->jmp_history_cnt = src->jmp_history_cnt;
1648 
1649 	/* if dst has more stack frames then src frame, free them, this is also
1650 	 * necessary in case of exceptional exits using bpf_throw.
1651 	 */
1652 	for (i = src->curframe + 1; i <= dst_state->curframe; i++) {
1653 		free_func_state(dst_state->frame[i]);
1654 		dst_state->frame[i] = NULL;
1655 	}
1656 	err = copy_reference_state(dst_state, src);
1657 	if (err)
1658 		return err;
1659 	dst_state->speculative = src->speculative;
1660 	dst_state->in_sleepable = src->in_sleepable;
1661 	dst_state->curframe = src->curframe;
1662 	dst_state->branches = src->branches;
1663 	dst_state->parent = src->parent;
1664 	dst_state->first_insn_idx = src->first_insn_idx;
1665 	dst_state->last_insn_idx = src->last_insn_idx;
1666 	dst_state->dfs_depth = src->dfs_depth;
1667 	dst_state->callback_unroll_depth = src->callback_unroll_depth;
1668 	dst_state->may_goto_depth = src->may_goto_depth;
1669 	dst_state->equal_state = src->equal_state;
1670 	for (i = 0; i <= src->curframe; i++) {
1671 		dst = dst_state->frame[i];
1672 		if (!dst) {
1673 			dst = kzalloc_obj(*dst, GFP_KERNEL_ACCOUNT);
1674 			if (!dst)
1675 				return -ENOMEM;
1676 			dst_state->frame[i] = dst;
1677 		}
1678 		err = copy_func_state(dst, src->frame[i]);
1679 		if (err)
1680 			return err;
1681 	}
1682 	return 0;
1683 }
1684 
state_htab_size(struct bpf_verifier_env * env)1685 static u32 state_htab_size(struct bpf_verifier_env *env)
1686 {
1687 	return env->prog->len;
1688 }
1689 
bpf_explored_state(struct bpf_verifier_env * env,int idx)1690 struct list_head *bpf_explored_state(struct bpf_verifier_env *env, int idx)
1691 {
1692 	struct bpf_verifier_state *cur = env->cur_state;
1693 	struct bpf_func_state *state = cur->frame[cur->curframe];
1694 
1695 	return &env->explored_states[(idx ^ state->callsite) % state_htab_size(env)];
1696 }
1697 
same_callsites(struct bpf_verifier_state * a,struct bpf_verifier_state * b)1698 static bool same_callsites(struct bpf_verifier_state *a, struct bpf_verifier_state *b)
1699 {
1700 	int fr;
1701 
1702 	if (a->curframe != b->curframe)
1703 		return false;
1704 
1705 	for (fr = a->curframe; fr >= 0; fr--)
1706 		if (a->frame[fr]->callsite != b->frame[fr]->callsite)
1707 			return false;
1708 
1709 	return true;
1710 }
1711 
bpf_free_backedges(struct bpf_scc_visit * visit)1712 void bpf_free_backedges(struct bpf_scc_visit *visit)
1713 {
1714 	struct bpf_scc_backedge *backedge, *next;
1715 
1716 	for (backedge = visit->backedges; backedge; backedge = next) {
1717 		bpf_free_verifier_state(&backedge->state, false);
1718 		next = backedge->next;
1719 		kfree(backedge);
1720 	}
1721 	visit->backedges = NULL;
1722 }
1723 
pop_stack(struct bpf_verifier_env * env,int * prev_insn_idx,int * insn_idx,bool pop_log)1724 static int pop_stack(struct bpf_verifier_env *env, int *prev_insn_idx,
1725 		     int *insn_idx, bool pop_log)
1726 {
1727 	struct bpf_verifier_state *cur = env->cur_state;
1728 	struct bpf_verifier_stack_elem *elem, *head = env->head;
1729 	int err;
1730 
1731 	if (env->head == NULL)
1732 		return -ENOENT;
1733 
1734 	if (cur) {
1735 		err = bpf_copy_verifier_state(cur, &head->st);
1736 		if (err)
1737 			return err;
1738 		bpf_diag_event_log_restore(env, head->diag_log_pos);
1739 	}
1740 	if (pop_log)
1741 		bpf_vlog_reset(&env->log, head->log_pos);
1742 	if (insn_idx)
1743 		*insn_idx = head->insn_idx;
1744 	if (prev_insn_idx)
1745 		*prev_insn_idx = head->prev_insn_idx;
1746 	elem = head->next;
1747 	bpf_free_verifier_state(&head->st, false);
1748 	kfree(head);
1749 	env->head = elem;
1750 	env->stack_size--;
1751 	return 0;
1752 }
1753 
error_recoverable_with_nospec(int err)1754 static bool error_recoverable_with_nospec(int err)
1755 {
1756 	/* Should only return true for non-fatal errors that are allowed to
1757 	 * occur during speculative verification. For these we can insert a
1758 	 * nospec and the program might still be accepted. Do not include
1759 	 * something like ENOMEM because it is likely to re-occur for the next
1760 	 * architectural path once it has been recovered-from in all speculative
1761 	 * paths.
1762 	 */
1763 	return err == -EPERM || err == -EACCES || err == -EINVAL;
1764 }
1765 
push_stack(struct bpf_verifier_env * env,int insn_idx,int prev_insn_idx,bool speculative)1766 static struct bpf_verifier_state *push_stack(struct bpf_verifier_env *env,
1767 					     int insn_idx, int prev_insn_idx,
1768 					     bool speculative)
1769 {
1770 	struct bpf_verifier_state *cur = env->cur_state;
1771 	struct bpf_verifier_stack_elem *elem;
1772 	int err;
1773 
1774 	elem = kzalloc_obj(struct bpf_verifier_stack_elem, GFP_KERNEL_ACCOUNT);
1775 	if (!elem)
1776 		return ERR_PTR(-ENOMEM);
1777 
1778 	elem->insn_idx = insn_idx;
1779 	elem->prev_insn_idx = prev_insn_idx;
1780 	elem->next = env->head;
1781 	elem->log_pos = env->log.end_pos;
1782 	elem->diag_log_pos = bpf_diag_event_log_save(env);
1783 	env->head = elem;
1784 	env->stack_size++;
1785 	err = bpf_copy_verifier_state(&elem->st, cur);
1786 	if (err)
1787 		return ERR_PTR(-ENOMEM);
1788 	elem->st.speculative |= speculative;
1789 	if (env->stack_size > BPF_COMPLEXITY_LIMIT_JMP_SEQ) {
1790 		verbose(env, "The sequence of %d jumps is too complex.\n",
1791 			env->stack_size);
1792 		return ERR_PTR(-E2BIG);
1793 	}
1794 	if (elem->st.parent) {
1795 		++elem->st.parent->branches;
1796 		/* WARN_ON(branches > 2) technically makes sense here,
1797 		 * but
1798 		 * 1. speculative states will bump 'branches' for non-branch
1799 		 * instructions
1800 		 * 2. is_state_visited() heuristics may decide not to create
1801 		 * a new state for a sequence of branches and all such current
1802 		 * and cloned states will be pointing to a single parent state
1803 		 * which might have large 'branches' count.
1804 		 */
1805 	}
1806 	return &elem->st;
1807 }
1808 
reg_arg_name(struct bpf_verifier_env * env,argno_t argno)1809 static const char *reg_arg_name(struct bpf_verifier_env *env, argno_t argno)
1810 {
1811 	char *buf = env->tmp_arg_name;
1812 	int len = sizeof(env->tmp_arg_name);
1813 	int arg, regno = reg_from_argno(argno);
1814 
1815 	if (regno >= 0) {
1816 		snprintf(buf, len, "R%d", regno);
1817 	} else {
1818 		arg = arg_from_argno(argno);
1819 		snprintf(buf, len, "*(R11-%u)", (arg - MAX_BPF_FUNC_REG_ARGS) * BPF_REG_SIZE);
1820 	}
1821 
1822 	return buf;
1823 }
1824 
1825 static const int caller_saved[CALLER_SAVED_REGS] = {
1826 	BPF_REG_0, BPF_REG_1, BPF_REG_2, BPF_REG_3, BPF_REG_4, BPF_REG_5
1827 };
1828 
bpf_diag_record_caller_saved(struct bpf_verifier_env * env,struct bpf_reg_state * regs)1829 static void bpf_diag_record_caller_saved(struct bpf_verifier_env *env,
1830 					 struct bpf_reg_state *regs)
1831 {
1832 	int i;
1833 
1834 	for (i = 1; i < CALLER_SAVED_REGS; i++) {
1835 		bpf_diag_record_scrub(env, &regs[caller_saved[i]],
1836 				      BPF_DIAG_MOD_CALLER_SAVED);
1837 	}
1838 }
1839 
1840 /* This helper doesn't clear reg->id */
___mark_reg_known(struct bpf_reg_state * reg,u64 imm)1841 static void ___mark_reg_known(struct bpf_reg_state *reg, u64 imm)
1842 {
1843 	reg->var_off = tnum_const(imm);
1844 	reg->r64 = cnum64_from_urange(imm, imm);
1845 	reg->r32 = cnum32_from_urange((u32)imm, (u32)imm);
1846 }
1847 
1848 /* Mark the unknown part of a register (variable offset or scalar value) as
1849  * known to have the value @imm.
1850  */
__mark_reg_known(struct bpf_reg_state * reg,u64 imm)1851 static void __mark_reg_known(struct bpf_reg_state *reg, u64 imm)
1852 {
1853 	/* Clear off and union(map_ptr, range) */
1854 	memset(((u8 *)reg) + sizeof(reg->type), 0,
1855 	       offsetof(struct bpf_reg_state, var_off) - sizeof(reg->type));
1856 	reg->id = 0;
1857 	reg->parent_id = 0;
1858 	___mark_reg_known(reg, imm);
1859 }
1860 
__mark_reg32_known(struct bpf_reg_state * reg,u64 imm)1861 static void __mark_reg32_known(struct bpf_reg_state *reg, u64 imm)
1862 {
1863 	reg->var_off = tnum_const_subreg(reg->var_off, imm);
1864 	reg->r32 = cnum32_from_urange((u32)imm, (u32)imm);
1865 }
1866 
1867 /* Mark the 'variable offset' part of a register as zero.  This should be
1868  * used only on registers holding a pointer type.
1869  */
__mark_reg_known_zero(struct bpf_reg_state * reg)1870 static void __mark_reg_known_zero(struct bpf_reg_state *reg)
1871 {
1872 	__mark_reg_known(reg, 0);
1873 }
1874 
__mark_reg_const_zero(const struct bpf_verifier_env * env,struct bpf_reg_state * reg)1875 static void __mark_reg_const_zero(const struct bpf_verifier_env *env, struct bpf_reg_state *reg)
1876 {
1877 	__mark_reg_known(reg, 0);
1878 	reg->type = SCALAR_VALUE;
1879 	/* all scalars are assumed imprecise initially (unless unprivileged,
1880 	 * in which case everything is forced to be precise)
1881 	 */
1882 	reg->precise = !env->bpf_capable;
1883 }
1884 
mark_reg_known_zero(struct bpf_verifier_env * env,struct bpf_reg_state * regs,u32 regno)1885 static void mark_reg_known_zero(struct bpf_verifier_env *env,
1886 				struct bpf_reg_state *regs, u32 regno)
1887 {
1888 	__mark_reg_known_zero(regs + regno);
1889 }
1890 
__mark_dynptr_reg(struct bpf_reg_state * reg,enum bpf_dynptr_type type,bool first_slot,int id,int parent_id)1891 static void __mark_dynptr_reg(struct bpf_reg_state *reg, enum bpf_dynptr_type type,
1892 			      bool first_slot, int id, int parent_id)
1893 {
1894 	/* reg->type has no meaning for STACK_DYNPTR, but when we set reg for
1895 	 * callback arguments, it does need to be CONST_PTR_TO_DYNPTR, so simply
1896 	 * set it unconditionally as it is ignored for STACK_DYNPTR anyway.
1897 	 */
1898 	__mark_reg_known_zero(reg);
1899 	reg->type = CONST_PTR_TO_DYNPTR;
1900 	/* Give each dynptr a unique id to uniquely associate slices to it. */
1901 	reg->id = id;
1902 	reg->parent_id = parent_id;
1903 	reg->dynptr.type = type;
1904 	reg->dynptr.first_slot = first_slot;
1905 }
1906 
1907 /*
1908  * Refine the return type of the bpf_map_lookup_elem() for special map types:
1909  * map-in-map, xskmap, sockmap and sockhash.
1910  */
refine_map_lookup_value(struct bpf_reg_state * reg)1911 static void refine_map_lookup_value(struct bpf_reg_state *reg)
1912 {
1913 	enum bpf_type_flag maybe_null = reg->type & PTR_MAYBE_NULL;
1914 	const struct bpf_map *map = reg->map_ptr;
1915 
1916 	if (map->inner_map_meta) {
1917 		reg->type = CONST_PTR_TO_MAP | maybe_null;
1918 		reg->map_ptr = map->inner_map_meta;
1919 		/* transfer reg's id which is unique for every map_lookup_elem
1920 		 * as UID of the inner map.
1921 		 */
1922 		if (btf_record_has_field(map->inner_map_meta->record,
1923 					 BPF_TIMER | BPF_WORKQUEUE | BPF_TASK_WORK))
1924 			reg->map_uid = reg->id;
1925 	} else if (map->map_type == BPF_MAP_TYPE_XSKMAP) {
1926 		reg->type = PTR_TO_XDP_SOCK | maybe_null;
1927 	} else if (map->map_type == BPF_MAP_TYPE_SOCKMAP ||
1928 		   map->map_type == BPF_MAP_TYPE_SOCKHASH) {
1929 		reg->type = PTR_TO_SOCKET | maybe_null;
1930 	}
1931 }
1932 
mark_ptr_not_null_reg(struct bpf_reg_state * reg)1933 static void mark_ptr_not_null_reg(struct bpf_reg_state *reg)
1934 {
1935 	reg->type &= ~PTR_MAYBE_NULL;
1936 }
1937 
mark_reg_graph_node(struct bpf_reg_state * regs,u32 regno,struct btf_field_graph_root * ds_head)1938 static void mark_reg_graph_node(struct bpf_reg_state *regs, u32 regno,
1939 				struct btf_field_graph_root *ds_head)
1940 {
1941 	__mark_reg_known(&regs[regno], ds_head->node_offset);
1942 	regs[regno].type = PTR_TO_BTF_ID | MEM_ALLOC;
1943 	regs[regno].btf = ds_head->btf;
1944 	regs[regno].btf_id = ds_head->value_btf_id;
1945 }
1946 
reg_is_pkt_pointer(const struct bpf_reg_state * reg)1947 static bool reg_is_pkt_pointer(const struct bpf_reg_state *reg)
1948 {
1949 	return type_is_pkt_pointer(reg->type);
1950 }
1951 
reg_is_pkt_pointer_any(const struct bpf_reg_state * reg)1952 static bool reg_is_pkt_pointer_any(const struct bpf_reg_state *reg)
1953 {
1954 	return reg_is_pkt_pointer(reg) ||
1955 	       reg->type == PTR_TO_PACKET_END;
1956 }
1957 
reg_is_dynptr_slice_pkt(const struct bpf_reg_state * reg)1958 static bool reg_is_dynptr_slice_pkt(const struct bpf_reg_state *reg)
1959 {
1960 	return base_type(reg->type) == PTR_TO_MEM &&
1961 	       (reg->type &
1962 		(DYNPTR_TYPE_SKB | DYNPTR_TYPE_XDP | DYNPTR_TYPE_SKB_META));
1963 }
1964 
1965 /* Unmodified PTR_TO_PACKET[_META,_END] register from ctx access. */
reg_is_init_pkt_pointer(const struct bpf_reg_state * reg,enum bpf_reg_type which)1966 static bool reg_is_init_pkt_pointer(const struct bpf_reg_state *reg,
1967 				    enum bpf_reg_type which)
1968 {
1969 	/* The register can already have a range from prior markings.
1970 	 * This is fine as long as it hasn't been advanced from its
1971 	 * origin.
1972 	 */
1973 	return reg->type == which &&
1974 	       reg->id == 0 &&
1975 	       tnum_equals_const(reg->var_off, 0);
1976 }
1977 
__mark_reg32_unbounded(struct bpf_reg_state * reg)1978 static void __mark_reg32_unbounded(struct bpf_reg_state *reg)
1979 {
1980 	reg->r32 = CNUM32_UNBOUNDED;
1981 }
1982 
__mark_reg64_unbounded(struct bpf_reg_state * reg)1983 static void __mark_reg64_unbounded(struct bpf_reg_state *reg)
1984 {
1985 	reg->r64 = CNUM64_UNBOUNDED;
1986 }
1987 
1988 /* Reset the min/max bounds of a register */
__mark_reg_unbounded(struct bpf_reg_state * reg)1989 static void __mark_reg_unbounded(struct bpf_reg_state *reg)
1990 {
1991 	__mark_reg64_unbounded(reg);
1992 	__mark_reg32_unbounded(reg);
1993 }
1994 
reset_reg64_and_tnum(struct bpf_reg_state * reg)1995 static void reset_reg64_and_tnum(struct bpf_reg_state *reg)
1996 {
1997 	__mark_reg64_unbounded(reg);
1998 	reg->var_off = tnum_unknown;
1999 }
2000 
reset_reg32_and_tnum(struct bpf_reg_state * reg)2001 static void reset_reg32_and_tnum(struct bpf_reg_state *reg)
2002 {
2003 	__mark_reg32_unbounded(reg);
2004 	reg->var_off = tnum_unknown;
2005 }
2006 
cnum32_from_tnum(struct tnum tnum)2007 static struct cnum32 cnum32_from_tnum(struct tnum tnum)
2008 {
2009 	tnum = tnum_subreg(tnum);
2010 	if ((tnum.mask & S32_MIN) || (tnum.value & S32_MIN))
2011 		/* min signed is max(sign bit) | min(other bits) */
2012 		/* max signed is min(sign bit) | max(other bits) */
2013 		return cnum32_from_srange(tnum.value | (tnum.mask & S32_MIN),
2014 					  tnum.value | (tnum.mask & S32_MAX));
2015 	else
2016 		return cnum32_from_urange(tnum.value, (tnum.value | tnum.mask));
2017 }
2018 
cnum64_from_tnum(struct tnum tnum)2019 static struct cnum64 cnum64_from_tnum(struct tnum tnum)
2020 {
2021 	if ((tnum.mask & S64_MIN) || (tnum.value & S64_MIN))
2022 		/* min signed is max(sign bit) | min(other bits) */
2023 		/* max signed is min(sign bit) | max(other bits) */
2024 		return cnum64_from_srange(tnum.value | (tnum.mask & S64_MIN),
2025 					  tnum.value | (tnum.mask & S64_MAX));
2026 	else
2027 		return cnum64_from_urange(tnum.value, (tnum.value | tnum.mask));
2028 }
2029 
__update_reg32_bounds(struct bpf_reg_state * reg)2030 static void __update_reg32_bounds(struct bpf_reg_state *reg)
2031 {
2032 	cnum32_intersect_with(&reg->r32, cnum32_from_tnum(reg->var_off));
2033 }
2034 
__update_reg64_bounds(struct bpf_reg_state * reg)2035 static void __update_reg64_bounds(struct bpf_reg_state *reg)
2036 {
2037 	u64 tnum_next, tmax;
2038 	bool umin_in_tnum;
2039 
2040 	cnum64_intersect_with(&reg->r64, cnum64_from_tnum(reg->var_off));
2041 
2042 	/* Check if u64 and tnum overlap in a single value */
2043 	tnum_next = tnum_step(reg->var_off, reg_umin(reg));
2044 	umin_in_tnum = (reg_umin(reg) & ~reg->var_off.mask) == reg->var_off.value;
2045 	tmax = reg->var_off.value | reg->var_off.mask;
2046 	if (umin_in_tnum && tnum_next > reg_umax(reg)) {
2047 		/* The u64 range and the tnum only overlap in umin.
2048 		 * u64:  ---[xxxxxx]-----
2049 		 * tnum: --xx----------x-
2050 		 */
2051 		___mark_reg_known(reg, reg_umin(reg));
2052 	} else if (!umin_in_tnum && tnum_next == tmax) {
2053 		/* The u64 range and the tnum only overlap in the maximum value
2054 		 * represented by the tnum, called tmax.
2055 		 * u64:  ---[xxxxxx]-----
2056 		 * tnum: xx-----x--------
2057 		 */
2058 		___mark_reg_known(reg, tmax);
2059 	} else if (!umin_in_tnum && tnum_next <= reg_umax(reg) &&
2060 		   tnum_step(reg->var_off, tnum_next) > reg_umax(reg)) {
2061 		/* The u64 range and the tnum only overlap in between umin
2062 		 * (excluded) and umax.
2063 		 * u64:  ---[xxxxxx]-----
2064 		 * tnum: xx----x-------x-
2065 		 */
2066 		___mark_reg_known(reg, tnum_next);
2067 	}
2068 }
2069 
__update_reg_bounds(struct bpf_reg_state * reg)2070 static void __update_reg_bounds(struct bpf_reg_state *reg)
2071 {
2072 	__update_reg32_bounds(reg);
2073 	__update_reg64_bounds(reg);
2074 }
2075 
deduce_bounds_32_from_64(struct bpf_reg_state * reg)2076 static void deduce_bounds_32_from_64(struct bpf_reg_state *reg)
2077 {
2078 	cnum32_intersect_with(&reg->r32, cnum32_from_cnum64(reg->r64));
2079 }
2080 
deduce_bounds_64_from_32(struct bpf_reg_state * reg)2081 static void deduce_bounds_64_from_32(struct bpf_reg_state *reg)
2082 {
2083 	reg->r64 = cnum64_cnum32_intersect(reg->r64, reg->r32);
2084 }
2085 
__reg_deduce_bounds(struct bpf_reg_state * reg)2086 static void __reg_deduce_bounds(struct bpf_reg_state *reg)
2087 {
2088 	deduce_bounds_32_from_64(reg);
2089 	deduce_bounds_64_from_32(reg);
2090 }
2091 
2092 /* Attempts to improve var_off based on unsigned min/max information */
__reg_bound_offset(struct bpf_reg_state * reg)2093 static void __reg_bound_offset(struct bpf_reg_state *reg)
2094 {
2095 	struct tnum var64_off = tnum_intersect(reg->var_off,
2096 					       tnum_range(reg_umin(reg),
2097 							  reg_umax(reg)));
2098 	struct tnum var32_off = tnum_intersect(tnum_subreg(var64_off),
2099 					       tnum_range(reg_u32_min(reg),
2100 							  reg_u32_max(reg)));
2101 
2102 	reg->var_off = tnum_or(tnum_clear_subreg(var64_off), var32_off);
2103 }
2104 
2105 static bool range_bounds_violation(struct bpf_reg_state *reg);
2106 
reg_bounds_sync(struct bpf_reg_state * reg)2107 static void reg_bounds_sync(struct bpf_reg_state *reg)
2108 {
2109 	/* If the input reg_state is invalid, we can exit early */
2110 	if (range_bounds_violation(reg))
2111 		return;
2112 	/* We might have learned new bounds from the var_off. */
2113 	__update_reg_bounds(reg);
2114 	/* We might have learned something about the sign bit. */
2115 	__reg_deduce_bounds(reg);
2116 	__reg_deduce_bounds(reg);
2117 	/* We might have learned some bits from the bounds. */
2118 	__reg_bound_offset(reg);
2119 	/* Intersecting with the old var_off might have improved our bounds
2120 	 * slightly, e.g. if umax was 0x7f...f and var_off was (0; 0xf...fc),
2121 	 * then new var_off is (0; 0x7f...fc) which improves our umax.
2122 	 */
2123 	__update_reg_bounds(reg);
2124 }
2125 
const_tnum_range_mismatch(struct bpf_reg_state * reg)2126 static bool const_tnum_range_mismatch(struct bpf_reg_state *reg)
2127 {
2128 	if (!tnum_is_const(reg->var_off))
2129 		return false;
2130 
2131 	return !cnum64_is_const(reg->r64) || reg->r64.base != reg->var_off.value;
2132 }
2133 
const_tnum_range_mismatch_32(struct bpf_reg_state * reg)2134 static bool const_tnum_range_mismatch_32(struct bpf_reg_state *reg)
2135 {
2136 	if (!tnum_subreg_is_const(reg->var_off))
2137 		return false;
2138 
2139 	return !cnum32_is_const(reg->r32) || reg->r32.base != tnum_subreg(reg->var_off).value;
2140 }
2141 
range_bounds_violation(struct bpf_reg_state * reg)2142 static bool range_bounds_violation(struct bpf_reg_state *reg)
2143 {
2144 	return cnum32_is_empty(reg->r32) || cnum64_is_empty(reg->r64);
2145 }
2146 
reg_bounds_sanity_check(struct bpf_verifier_env * env,struct bpf_reg_state * reg,const char * ctx)2147 static int reg_bounds_sanity_check(struct bpf_verifier_env *env,
2148 				   struct bpf_reg_state *reg, const char *ctx)
2149 {
2150 	const char *msg;
2151 
2152 	if (range_bounds_violation(reg)) {
2153 		msg = "range bounds violation";
2154 		goto out;
2155 	}
2156 
2157 	if (const_tnum_range_mismatch(reg)) {
2158 		msg = "const tnum out of sync with range bounds";
2159 		goto out;
2160 	}
2161 
2162 	if (const_tnum_range_mismatch_32(reg)) {
2163 		msg = "const subreg tnum out of sync with range bounds";
2164 		goto out;
2165 	}
2166 
2167 	return 0;
2168 out:
2169 	verifier_bug(env, "REG INVARIANTS VIOLATION (%s): %s r64={.base=%#llx, .size=%#llx} "
2170 		     "r32={.base=%#x, .size=%#x} var_off=(%#llx, %#llx)",
2171 		     ctx, msg,
2172 		     reg->r64.base, reg->r64.size,
2173 		     reg->r32.base, reg->r32.size,
2174 		     reg->var_off.value, reg->var_off.mask);
2175 	if (env->test_reg_invariants)
2176 		return -EFAULT;
2177 	__mark_reg_unbounded(reg);
2178 	return 0;
2179 }
2180 
2181 /* Mark a register as having a completely unknown (scalar) value. */
bpf_mark_reg_unknown_imprecise(struct bpf_reg_state * reg)2182 void bpf_mark_reg_unknown_imprecise(struct bpf_reg_state *reg)
2183 {
2184 	memset(reg, 0, sizeof(*reg));
2185 	reg->type = SCALAR_VALUE;
2186 	reg->var_off = tnum_unknown;
2187 	__mark_reg_unbounded(reg);
2188 }
2189 
2190 /* Mark a register as having a completely unknown (scalar) value,
2191  * initialize .precise as true when not bpf capable.
2192  */
__mark_reg_unknown(const struct bpf_verifier_env * env,struct bpf_reg_state * reg)2193 static void __mark_reg_unknown(const struct bpf_verifier_env *env,
2194 			       struct bpf_reg_state *reg)
2195 {
2196 	bpf_mark_reg_unknown_imprecise(reg);
2197 	reg->precise = !env->bpf_capable;
2198 }
2199 
mark_reg_unknown(struct bpf_verifier_env * env,struct bpf_reg_state * regs,u32 regno)2200 static void mark_reg_unknown(struct bpf_verifier_env *env,
2201 			     struct bpf_reg_state *regs, u32 regno)
2202 {
2203 	__mark_reg_unknown(env, regs + regno);
2204 }
2205 
__mark_reg_s32_range(struct bpf_verifier_env * env,struct bpf_reg_state * regs,u32 regno,s32 s32_min,s32 s32_max)2206 static int __mark_reg_s32_range(struct bpf_verifier_env *env,
2207 				struct bpf_reg_state *regs,
2208 				u32 regno,
2209 				s32 s32_min,
2210 				s32 s32_max)
2211 {
2212 	struct bpf_reg_state *reg = regs + regno;
2213 
2214 	reg_set_srange32(reg,
2215 			 max_t(s32, reg_s32_min(reg), s32_min),
2216 			 min_t(s32, reg_s32_max(reg), s32_max));
2217 	reg_set_srange64(reg,
2218 			 max_t(s64, reg_smin(reg), s32_min),
2219 			 min_t(s64, reg_smax(reg), s32_max));
2220 
2221 	reg_bounds_sync(reg);
2222 
2223 	return reg_bounds_sanity_check(env, reg, "s32_range");
2224 }
2225 
bpf_mark_reg_not_init(const struct bpf_verifier_env * env,struct bpf_reg_state * reg)2226 void bpf_mark_reg_not_init(const struct bpf_verifier_env *env,
2227 			   struct bpf_reg_state *reg)
2228 {
2229 	__mark_reg_unknown(env, reg);
2230 	reg->type = NOT_INIT;
2231 }
2232 
mark_btf_ld_reg(struct bpf_verifier_env * env,struct bpf_reg_state * regs,u32 regno,enum bpf_reg_type reg_type,struct btf * btf,u32 btf_id,enum bpf_type_flag flag)2233 static int mark_btf_ld_reg(struct bpf_verifier_env *env,
2234 			   struct bpf_reg_state *regs, u32 regno,
2235 			   enum bpf_reg_type reg_type,
2236 			   struct btf *btf, u32 btf_id,
2237 			   enum bpf_type_flag flag)
2238 {
2239 	switch (reg_type) {
2240 	case SCALAR_VALUE:
2241 		mark_reg_unknown(env, regs, regno);
2242 		return 0;
2243 	case PTR_TO_BTF_ID:
2244 		mark_reg_known_zero(env, regs, regno);
2245 		regs[regno].type = PTR_TO_BTF_ID | flag;
2246 		regs[regno].btf = btf;
2247 		regs[regno].btf_id = btf_id;
2248 		if (type_may_be_null(flag))
2249 			regs[regno].id = ++env->id_gen;
2250 		return 0;
2251 	case PTR_TO_MEM:
2252 		mark_reg_known_zero(env, regs, regno);
2253 		regs[regno].type = PTR_TO_MEM | flag;
2254 		regs[regno].mem_size = 0;
2255 		return 0;
2256 	default:
2257 		verifier_bug(env, "unexpected reg_type %d in %s\n", reg_type, __func__);
2258 		return -EFAULT;
2259 	}
2260 }
2261 
init_reg_state(struct bpf_verifier_env * env,struct bpf_func_state * state)2262 static void init_reg_state(struct bpf_verifier_env *env,
2263 			   struct bpf_func_state *state)
2264 {
2265 	struct bpf_reg_state *regs = state->regs;
2266 	int i;
2267 
2268 	for (i = 0; i < MAX_BPF_REG; i++) {
2269 		bpf_mark_reg_not_init(env, &regs[i]);
2270 	}
2271 
2272 	/* frame pointer */
2273 	regs[BPF_REG_FP].type = PTR_TO_STACK;
2274 	mark_reg_known_zero(env, regs, BPF_REG_FP);
2275 	regs[BPF_REG_FP].frameno = state->frameno;
2276 }
2277 
retval_range(s32 minval,s32 maxval)2278 static struct bpf_retval_range retval_range(s32 minval, s32 maxval)
2279 {
2280 	/*
2281 	 * return_32bit is set to false by default and set explicitly
2282 	 * by the caller when necessary.
2283 	 */
2284 	return (struct bpf_retval_range){ minval, maxval, false };
2285 }
2286 
init_func_state(struct bpf_verifier_env * env,struct bpf_func_state * state,int callsite,int frameno,int subprogno)2287 static void init_func_state(struct bpf_verifier_env *env,
2288 			    struct bpf_func_state *state,
2289 			    int callsite, int frameno, int subprogno)
2290 {
2291 	state->callsite = callsite;
2292 	state->frameno = frameno;
2293 	bpf_diag_init_frame(env, state);
2294 	state->subprogno = subprogno;
2295 	state->callback_ret_range = retval_range(0, 0);
2296 	init_reg_state(env, state);
2297 	mark_verifier_state_scratched(env);
2298 }
2299 
2300 /* Similar to push_stack(), but for async callbacks */
push_async_cb(struct bpf_verifier_env * env,int insn_idx,int prev_insn_idx,int subprog,bool is_sleepable)2301 static struct bpf_verifier_state *push_async_cb(struct bpf_verifier_env *env,
2302 						int insn_idx, int prev_insn_idx,
2303 						int subprog, bool is_sleepable)
2304 {
2305 	struct bpf_verifier_stack_elem *elem;
2306 	struct bpf_func_state *frame;
2307 
2308 	elem = kzalloc_obj(struct bpf_verifier_stack_elem, GFP_KERNEL_ACCOUNT);
2309 	if (!elem)
2310 		return ERR_PTR(-ENOMEM);
2311 
2312 	elem->insn_idx = insn_idx;
2313 	elem->prev_insn_idx = prev_insn_idx;
2314 	elem->next = env->head;
2315 	elem->log_pos = env->log.end_pos;
2316 	elem->diag_log_pos = bpf_diag_event_log_save(env);
2317 	env->head = elem;
2318 	env->stack_size++;
2319 	if (env->stack_size > BPF_COMPLEXITY_LIMIT_JMP_SEQ) {
2320 		verbose(env,
2321 			"The sequence of %d jumps is too complex for async cb.\n",
2322 			env->stack_size);
2323 		return ERR_PTR(-E2BIG);
2324 	}
2325 	/* Unlike push_stack() do not bpf_copy_verifier_state().
2326 	 * The caller state doesn't matter.
2327 	 * This is async callback. It starts in a fresh stack.
2328 	 * Initialize it similar to do_check_common().
2329 	 */
2330 	elem->st.branches = 1;
2331 	elem->st.in_sleepable = is_sleepable;
2332 	frame = kzalloc_obj(*frame, GFP_KERNEL_ACCOUNT);
2333 	if (!frame)
2334 		return ERR_PTR(-ENOMEM);
2335 	init_func_state(env, frame,
2336 			BPF_MAIN_FUNC /* callsite */,
2337 			0 /* frameno within this callchain */,
2338 			subprog /* subprog number within this prog */);
2339 	elem->st.frame[0] = frame;
2340 	return &elem->st;
2341 }
2342 
cmp_subprogs(const void * a,const void * b)2343 static int cmp_subprogs(const void *a, const void *b)
2344 {
2345 	return ((struct bpf_subprog_info *)a)->start -
2346 	       ((struct bpf_subprog_info *)b)->start;
2347 }
2348 
2349 /* Find subprogram that contains instruction at 'off' */
bpf_find_containing_subprog(struct bpf_verifier_env * env,int off)2350 struct bpf_subprog_info *bpf_find_containing_subprog(struct bpf_verifier_env *env, int off)
2351 {
2352 	struct bpf_subprog_info *vals = env->subprog_info;
2353 	int l, r, m;
2354 
2355 	if (off >= env->prog->len || off < 0 || env->subprog_cnt == 0)
2356 		return NULL;
2357 
2358 	l = 0;
2359 	r = env->subprog_cnt - 1;
2360 	while (l < r) {
2361 		m = l + (r - l + 1) / 2;
2362 		if (vals[m].start <= off)
2363 			l = m;
2364 		else
2365 			r = m - 1;
2366 	}
2367 	return &vals[l];
2368 }
2369 
2370 /* Find subprogram that starts exactly at 'off' */
bpf_find_subprog(struct bpf_verifier_env * env,int off)2371 int bpf_find_subprog(struct bpf_verifier_env *env, int off)
2372 {
2373 	struct bpf_subprog_info *p;
2374 
2375 	p = bpf_find_containing_subprog(env, off);
2376 	if (!p || p->start != off)
2377 		return -ENOENT;
2378 	return p - env->subprog_info;
2379 }
2380 
add_subprog(struct bpf_verifier_env * env,int off)2381 static int add_subprog(struct bpf_verifier_env *env, int off)
2382 {
2383 	int insn_cnt = env->prog->len;
2384 	int ret;
2385 
2386 	if (off >= insn_cnt || off < 0) {
2387 		verbose(env, "call to invalid destination\n");
2388 		return -EINVAL;
2389 	}
2390 	ret = bpf_find_subprog(env, off);
2391 	if (ret >= 0)
2392 		return ret;
2393 	if (env->subprog_cnt >= BPF_MAX_SUBPROGS) {
2394 		verbose(env, "too many subprograms\n");
2395 		return -E2BIG;
2396 	}
2397 	/* determine subprog starts. The end is one before the next starts */
2398 	env->subprog_info[env->subprog_cnt++].start = off;
2399 	sort(env->subprog_info, env->subprog_cnt,
2400 	     sizeof(env->subprog_info[0]), cmp_subprogs, NULL);
2401 	return env->subprog_cnt - 1;
2402 }
2403 
bpf_find_exception_callback_insn_off(struct bpf_verifier_env * env)2404 static int bpf_find_exception_callback_insn_off(struct bpf_verifier_env *env)
2405 {
2406 	struct bpf_prog_aux *aux = env->prog->aux;
2407 	struct btf *btf = aux->btf;
2408 	const struct btf_type *t;
2409 	u32 main_btf_id, id;
2410 	const char *name;
2411 	int ret, i;
2412 
2413 	/* Non-zero func_info_cnt implies valid btf */
2414 	if (!aux->func_info_cnt)
2415 		return 0;
2416 	main_btf_id = aux->func_info[0].type_id;
2417 
2418 	t = btf_type_by_id(btf, main_btf_id);
2419 	if (!t) {
2420 		verbose(env, "invalid btf id for main subprog in func_info\n");
2421 		return -EINVAL;
2422 	}
2423 
2424 	name = btf_find_decl_tag_value(btf, t, -1, "exception_callback:");
2425 	if (IS_ERR(name)) {
2426 		ret = PTR_ERR(name);
2427 		/* If there is no tag present, there is no exception callback */
2428 		if (ret == -ENOENT)
2429 			ret = 0;
2430 		else if (ret == -EEXIST)
2431 			verbose(env, "multiple exception callback tags for main subprog\n");
2432 		return ret;
2433 	}
2434 
2435 	ret = btf_find_by_name_kind(btf, name, BTF_KIND_FUNC);
2436 	if (ret < 0) {
2437 		verbose(env, "exception callback '%s' could not be found in BTF\n", name);
2438 		return ret;
2439 	}
2440 	id = ret;
2441 	t = btf_type_by_id(btf, id);
2442 	if (btf_func_linkage(t) != BTF_FUNC_GLOBAL) {
2443 		verbose(env, "exception callback '%s' must have global linkage\n", name);
2444 		return -EINVAL;
2445 	}
2446 	ret = 0;
2447 	for (i = 0; i < aux->func_info_cnt; i++) {
2448 		if (aux->func_info[i].type_id != id)
2449 			continue;
2450 		ret = aux->func_info[i].insn_off;
2451 		/* Further func_info and subprog checks will also happen
2452 		 * later, so assume this is the right insn_off for now.
2453 		 */
2454 		if (!ret) {
2455 			verbose(env, "invalid exception callback insn_off in func_info: 0\n");
2456 			ret = -EINVAL;
2457 		}
2458 	}
2459 	if (!ret) {
2460 		verbose(env, "exception callback type id not found in func_info\n");
2461 		ret = -EINVAL;
2462 	}
2463 	return ret;
2464 }
2465 
2466 #define MAX_KFUNC_BTFS	256
2467 
2468 struct bpf_kfunc_btf {
2469 	struct btf *btf;
2470 	struct module *module;
2471 	u16 offset;
2472 };
2473 
2474 struct bpf_kfunc_btf_tab {
2475 	struct bpf_kfunc_btf descs[MAX_KFUNC_BTFS];
2476 	u32 nr_descs;
2477 };
2478 
kfunc_desc_cmp_by_id_off(const void * a,const void * b)2479 static int kfunc_desc_cmp_by_id_off(const void *a, const void *b)
2480 {
2481 	const struct bpf_kfunc_desc *d0 = a;
2482 	const struct bpf_kfunc_desc *d1 = b;
2483 
2484 	/* func_id is not greater than BTF_MAX_TYPE */
2485 	return d0->func_id - d1->func_id ?: d0->offset - d1->offset;
2486 }
2487 
kfunc_btf_cmp_by_off(const void * a,const void * b)2488 static int kfunc_btf_cmp_by_off(const void *a, const void *b)
2489 {
2490 	const struct bpf_kfunc_btf *d0 = a;
2491 	const struct bpf_kfunc_btf *d1 = b;
2492 
2493 	return d0->offset - d1->offset;
2494 }
2495 
2496 static struct bpf_kfunc_desc *
find_kfunc_desc(const struct bpf_prog * prog,u32 func_id,u16 offset)2497 find_kfunc_desc(const struct bpf_prog *prog, u32 func_id, u16 offset)
2498 {
2499 	struct bpf_kfunc_desc desc = {
2500 		.func_id = func_id,
2501 		.offset = offset,
2502 	};
2503 	struct bpf_kfunc_desc_tab *tab;
2504 
2505 	tab = prog->aux->kfunc_tab;
2506 	return bsearch(&desc, tab->descs, tab->nr_descs,
2507 		       sizeof(tab->descs[0]), kfunc_desc_cmp_by_id_off);
2508 }
2509 
bpf_get_kfunc_addr(const struct bpf_prog * prog,u32 func_id,u16 btf_fd_idx,u8 ** func_addr)2510 int bpf_get_kfunc_addr(const struct bpf_prog *prog, u32 func_id,
2511 		       u16 btf_fd_idx, u8 **func_addr)
2512 {
2513 	const struct bpf_kfunc_desc *desc;
2514 
2515 	desc = find_kfunc_desc(prog, func_id, btf_fd_idx);
2516 	if (!desc)
2517 		return -EFAULT;
2518 
2519 	*func_addr = (u8 *)desc->addr;
2520 	return 0;
2521 }
2522 
2523 #define BPF_FD_SLOT_BTF	1UL
2524 
fd_slot_set_map(struct bpf_fd_array * slot,struct bpf_map * map)2525 static void fd_slot_set_map(struct bpf_fd_array *slot, struct bpf_map *map)
2526 {
2527 	slot->val = (unsigned long)map;
2528 }
2529 
fd_slot_set_btf(struct bpf_fd_array * slot,struct btf * btf)2530 static void fd_slot_set_btf(struct bpf_fd_array *slot, struct btf *btf)
2531 {
2532 	slot->val = (unsigned long)btf | BPF_FD_SLOT_BTF;
2533 }
2534 
fd_slot_map(struct bpf_fd_array slot)2535 static struct bpf_map *fd_slot_map(struct bpf_fd_array slot)
2536 {
2537 	if (slot.val & BPF_FD_SLOT_BTF)
2538 		return NULL;
2539 	return (struct bpf_map *)slot.val;
2540 }
2541 
fd_slot_btf(struct bpf_fd_array slot)2542 static struct btf *fd_slot_btf(struct bpf_fd_array slot)
2543 {
2544 	if (!(slot.val & BPF_FD_SLOT_BTF))
2545 		return NULL;
2546 	return (struct btf *)(slot.val & ~BPF_FD_SLOT_BTF);
2547 }
2548 
2549 static struct btf *
fd_array_get_btf_continuous(struct bpf_verifier_env * env,u32 idx)2550 fd_array_get_btf_continuous(struct bpf_verifier_env *env, u32 idx)
2551 {
2552 	struct btf *btf;
2553 
2554 	if (idx >= env->fd_array_cnt) {
2555 		verbose(env, "kfunc fd_idx %u out of bounds, fd_array_cnt %u\n",
2556 			idx, env->fd_array_cnt);
2557 		return ERR_PTR(-EINVAL);
2558 	}
2559 	btf = fd_slot_btf(env->fd_array[idx]);
2560 	if (!btf) {
2561 		verbose(env, "kfunc fd_idx %u is not a module BTF\n", idx);
2562 		return ERR_PTR(-EINVAL);
2563 	}
2564 	btf_get(btf);
2565 	return btf;
2566 }
2567 
2568 static struct btf *
fd_array_get_btf_sparse(struct bpf_verifier_env * env,u32 idx)2569 fd_array_get_btf_sparse(struct bpf_verifier_env *env, u32 idx)
2570 {
2571 	struct btf *btf;
2572 	int btf_fd;
2573 
2574 	if (copy_from_bpfptr_offset(&btf_fd, env->fd_array_raw,
2575 				    (size_t)idx * sizeof(btf_fd), sizeof(btf_fd)))
2576 		return ERR_PTR(-EFAULT);
2577 	btf = btf_get_by_fd(btf_fd);
2578 	if (IS_ERR(btf)) {
2579 		verbose(env, "invalid module BTF fd specified\n");
2580 		return btf;
2581 	}
2582 	return btf;
2583 }
2584 
fd_array_get_btf(struct bpf_verifier_env * env,u32 idx)2585 static struct btf *fd_array_get_btf(struct bpf_verifier_env *env, u32 idx)
2586 {
2587 	if (env->signature) {
2588 		verbose(env, "signed program cannot bind any BTF\n");
2589 		return ERR_PTR(-EACCES);
2590 	}
2591 	if (env->fd_array)
2592 		return fd_array_get_btf_continuous(env, idx);
2593 	if (!bpfptr_is_null(env->fd_array_raw))
2594 		return fd_array_get_btf_sparse(env, idx);
2595 
2596 	verbose(env, "kfunc offset > 0 without fd_array is invalid\n");
2597 	return ERR_PTR(-EPROTO);
2598 }
2599 
__find_kfunc_desc_btf(struct bpf_verifier_env * env,s16 offset)2600 static struct btf *__find_kfunc_desc_btf(struct bpf_verifier_env *env,
2601 					 s16 offset)
2602 {
2603 	struct bpf_kfunc_btf kf_btf = { .offset = offset };
2604 	struct bpf_kfunc_btf_tab *tab;
2605 	struct bpf_kfunc_btf *b;
2606 	struct module *mod;
2607 	struct btf *btf;
2608 
2609 	tab = env->prog->aux->kfunc_btf_tab;
2610 	b = bsearch(&kf_btf, tab->descs, tab->nr_descs,
2611 		    sizeof(tab->descs[0]), kfunc_btf_cmp_by_off);
2612 	if (!b) {
2613 		if (tab->nr_descs == MAX_KFUNC_BTFS) {
2614 			verbose(env, "too many different module BTFs\n");
2615 			return ERR_PTR(-E2BIG);
2616 		}
2617 
2618 		btf = fd_array_get_btf(env, offset);
2619 		if (IS_ERR(btf))
2620 			return btf;
2621 		if (!btf_is_module(btf)) {
2622 			verbose(env, "BTF fd for kfunc is not a module BTF\n");
2623 			btf_put(btf);
2624 			return ERR_PTR(-EINVAL);
2625 		}
2626 
2627 		mod = btf_try_get_module(btf);
2628 		if (!mod) {
2629 			btf_put(btf);
2630 			return ERR_PTR(-ENXIO);
2631 		}
2632 
2633 		b = &tab->descs[tab->nr_descs++];
2634 		b->btf = btf;
2635 		b->module = mod;
2636 		b->offset = offset;
2637 
2638 		/* sort() reorders entries by value, so b may no longer point
2639 		 * to the right entry after this
2640 		 */
2641 		sort(tab->descs, tab->nr_descs, sizeof(tab->descs[0]),
2642 		     kfunc_btf_cmp_by_off, NULL);
2643 	} else {
2644 		btf = b->btf;
2645 	}
2646 
2647 	return btf;
2648 }
2649 
bpf_free_kfunc_btf_tab(struct bpf_kfunc_btf_tab * tab)2650 void bpf_free_kfunc_btf_tab(struct bpf_kfunc_btf_tab *tab)
2651 {
2652 	if (!tab)
2653 		return;
2654 
2655 	while (tab->nr_descs--) {
2656 		module_put(tab->descs[tab->nr_descs].module);
2657 		btf_put(tab->descs[tab->nr_descs].btf);
2658 	}
2659 	kfree(tab);
2660 }
2661 
find_kfunc_desc_btf(struct bpf_verifier_env * env,s16 offset)2662 static struct btf *find_kfunc_desc_btf(struct bpf_verifier_env *env, s16 offset)
2663 {
2664 	if (offset) {
2665 		if (offset < 0) {
2666 			/* In the future, this can be allowed to increase limit
2667 			 * of fd index into fd_array, interpreted as u16.
2668 			 */
2669 			verbose(env, "negative offset disallowed for kernel module function call\n");
2670 			return ERR_PTR(-EINVAL);
2671 		}
2672 
2673 		return __find_kfunc_desc_btf(env, offset);
2674 	}
2675 	return btf_vmlinux ?: ERR_PTR(-ENOENT);
2676 }
2677 
find_kfunc_desc_btf_cached(struct bpf_verifier_env * env,s16 offset)2678 static struct btf *find_kfunc_desc_btf_cached(struct bpf_verifier_env *env, s16 offset)
2679 {
2680 	struct bpf_kfunc_btf kf_btf = { .offset = offset };
2681 	struct bpf_kfunc_btf_tab *tab;
2682 	struct bpf_kfunc_btf *b;
2683 
2684 	if (!offset)
2685 		return btf_vmlinux ?: ERR_PTR(-ENOENT);
2686 	if (offset < 0)
2687 		return ERR_PTR(-EINVAL);
2688 
2689 	tab = env->prog->aux->kfunc_btf_tab;
2690 	if (!tab)
2691 		return ERR_PTR(-ENOENT);
2692 
2693 	b = bsearch(&kf_btf, tab->descs, tab->nr_descs,
2694 		    sizeof(tab->descs[0]), kfunc_btf_cmp_by_off);
2695 	return b ? b->btf : ERR_PTR(-ENOENT);
2696 }
2697 
2698 #define KF_IMPL_SUFFIX "_impl"
2699 
find_kfunc_impl_proto(struct bpf_verifier_log * log,struct btf * btf,const char * func_name)2700 static const struct btf_type *find_kfunc_impl_proto(struct bpf_verifier_log *log,
2701 						    struct btf *btf,
2702 						    const char *func_name)
2703 {
2704 	const struct btf_type *func;
2705 	char buf[KSYM_NAME_LEN];
2706 	s32 impl_id;
2707 	int len;
2708 
2709 	len = snprintf(buf, sizeof(buf), "%s%s", func_name, KF_IMPL_SUFFIX);
2710 	if (len < 0 || len >= sizeof(buf)) {
2711 		bpf_log(log, "function name %s%s is too long\n",
2712 			func_name, KF_IMPL_SUFFIX);
2713 		return NULL;
2714 	}
2715 
2716 	impl_id = btf_find_by_name_kind(btf, buf, BTF_KIND_FUNC);
2717 	if (impl_id <= 0) {
2718 		bpf_log(log, "cannot find function %s in BTF\n", buf);
2719 		return NULL;
2720 	}
2721 
2722 	func = btf_type_by_id(btf, impl_id);
2723 
2724 	return btf_type_by_id(btf, func->type);
2725 }
2726 
fetch_kfunc_meta(struct bpf_verifier_env * env,s32 func_id,s16 offset,struct bpf_kfunc_meta * kfunc)2727 static int fetch_kfunc_meta(struct bpf_verifier_env *env,
2728 			    s32 func_id,
2729 			    s16 offset,
2730 			    struct bpf_kfunc_meta *kfunc)
2731 {
2732 	const struct btf_type *func, *func_proto;
2733 	const char *func_name;
2734 	u32 *kfunc_flags;
2735 	struct btf *btf;
2736 
2737 	if (func_id <= 0) {
2738 		verbose(env, "invalid kernel function btf_id %d\n", func_id);
2739 		return -EINVAL;
2740 	}
2741 
2742 	btf = find_kfunc_desc_btf(env, offset);
2743 	if (IS_ERR(btf)) {
2744 		verbose(env, "failed to find BTF for kernel function\n");
2745 		return PTR_ERR(btf);
2746 	}
2747 
2748 	/*
2749 	 * Note that kfunc_flags may be NULL at this point, which
2750 	 * means that we couldn't find func_id in any relevant
2751 	 * kfunc_id_set. This most likely indicates an invalid kfunc
2752 	 * call.  However we don't fail with an error here,
2753 	 * and let the caller decide what to do with NULL kfunc->flags.
2754 	 */
2755 	kfunc_flags = btf_kfunc_flags(btf, func_id, env->prog);
2756 
2757 	func = btf_type_by_id(btf, func_id);
2758 	if (!func || !btf_type_is_func(func)) {
2759 		verbose(env, "kernel btf_id %d is not a function\n", func_id);
2760 		return -EINVAL;
2761 	}
2762 
2763 	func_name = btf_name_by_offset(btf, func->name_off);
2764 
2765 	/*
2766 	 * An actual prototype of a kfunc with KF_IMPLICIT_ARGS flag
2767 	 * can be found through the counterpart _impl kfunc.
2768 	 */
2769 	if (kfunc_flags && (*kfunc_flags & KF_IMPLICIT_ARGS))
2770 		func_proto = find_kfunc_impl_proto(&env->log, btf, func_name);
2771 	else
2772 		func_proto = btf_type_by_id(btf, func->type);
2773 
2774 	if (!func_proto || !btf_type_is_func_proto(func_proto)) {
2775 		verbose(env, "kernel function btf_id %d does not have a valid func_proto\n",
2776 			func_id);
2777 		return -EINVAL;
2778 	}
2779 
2780 	memset(kfunc, 0, sizeof(*kfunc));
2781 	kfunc->btf = btf;
2782 	kfunc->id = func_id;
2783 	kfunc->name = func_name;
2784 	kfunc->proto = func_proto;
2785 	kfunc->flags = kfunc_flags;
2786 
2787 	return 0;
2788 }
2789 
2790 static int gen_kfunc_arg_proto(struct bpf_verifier_env *env, struct bpf_call_arg_meta *meta,
2791 			       struct bpf_func_proto *proto);
2792 
bpf_add_kfunc_call(struct bpf_verifier_env * env,u32 func_id,u16 offset)2793 int bpf_add_kfunc_call(struct bpf_verifier_env *env, u32 func_id, u16 offset)
2794 {
2795 	struct bpf_call_arg_meta meta;
2796 	struct bpf_kfunc_btf_tab *btf_tab;
2797 	struct btf_func_model func_model;
2798 	struct bpf_kfunc_desc_tab *tab;
2799 	struct bpf_prog_aux *prog_aux;
2800 	struct bpf_kfunc_meta kfunc;
2801 	struct bpf_kfunc_desc *desc;
2802 	unsigned long addr;
2803 	int err;
2804 
2805 	prog_aux = env->prog->aux;
2806 	tab = prog_aux->kfunc_tab;
2807 	btf_tab = prog_aux->kfunc_btf_tab;
2808 	if (!tab) {
2809 		if (!btf_vmlinux) {
2810 			verbose(env, "calling kernel function is not supported without CONFIG_DEBUG_INFO_BTF\n");
2811 			return -ENOTSUPP;
2812 		}
2813 
2814 		if (!env->prog->jit_requested) {
2815 			verbose(env, "JIT is required for calling kernel function\n");
2816 			return -ENOTSUPP;
2817 		}
2818 
2819 		if (!bpf_jit_supports_kfunc_call()) {
2820 			verbose(env, "JIT does not support calling kernel function\n");
2821 			return -ENOTSUPP;
2822 		}
2823 
2824 		if (!env->prog->gpl_compatible) {
2825 			verbose(env, "cannot call kernel function from non-GPL compatible program\n");
2826 			return -EINVAL;
2827 		}
2828 
2829 		tab = kzalloc_obj(*tab, GFP_KERNEL_ACCOUNT);
2830 		if (!tab)
2831 			return -ENOMEM;
2832 		prog_aux->kfunc_tab = tab;
2833 	}
2834 
2835 	env->prog->jit_required = 1;
2836 
2837 	/* func_id == 0 is always invalid, but instead of returning an error, be
2838 	 * conservative and wait until the code elimination pass before returning
2839 	 * error, so that invalid calls that get pruned out can be in BPF programs
2840 	 * loaded from userspace.  It is also required that offset be untouched
2841 	 * for such calls.
2842 	 */
2843 	if (!func_id && !offset)
2844 		return 0;
2845 
2846 	if (!btf_tab && offset) {
2847 		btf_tab = kzalloc_obj(*btf_tab, GFP_KERNEL_ACCOUNT);
2848 		if (!btf_tab)
2849 			return -ENOMEM;
2850 		prog_aux->kfunc_btf_tab = btf_tab;
2851 	}
2852 
2853 	if (find_kfunc_desc(env->prog, func_id, offset))
2854 		return 0;
2855 
2856 	if (tab->nr_descs == MAX_KFUNC_DESCS) {
2857 		verbose(env, "too many different kernel function calls\n");
2858 		return -E2BIG;
2859 	}
2860 
2861 	err = fetch_kfunc_meta(env, func_id, offset, &kfunc);
2862 	if (err)
2863 		return err;
2864 
2865 	addr = kallsyms_lookup_name(kfunc.name);
2866 	if (!addr) {
2867 		verbose(env, "cannot find address for kernel function %s\n", kfunc.name);
2868 		return -EINVAL;
2869 	}
2870 
2871 	if (bpf_dev_bound_kfunc_id(func_id)) {
2872 		err = bpf_dev_bound_kfunc_check(&env->log, prog_aux);
2873 		if (err)
2874 			return err;
2875 	}
2876 
2877 	err = btf_distill_func_proto(&env->log, kfunc.btf, kfunc.proto, kfunc.name, &func_model);
2878 	if (err)
2879 		return err;
2880 
2881 	memset(&meta, 0, sizeof(meta));
2882 	meta.btf = kfunc.btf;
2883 	meta.func_id = kfunc.id;
2884 	meta.func_proto = kfunc.proto;
2885 	meta.func_name = kfunc.name;
2886 	meta.kfunc_flags = kfunc.flags ? *kfunc.flags : 0;
2887 
2888 	tab = krealloc(tab, struct_size(tab, descs, tab->nr_descs + 1), GFP_KERNEL_ACCOUNT);
2889 	if (!tab)
2890 		return -ENOMEM;
2891 	prog_aux->kfunc_tab = tab;
2892 
2893 	desc = &tab->descs[tab->nr_descs];
2894 	memset(desc, 0, sizeof(*desc));
2895 
2896 	err = gen_kfunc_arg_proto(env, &meta, &desc->proto);
2897 	if (err)
2898 		return err;
2899 
2900 	desc->func_id = func_id;
2901 	desc->offset = offset;
2902 	desc->addr = addr;
2903 	desc->func_model = func_model;
2904 	tab->nr_descs++;
2905 	sort(tab->descs, tab->nr_descs, sizeof(tab->descs[0]),
2906 	     kfunc_desc_cmp_by_id_off, NULL);
2907 	return 0;
2908 }
2909 
add_subprogs(struct bpf_verifier_env * env)2910 static int add_subprogs(struct bpf_verifier_env *env)
2911 {
2912 	struct bpf_subprog_info *subprog = env->subprog_info;
2913 	int i, ret, insn_cnt = env->prog->len, ex_cb_insn;
2914 	struct bpf_insn *insn = env->prog->insnsi;
2915 	const char *operation, *suggestion;
2916 
2917 	/* Add entry function. */
2918 	ret = add_subprog(env, 0);
2919 	if (ret)
2920 		return ret;
2921 
2922 	for (i = 0; i < insn_cnt; i++, insn++) {
2923 		if (!bpf_pseudo_func(insn) && !bpf_pseudo_call(insn))
2924 			continue;
2925 
2926 		if (!env->bpf_capable) {
2927 			if (bpf_pseudo_func(insn)) {
2928 				operation = "BPF function reference";
2929 				suggestion = "Load this program with the required capability, or avoid BPF function references in unprivileged programs.";
2930 			} else {
2931 				operation = "BPF-to-BPF function call";
2932 				suggestion = "Load this program with the required capability, or avoid BPF-to-BPF function calls in unprivileged programs.";
2933 			}
2934 			verbose(env, "loading/calling other bpf or kernel functions are allowed for CAP_BPF and CAP_SYS_ADMIN\n");
2935 			bpf_diag_policy(
2936 				env, i, operation,
2937 				"loading or calling other BPF functions requires CAP_BPF or CAP_SYS_ADMIN",
2938 				suggestion);
2939 			return -EPERM;
2940 		}
2941 
2942 		ret = add_subprog(env, i + insn->imm + 1);
2943 		if (ret < 0)
2944 			return ret;
2945 	}
2946 
2947 	ret = bpf_find_exception_callback_insn_off(env);
2948 	if (ret < 0)
2949 		return ret;
2950 	ex_cb_insn = ret;
2951 
2952 	/* If ex_cb_insn > 0, this means that the main program has a subprog
2953 	 * marked using BTF decl tag to serve as the exception callback.
2954 	 */
2955 	if (ex_cb_insn) {
2956 		ret = add_subprog(env, ex_cb_insn);
2957 		if (ret < 0)
2958 			return ret;
2959 		for (i = 1; i < env->subprog_cnt; i++) {
2960 			if (env->subprog_info[i].start != ex_cb_insn)
2961 				continue;
2962 			env->exception_callback_subprog = i;
2963 			bpf_mark_subprog_exc_cb(env, i);
2964 			break;
2965 		}
2966 	}
2967 
2968 	/* Add a fake 'exit' subprog which could simplify subprog iteration
2969 	 * logic. 'subprog_cnt' should not be increased.
2970 	 */
2971 	subprog[env->subprog_cnt].start = insn_cnt;
2972 
2973 	if (env->log.level & BPF_LOG_LEVEL2)
2974 		for (i = 0; i < env->subprog_cnt; i++)
2975 			verbose(env, "func#%d @%d\n", i, subprog[i].start);
2976 
2977 	return 0;
2978 }
2979 
add_kfuncs(struct bpf_verifier_env * env)2980 static int add_kfuncs(struct bpf_verifier_env *env)
2981 {
2982 	struct bpf_insn *insn = env->prog->insnsi;
2983 	int i, ret, insn_cnt = env->prog->len;
2984 
2985 	for (i = 0; i < insn_cnt; i++, insn++) {
2986 		if (!bpf_pseudo_kfunc_call(insn))
2987 			continue;
2988 
2989 		if (!env->bpf_capable) {
2990 			verbose(env, "loading/calling other bpf or kernel functions are allowed for CAP_BPF and CAP_SYS_ADMIN\n");
2991 			bpf_diag_policy(
2992 				env, i, "kernel function call",
2993 				"calling kernel functions requires CAP_BPF or CAP_SYS_ADMIN",
2994 				"Load this program with the required capability, or avoid kernel function calls in unprivileged programs.");
2995 			return -EPERM;
2996 		}
2997 
2998 		ret = bpf_add_kfunc_call(env, insn->imm, insn->off);
2999 		if (ret < 0)
3000 			return ret;
3001 	}
3002 
3003 	return 0;
3004 }
3005 
check_subprogs(struct bpf_verifier_env * env)3006 static int check_subprogs(struct bpf_verifier_env *env)
3007 {
3008 	int i, subprog_start, subprog_end, off, cur_subprog = 0;
3009 	struct bpf_subprog_info *subprog = env->subprog_info;
3010 	struct bpf_insn *insn = env->prog->insnsi;
3011 	int insn_cnt = env->prog->len;
3012 
3013 	/* now check that all jumps are within the same subprog */
3014 	subprog_start = subprog[cur_subprog].start;
3015 	subprog_end = subprog[cur_subprog + 1].start;
3016 	for (i = 0; i < insn_cnt; i++) {
3017 		u8 code = insn[i].code;
3018 
3019 		if (code == (BPF_JMP | BPF_CALL) &&
3020 		    insn[i].src_reg == 0 &&
3021 		    insn[i].imm == BPF_FUNC_tail_call) {
3022 			subprog[cur_subprog].has_tail_call = true;
3023 			subprog[cur_subprog].tail_call_reachable = true;
3024 		}
3025 		if (BPF_CLASS(code) == BPF_LD &&
3026 		    (BPF_MODE(code) == BPF_ABS || BPF_MODE(code) == BPF_IND))
3027 			subprog[cur_subprog].has_ld_abs = true;
3028 		if (BPF_CLASS(code) != BPF_JMP && BPF_CLASS(code) != BPF_JMP32)
3029 			goto next;
3030 		if (BPF_OP(code) == BPF_CALL)
3031 			goto next;
3032 		if (BPF_OP(code) == BPF_EXIT) {
3033 			subprog[cur_subprog].exit_idx = i;
3034 			goto next;
3035 		}
3036 		off = i + bpf_jmp_offset(&insn[i]) + 1;
3037 		if (off < subprog_start || off >= subprog_end) {
3038 			verbose(env, "jump out of range from insn %d to %d\n", i, off);
3039 			bpf_diag_program_structure(
3040 				env, i, "jump out of range",
3041 				"Keep branch targets within the same subprogram, or use an explicit subprogram call.",
3042 				"Instruction %d jumps to instruction %d, but subprogram %d only contains instructions %d through %d. "
3043 				"A branch target must stay inside the same subprogram.",
3044 				i, off, cur_subprog, subprog_start, subprog_end - 1);
3045 			return -EINVAL;
3046 		}
3047 next:
3048 		if (i == subprog_end - 1) {
3049 			/* to avoid fall-through from one subprog into another
3050 			 * the last insn of the subprog should be either exit
3051 			 * or unconditional jump back or bpf_throw call
3052 			 */
3053 			if (code != (BPF_JMP | BPF_EXIT) &&
3054 			    code != (BPF_JMP32 | BPF_JA) &&
3055 			    code != (BPF_JMP | BPF_JA)) {
3056 				verbose(env, "last insn is not an exit or jmp\n");
3057 				bpf_diag_program_structure(
3058 					env, i, "subprogram can fall through",
3059 					"End each subprogram with an exit or an explicit jump that keeps control flow inside the subprogram.",
3060 					"Subprogram %d reaches its last instruction %d without an exit or jump, so control could continue into the next subprogram.",
3061 					cur_subprog, i);
3062 				return -EINVAL;
3063 			}
3064 			subprog_start = subprog_end;
3065 			cur_subprog++;
3066 			if (cur_subprog < env->subprog_cnt)
3067 				subprog_end = subprog[cur_subprog + 1].start;
3068 		}
3069 	}
3070 	return 0;
3071 }
3072 
3073 /*
3074  * Sort subprogs in topological order so that leaf subprogs come first and
3075  * their callers come later. This is a DFS post-order traversal of the call
3076  * graph. Scan only reachable instructions (those in the computed postorder) of
3077  * the current subprog to discover callees (direct subprogs and sync
3078  * callbacks).
3079  */
sort_subprogs_topo(struct bpf_verifier_env * env)3080 static int sort_subprogs_topo(struct bpf_verifier_env *env)
3081 {
3082 	struct bpf_subprog_info *si = env->subprog_info;
3083 	int *insn_postorder = env->cfg.insn_postorder;
3084 	struct bpf_insn *insn = env->prog->insnsi;
3085 	int cnt = env->subprog_cnt;
3086 	int *dfs_stack = NULL;
3087 	int top = 0, order = 0;
3088 	int i, ret = 0;
3089 	u8 *color = NULL;
3090 
3091 	color = kvzalloc_objs(*color, cnt, GFP_KERNEL_ACCOUNT);
3092 	dfs_stack = kvmalloc_objs(*dfs_stack, cnt, GFP_KERNEL_ACCOUNT);
3093 	if (!color || !dfs_stack) {
3094 		ret = -ENOMEM;
3095 		goto out;
3096 	}
3097 
3098 	/*
3099 	 * DFS post-order traversal.
3100 	 * Color values: 0 = unvisited, 1 = on stack, 2 = done.
3101 	 */
3102 	for (i = 0; i < cnt; i++) {
3103 		if (color[i])
3104 			continue;
3105 		color[i] = 1;
3106 		dfs_stack[top++] = i;
3107 
3108 		while (top > 0) {
3109 			int cur = dfs_stack[top - 1];
3110 			int po_start = si[cur].postorder_start;
3111 			int po_end = si[cur + 1].postorder_start;
3112 			bool pushed = false;
3113 			int j;
3114 
3115 			for (j = po_start; j < po_end; j++) {
3116 				int idx = insn_postorder[j];
3117 				int callee;
3118 
3119 				if (!bpf_pseudo_call(&insn[idx]) && !bpf_pseudo_func(&insn[idx]))
3120 					continue;
3121 				callee = bpf_find_subprog(env, idx + insn[idx].imm + 1);
3122 				if (callee < 0) {
3123 					ret = -EFAULT;
3124 					goto out;
3125 				}
3126 				if (color[callee] == 2)
3127 					continue;
3128 				if (color[callee] == 1) {
3129 					if (bpf_pseudo_func(&insn[idx]))
3130 						continue;
3131 					verbose(env, "recursive call from %s() to %s()\n",
3132 						bpf_subprog_name(env, cur),
3133 						bpf_subprog_name(env, callee));
3134 					bpf_diag_program_structure(
3135 						env, idx, "recursive subprogram call",
3136 						"Rewrite the recursion as an explicit bounded loop, or split the logic so subprogram calls do not form a cycle.",
3137 						"This bpf2bpf call would make the subprogram call graph recursive. "
3138 						"The verifier requires a finite, acyclic call graph so it can bound stack depth and analysis.");
3139 					ret = -EINVAL;
3140 					goto out;
3141 				}
3142 				color[callee] = 1;
3143 				dfs_stack[top++] = callee;
3144 				pushed = true;
3145 				break;
3146 			}
3147 
3148 			if (!pushed) {
3149 				color[cur] = 2;
3150 				env->subprog_topo_order[order++] = cur;
3151 				top--;
3152 			}
3153 		}
3154 	}
3155 
3156 	if (env->log.level & BPF_LOG_LEVEL2)
3157 		for (i = 0; i < cnt; i++)
3158 			verbose(env, "topo_order[%d] = %s\n",
3159 				i, bpf_subprog_name(env, env->subprog_topo_order[i]));
3160 out:
3161 	kvfree(dfs_stack);
3162 	kvfree(color);
3163 	return ret;
3164 }
3165 
mark_stack_slots_scratched(struct bpf_verifier_env * env,int spi,int nr_slots)3166 static void mark_stack_slots_scratched(struct bpf_verifier_env *env,
3167 				       int spi, int nr_slots)
3168 {
3169 	int i;
3170 
3171 	for (i = 0; i < nr_slots; i++)
3172 		mark_stack_slot_scratched(env, spi - i);
3173 }
3174 
__check_reg_arg(struct bpf_verifier_env * env,struct bpf_reg_state * regs,u32 regno,enum bpf_reg_arg_type t)3175 static int __check_reg_arg(struct bpf_verifier_env *env, struct bpf_reg_state *regs, u32 regno,
3176 			   enum bpf_reg_arg_type t)
3177 {
3178 	struct bpf_reg_state *reg;
3179 
3180 	mark_reg_scratched(env, regno);
3181 
3182 	reg = &regs[regno];
3183 	if (t == SRC_OP) {
3184 		/* check whether register used as source operand can be read */
3185 		if (reg->type == NOT_INIT) {
3186 			verbose(env, "R%d !read_ok\n", regno);
3187 			bpf_diag_unreadable_reg(env, env->insn_idx, regno);
3188 			return -EACCES;
3189 		}
3190 		/* We don't need to worry about FP liveness because it's read-only */
3191 		if (regno == BPF_REG_FP)
3192 			return 0;
3193 
3194 		return 0;
3195 	} else {
3196 		/* check whether register used as dest operand can be written to */
3197 		if (regno == BPF_REG_FP) {
3198 			verbose(env, "frame pointer is read only\n");
3199 			return -EACCES;
3200 		}
3201 		if (t == DST_OP)
3202 			mark_reg_unknown(env, regs, regno);
3203 	}
3204 	return 0;
3205 }
3206 
check_reg_arg(struct bpf_verifier_env * env,u32 regno,enum bpf_reg_arg_type t)3207 static int check_reg_arg(struct bpf_verifier_env *env, u32 regno,
3208 			 enum bpf_reg_arg_type t)
3209 {
3210 	struct bpf_verifier_state *vstate = env->cur_state;
3211 	struct bpf_func_state *state = vstate->frame[vstate->curframe];
3212 
3213 	return __check_reg_arg(env, state->regs, regno, t);
3214 }
3215 
mark_indirect_target(struct bpf_verifier_env * env,int idx)3216 static void mark_indirect_target(struct bpf_verifier_env *env, int idx)
3217 {
3218 	env->insn_aux_data[idx].indirect_target = true;
3219 }
3220 
3221 #define LR_FRAMENO_BITS	4
3222 #define LR_SPI_BITS	6
3223 #define LR_ENTRY_BITS	(LR_SPI_BITS + LR_FRAMENO_BITS + 1)
3224 #define LR_SIZE_BITS	4
3225 #define LR_FRAMENO_MASK	((1ull << LR_FRAMENO_BITS) - 1)
3226 #define LR_SPI_MASK	((1ull << LR_SPI_BITS)     - 1)
3227 #define LR_SIZE_MASK	((1ull << LR_SIZE_BITS)    - 1)
3228 #define LR_SPI_OFF	LR_FRAMENO_BITS
3229 #define LR_IS_REG_OFF	(LR_SPI_BITS + LR_FRAMENO_BITS)
3230 #define LINKED_REGS_MAX	5
3231 
3232 static_assert(MAX_CALL_FRAMES <= (1 << LR_FRAMENO_BITS));
3233 static_assert(LINKED_REGS_MAX < (1 << LR_SIZE_BITS));
3234 static_assert(LINKED_REGS_MAX * LR_ENTRY_BITS + LR_SIZE_BITS <= 64);
3235 
3236 struct linked_reg {
3237 	u8 frameno;
3238 	union {
3239 		u8 spi;
3240 		u8 regno;
3241 	};
3242 	bool is_reg;
3243 };
3244 
3245 struct linked_regs {
3246 	int cnt;
3247 	struct linked_reg entries[LINKED_REGS_MAX];
3248 };
3249 
linked_regs_push(struct linked_regs * s)3250 static struct linked_reg *linked_regs_push(struct linked_regs *s)
3251 {
3252 	if (s->cnt < LINKED_REGS_MAX)
3253 		return &s->entries[s->cnt++];
3254 
3255 	return NULL;
3256 }
3257 
3258 /*
3259  * Use u64 as a vector of 5 11-bit values, use first 4-bits to track
3260  * number of elements currently in stack.
3261  * Pack one history entry for linked registers as 11 bits in the following format:
3262  * - 4-bits frameno
3263  * - 6-bits spi_or_reg
3264  * - 1-bit  is_reg
3265  */
linked_regs_pack(struct linked_regs * s)3266 static u64 linked_regs_pack(struct linked_regs *s)
3267 {
3268 	u64 val = 0;
3269 	int i;
3270 
3271 	for (i = 0; i < s->cnt; ++i) {
3272 		struct linked_reg *e = &s->entries[i];
3273 		u64 tmp = 0;
3274 
3275 		tmp |= e->frameno;
3276 		tmp |= e->spi << LR_SPI_OFF;
3277 		tmp |= (e->is_reg ? 1 : 0) << LR_IS_REG_OFF;
3278 
3279 		val <<= LR_ENTRY_BITS;
3280 		val |= tmp;
3281 	}
3282 	val <<= LR_SIZE_BITS;
3283 	val |= s->cnt;
3284 	return val;
3285 }
3286 
linked_regs_unpack(u64 val,struct linked_regs * s)3287 static void linked_regs_unpack(u64 val, struct linked_regs *s)
3288 {
3289 	int i;
3290 
3291 	s->cnt = val & LR_SIZE_MASK;
3292 	val >>= LR_SIZE_BITS;
3293 
3294 	for (i = 0; i < s->cnt; ++i) {
3295 		struct linked_reg *e = &s->entries[i];
3296 
3297 		e->frameno =  val & LR_FRAMENO_MASK;
3298 		e->spi     = (val >> LR_SPI_OFF) & LR_SPI_MASK;
3299 		e->is_reg  = (val >> LR_IS_REG_OFF) & 0x1;
3300 		val >>= LR_ENTRY_BITS;
3301 	}
3302 }
3303 
bpf_disasm_kfunc_name(void * data,const struct bpf_insn * insn)3304 const char *bpf_disasm_kfunc_name(void *data, const struct bpf_insn *insn)
3305 {
3306 	const struct btf_type *func;
3307 	struct btf *desc_btf;
3308 
3309 	if (insn->src_reg != BPF_PSEUDO_KFUNC_CALL)
3310 		return NULL;
3311 
3312 	desc_btf = find_kfunc_desc_btf_cached(data, insn->off);
3313 	if (IS_ERR(desc_btf))
3314 		return "<error>";
3315 
3316 	func = btf_type_by_id(desc_btf, insn->imm);
3317 	if (!func || !btf_type_is_func(func))
3318 		return "<error>";
3319 	return btf_name_by_offset(desc_btf, func->name_off);
3320 }
3321 
bpf_verbose_insn(struct bpf_verifier_env * env,struct bpf_insn * insn)3322 void bpf_verbose_insn(struct bpf_verifier_env *env, struct bpf_insn *insn)
3323 {
3324 	const struct bpf_insn_cbs cbs = {
3325 		.cb_call	= bpf_disasm_kfunc_name,
3326 		.cb_print	= verbose,
3327 		.private_data	= env,
3328 	};
3329 
3330 	print_bpf_insn(&cbs, insn, env->allow_ptr_leaks);
3331 }
3332 
3333 /* If any register R in hist->linked_regs is marked as precise in bt,
3334  * do bt_set_frame_{reg,slot}(bt, R) for all registers in hist->linked_regs.
3335  */
bpf_bt_sync_linked_regs(struct backtrack_state * bt,struct bpf_jmp_history_entry * hist)3336 void bpf_bt_sync_linked_regs(struct backtrack_state *bt, struct bpf_jmp_history_entry *hist)
3337 {
3338 	struct linked_regs linked_regs;
3339 	bool some_precise = false;
3340 	int i;
3341 
3342 	if (!hist || hist->linked_regs == 0)
3343 		return;
3344 
3345 	linked_regs_unpack(hist->linked_regs, &linked_regs);
3346 	for (i = 0; i < linked_regs.cnt; ++i) {
3347 		struct linked_reg *e = &linked_regs.entries[i];
3348 
3349 		if ((e->is_reg && bt_is_frame_reg_set(bt, e->frameno, e->regno)) ||
3350 		    (!e->is_reg && bt_is_frame_slot_set(bt, e->frameno, e->spi))) {
3351 			some_precise = true;
3352 			break;
3353 		}
3354 	}
3355 
3356 	if (!some_precise)
3357 		return;
3358 
3359 	for (i = 0; i < linked_regs.cnt; ++i) {
3360 		struct linked_reg *e = &linked_regs.entries[i];
3361 
3362 		if (e->is_reg)
3363 			bpf_bt_set_frame_reg(bt, e->frameno, e->regno);
3364 		else
3365 			bpf_bt_set_frame_slot(bt, e->frameno, e->spi);
3366 	}
3367 }
3368 
mark_chain_precision(struct bpf_verifier_env * env,int regno)3369 int mark_chain_precision(struct bpf_verifier_env *env, int regno)
3370 {
3371 	return bpf_mark_chain_precision(env, env->cur_state, regno, NULL);
3372 }
3373 
3374 /* mark_chain_precision_batch() assumes that env->bt is set in the caller to
3375  * desired reg and stack masks across all relevant frames
3376  */
mark_chain_precision_batch(struct bpf_verifier_env * env,struct bpf_verifier_state * starting_state)3377 static int mark_chain_precision_batch(struct bpf_verifier_env *env,
3378 				      struct bpf_verifier_state *starting_state)
3379 {
3380 	return bpf_mark_chain_precision(env, starting_state, -1, NULL);
3381 }
3382 
3383 /* check if register is a constant scalar value */
is_reg_const(struct bpf_reg_state * reg,bool subreg32)3384 static bool is_reg_const(struct bpf_reg_state *reg, bool subreg32)
3385 {
3386 	return reg->type == SCALAR_VALUE &&
3387 	       tnum_is_const(subreg32 ? tnum_subreg(reg->var_off) : reg->var_off);
3388 }
3389 
3390 /* assuming is_reg_const() is true, return constant value of a register */
reg_const_value(struct bpf_reg_state * reg,bool subreg32)3391 static u64 reg_const_value(struct bpf_reg_state *reg, bool subreg32)
3392 {
3393 	return subreg32 ? tnum_subreg(reg->var_off).value : reg->var_off.value;
3394 }
3395 
is_pointer_regtype(enum bpf_reg_type type)3396 static bool is_pointer_regtype(enum bpf_reg_type type)
3397 {
3398 	return type != SCALAR_VALUE && type != NOT_INIT;
3399 }
3400 
__is_pointer_value(bool allow_ptr_leaks,const struct bpf_reg_state * reg)3401 static bool __is_pointer_value(bool allow_ptr_leaks,
3402 			       const struct bpf_reg_state *reg)
3403 {
3404 	if (allow_ptr_leaks)
3405 		return false;
3406 
3407 	return is_pointer_regtype(reg->type);
3408 }
3409 
clear_scalar_id(struct bpf_reg_state * reg)3410 static void clear_scalar_id(struct bpf_reg_state *reg)
3411 {
3412 	reg->id = 0;
3413 	reg->delta = 0;
3414 }
3415 
assign_scalar_id_before_mov(struct bpf_verifier_env * env,struct bpf_reg_state * src_reg)3416 static void assign_scalar_id_before_mov(struct bpf_verifier_env *env,
3417 					struct bpf_reg_state *src_reg)
3418 {
3419 	if (src_reg->type != SCALAR_VALUE)
3420 		return;
3421 	/*
3422 	 * The verifier is processing rX = rY insn and
3423 	 * rY->id has special linked register already.
3424 	 * Cleared it, since multiple rX += const are not supported.
3425 	 */
3426 	if (src_reg->id & BPF_ADD_CONST)
3427 		clear_scalar_id(src_reg);
3428 	/*
3429 	 * Ensure that src_reg has a valid ID that will be copied to
3430 	 * dst_reg and then will be used by sync_linked_regs() to
3431 	 * propagate min/max range.
3432 	 */
3433 	if (!src_reg->id && !tnum_is_const(src_reg->var_off))
3434 		src_reg->id = ++env->id_gen;
3435 }
3436 
save_register_state(struct bpf_verifier_env * env,struct bpf_func_state * state,int spi,struct bpf_reg_state * reg,int size)3437 static void save_register_state(struct bpf_verifier_env *env,
3438 				struct bpf_func_state *state,
3439 				int spi, struct bpf_reg_state *reg,
3440 				int size)
3441 {
3442 	int i;
3443 
3444 	bpf_diag_mod_begin(env, &state->stack[spi].spilled_ptr, reg, BPF_DIAG_MOD_SPILL);
3445 	state->stack[spi].spilled_ptr = *reg;
3446 
3447 	for (i = BPF_REG_SIZE; i > BPF_REG_SIZE - size; i--)
3448 		state->stack[spi].slot_type[i - 1] = STACK_SPILL;
3449 
3450 	/* size < 8 bytes spill */
3451 	for (; i; i--)
3452 		mark_stack_slot_misc(env, &state->stack[spi].slot_type[i - 1]);
3453 
3454 	bpf_diag_mod_end(env);
3455 }
3456 
is_bpf_st_mem(struct bpf_insn * insn)3457 static bool is_bpf_st_mem(struct bpf_insn *insn)
3458 {
3459 	return BPF_CLASS(insn->code) == BPF_ST && BPF_MODE(insn->code) == BPF_MEM;
3460 }
3461 
get_reg_width(struct bpf_reg_state * reg)3462 static int get_reg_width(struct bpf_reg_state *reg)
3463 {
3464 	return fls64(reg_umax(reg));
3465 }
3466 
3467 /* See comment for mark_fastcall_pattern_for_call() */
check_fastcall_stack_contract(struct bpf_verifier_env * env,struct bpf_func_state * state,int insn_idx,int off)3468 static void check_fastcall_stack_contract(struct bpf_verifier_env *env,
3469 					  struct bpf_func_state *state, int insn_idx, int off)
3470 {
3471 	struct bpf_subprog_info *subprog = &env->subprog_info[state->subprogno];
3472 	struct bpf_insn_aux_data *aux = env->insn_aux_data;
3473 	int i;
3474 
3475 	if (subprog->fastcall_stack_off <= off || aux[insn_idx].fastcall_pattern)
3476 		return;
3477 	/* access to the region [max_stack_depth .. fastcall_stack_off)
3478 	 * from something that is not a part of the fastcall pattern,
3479 	 * disable fastcall rewrites for current subprogram by setting
3480 	 * fastcall_stack_off to a value smaller than any possible offset.
3481 	 */
3482 	subprog->fastcall_stack_off = S16_MIN;
3483 	/* reset fastcall aux flags within subprogram,
3484 	 * happens at most once per subprogram
3485 	 */
3486 	for (i = subprog->start; i < (subprog + 1)->start; ++i) {
3487 		aux[i].fastcall_spills_num = 0;
3488 		aux[i].fastcall_pattern = 0;
3489 	}
3490 }
3491 
scrub_special_slot(struct bpf_func_state * state,int spi)3492 static void scrub_special_slot(struct bpf_func_state *state, int spi)
3493 {
3494 	int i;
3495 
3496 	/* regular write of data into stack destroys any spilled ptr */
3497 	state->stack[spi].spilled_ptr.type = NOT_INIT;
3498 	/* Mark slots as STACK_MISC if they belonged to spilled ptr/dynptr/iter. */
3499 	if (is_stack_slot_special(&state->stack[spi]))
3500 		for (i = 0; i < BPF_REG_SIZE; i++)
3501 			scrub_spilled_slot(&state->stack[spi].slot_type[i]);
3502 }
3503 
3504 /* check_stack_{read,write}_fixed_off functions track spill/fill of registers,
3505  * stack boundary and alignment are checked in check_mem_access()
3506  */
check_stack_write_fixed_off(struct bpf_verifier_env * env,struct bpf_func_state * state,int off,int size,int value_regno,int insn_idx)3507 static int check_stack_write_fixed_off(struct bpf_verifier_env *env,
3508 				       /* stack frame we're writing to */
3509 				       struct bpf_func_state *state,
3510 				       int off, int size, int value_regno,
3511 				       int insn_idx)
3512 {
3513 	struct bpf_func_state *cur; /* state of the current function */
3514 	int i, slot = -off - 1, spi = slot / BPF_REG_SIZE, err;
3515 	struct bpf_insn *insn = &env->prog->insnsi[insn_idx];
3516 	struct bpf_reg_state *reg = NULL;
3517 	int insn_flags = INSN_F_STACK_ACCESS;
3518 	int hist_spi = spi, hist_frame = state->frameno;
3519 
3520 	/* caller checked that off % size == 0 and -MAX_BPF_STACK <= off < 0,
3521 	 * so it's aligned access and [off, off + size) are within stack limits
3522 	 */
3523 	if (!env->allow_ptr_leaks &&
3524 	    bpf_is_spilled_reg(&state->stack[spi]) &&
3525 	    !bpf_is_spilled_scalar_reg(&state->stack[spi]) &&
3526 	    size != BPF_REG_SIZE) {
3527 		const char *reason;
3528 
3529 		verbose(env, "attempt to corrupt spilled pointer on stack\n");
3530 		reason = bpf_diag_fmt(env,
3531 				      "This store writes %d bytes at stack offset %d into a stack slot that currently holds a spilled pointer. "
3532 			"Partial writes to spilled pointers are rejected because they can corrupt pointer metadata and leak kernel pointers.",
3533 			size, off);
3534 		bpf_diag_memory(
3535 			env, insn_idx, "stack spill corruption", reason,
3536 			"Write the full 8-byte spilled pointer slot, or use a separate stack slot for scalar data before overwriting only part of it.");
3537 		return -EACCES;
3538 	}
3539 
3540 	cur = env->cur_state->frame[env->cur_state->curframe];
3541 	if (value_regno >= 0)
3542 		reg = &cur->regs[value_regno];
3543 	if (!env->bypass_spec_v4) {
3544 		bool sanitize = reg && is_pointer_regtype(reg->type);
3545 
3546 		for (i = 0; i < size; i++) {
3547 			u8 type = state->stack[spi].slot_type[(slot - i) %
3548 							      BPF_REG_SIZE];
3549 
3550 			if (type != STACK_MISC && type != STACK_ZERO) {
3551 				sanitize = true;
3552 				break;
3553 			}
3554 		}
3555 
3556 		if (sanitize)
3557 			env->insn_aux_data[insn_idx].nospec_result = true;
3558 	}
3559 
3560 	err = destroy_if_dynptr_stack_slot(env, state, spi);
3561 	if (err)
3562 		return err;
3563 
3564 	check_fastcall_stack_contract(env, state, insn_idx, off);
3565 	mark_stack_slot_scratched(env, spi);
3566 	if (reg && !(off % BPF_REG_SIZE) && reg->type == SCALAR_VALUE && env->bpf_capable) {
3567 		bool reg_value_fits;
3568 
3569 		reg_value_fits = get_reg_width(reg) <= BITS_PER_BYTE * size;
3570 		/* Make sure that reg had an ID to build a relation on spill. */
3571 		if (reg_value_fits)
3572 			assign_scalar_id_before_mov(env, reg);
3573 		save_register_state(env, state, spi, reg, size);
3574 		/* Break the relation on a narrowing spill. */
3575 		if (!reg_value_fits)
3576 			state->stack[spi].spilled_ptr.id = 0;
3577 	} else if (!reg && !(off % BPF_REG_SIZE) && is_bpf_st_mem(insn) &&
3578 		   env->bpf_capable) {
3579 		struct bpf_reg_state *tmp_reg = &env->fake_reg[0];
3580 
3581 		memset(tmp_reg, 0, sizeof(*tmp_reg));
3582 		__mark_reg_known(tmp_reg, insn->imm);
3583 		tmp_reg->type = SCALAR_VALUE;
3584 		save_register_state(env, state, spi, tmp_reg, size);
3585 	} else if (reg && is_pointer_regtype(reg->type)) {
3586 		/* register containing pointer is being spilled into stack */
3587 		if (size != BPF_REG_SIZE) {
3588 			verbose_linfo(env, insn_idx, "; ");
3589 			verbose(env, "invalid size of register spill\n");
3590 			return -EACCES;
3591 		}
3592 		if (state != cur && reg->type == PTR_TO_STACK) {
3593 			verbose(env, "cannot spill pointers to stack into stack frame of the caller\n");
3594 			return -EINVAL;
3595 		}
3596 		save_register_state(env, state, spi, reg, size);
3597 	} else {
3598 		u8 type = STACK_MISC;
3599 
3600 		if (bpf_is_spilled_reg(&state->stack[spi]))
3601 			bpf_diag_record_scrub(env, &state->stack[spi].spilled_ptr,
3602 					      BPF_DIAG_MOD_WRITE);
3603 		scrub_special_slot(state, spi);
3604 
3605 		/* when we zero initialize stack slots mark them as such */
3606 		if ((reg && bpf_register_is_null(reg)) ||
3607 		    (!reg && is_bpf_st_mem(insn) && insn->imm == 0)) {
3608 			/* STACK_ZERO case happened because register spill
3609 			 * wasn't properly aligned at the stack slot boundary,
3610 			 * so it's not a register spill anymore; force
3611 			 * originating register to be precise to make
3612 			 * STACK_ZERO correct for subsequent states
3613 			 */
3614 			err = mark_chain_precision(env, value_regno);
3615 			if (err)
3616 				return err;
3617 			type = STACK_ZERO;
3618 		}
3619 
3620 		/* Mark slots affected by this stack write. */
3621 		for (i = 0; i < size; i++)
3622 			state->stack[spi].slot_type[(slot - i) % BPF_REG_SIZE] = type;
3623 		insn_flags = 0; /* not a register spill */
3624 	}
3625 
3626 	if (insn_flags)
3627 		return bpf_push_jmp_history(env, env->cur_state, insn_flags,
3628 					    hist_spi, hist_frame, 0);
3629 	return 0;
3630 }
3631 
3632 /* Write the stack: 'stack[ptr_reg + off] = value_regno'. 'ptr_reg' is
3633  * known to contain a variable offset.
3634  * This function checks whether the write is permitted and conservatively
3635  * tracks the effects of the write, considering that each stack slot in the
3636  * dynamic range is potentially written to.
3637  *
3638  * 'value_regno' can be -1, meaning that an unknown value is being written to
3639  * the stack.
3640  *
3641  * Spilled pointers in range are not marked as written because we don't know
3642  * what's going to be actually written. This means that read propagation for
3643  * future reads cannot be terminated by this write.
3644  *
3645  * For privileged programs, uninitialized stack slots are considered
3646  * initialized by this write (even though we don't know exactly what offsets
3647  * are going to be written to). The idea is that we don't want the verifier to
3648  * reject future reads that access slots written to through variable offsets.
3649  */
check_stack_write_var_off(struct bpf_verifier_env * env,struct bpf_func_state * state,struct bpf_reg_state * ptr_reg,int off,int size,int value_regno,int insn_idx)3650 static int check_stack_write_var_off(struct bpf_verifier_env *env,
3651 				     /* func where register points to */
3652 				     struct bpf_func_state *state,
3653 				     struct bpf_reg_state *ptr_reg, int off, int size,
3654 				     int value_regno, int insn_idx)
3655 {
3656 	struct bpf_func_state *cur; /* state of the current function */
3657 	int min_off, max_off;
3658 	int i, err;
3659 	struct bpf_reg_state *value_reg = NULL;
3660 	struct bpf_insn *insn = &env->prog->insnsi[insn_idx];
3661 	bool writing_zero = false;
3662 	/* set if the fact that we're writing a zero is used to let any
3663 	 * stack slots remain STACK_ZERO
3664 	 */
3665 	bool zero_used = false;
3666 
3667 	cur = env->cur_state->frame[env->cur_state->curframe];
3668 	min_off = reg_smin(ptr_reg) + off;
3669 	max_off = reg_smax(ptr_reg) + off + size;
3670 	if (value_regno >= 0)
3671 		value_reg = &cur->regs[value_regno];
3672 	if ((value_reg && bpf_register_is_null(value_reg)) ||
3673 	    (!value_reg && is_bpf_st_mem(insn) && insn->imm == 0))
3674 		writing_zero = true;
3675 
3676 	for (i = min_off; i < max_off; i++) {
3677 		int spi;
3678 
3679 		spi = bpf_get_spi(i);
3680 		err = destroy_if_dynptr_stack_slot(env, state, spi);
3681 		if (err)
3682 			return err;
3683 	}
3684 
3685 	check_fastcall_stack_contract(env, state, insn_idx, min_off);
3686 	/* Variable offset writes destroy any spilled pointers in range. */
3687 	for (i = min_off; i < max_off; i++) {
3688 		u8 new_type, *stype;
3689 		int slot, spi;
3690 
3691 		slot = -i - 1;
3692 		spi = slot / BPF_REG_SIZE;
3693 		stype = &state->stack[spi].slot_type[slot % BPF_REG_SIZE];
3694 		mark_stack_slot_scratched(env, spi);
3695 
3696 		if (!env->allow_ptr_leaks && *stype != STACK_MISC && *stype != STACK_ZERO) {
3697 			/* Reject the write if range we may write to has not
3698 			 * been initialized beforehand. If we didn't reject
3699 			 * here, the ptr status would be erased below (even
3700 			 * though not all slots are actually overwritten),
3701 			 * possibly opening the door to leaks.
3702 			 *
3703 			 * We do however catch STACK_INVALID case below, and
3704 			 * only allow reading possibly uninitialized memory
3705 			 * later for CAP_PERFMON, as the write may not happen to
3706 			 * that slot.
3707 			 */
3708 			verbose(env, "spilled ptr in range of var-offset stack write; insn %d, ptr off: %d",
3709 				insn_idx, i);
3710 			return -EINVAL;
3711 		}
3712 
3713 		/* If writing_zero and the spi slot contains a spill of value 0,
3714 		 * maintain the spill type.
3715 		 */
3716 		if (writing_zero && *stype == STACK_SPILL &&
3717 		    bpf_is_spilled_scalar_reg(&state->stack[spi])) {
3718 			struct bpf_reg_state *spill_reg = &state->stack[spi].spilled_ptr;
3719 
3720 			if (tnum_is_const(spill_reg->var_off) && spill_reg->var_off.value == 0) {
3721 				zero_used = true;
3722 				continue;
3723 			}
3724 		}
3725 
3726 		/*
3727 		 * Scrub slots if variable-offset stack write goes over spilled pointers.
3728 		 * Otherwise bpf_is_spilled_reg() may == true && spilled_ptr.type == NOT_INIT
3729 		 * and valid program is rejected by check_stack_read_fixed_off()
3730 		 * with obscure "invalid size of register fill" message.
3731 		 */
3732 		scrub_special_slot(state, spi);
3733 
3734 		/* Update the slot type. */
3735 		new_type = STACK_MISC;
3736 		if (writing_zero && *stype == STACK_ZERO) {
3737 			new_type = STACK_ZERO;
3738 			zero_used = true;
3739 		}
3740 		/* If the slot is STACK_INVALID, we check whether it's OK to
3741 		 * pretend that it will be initialized by this write. The slot
3742 		 * might not actually be written to, and so if we mark it as
3743 		 * initialized future reads might leak uninitialized memory.
3744 		 * For privileged programs, we will accept such reads to slots
3745 		 * that may or may not be written because, if we're reject
3746 		 * them, the error would be too confusing.
3747 		 * Conservatively, treat STACK_POISON in a similar way.
3748 		 */
3749 		if ((*stype == STACK_INVALID || *stype == STACK_POISON) &&
3750 		    !env->allow_uninit_stack) {
3751 			verbose(env, "uninit stack in range of var-offset write prohibited for !root; insn %d, off: %d",
3752 					insn_idx, i);
3753 			return -EINVAL;
3754 		}
3755 		*stype = new_type;
3756 	}
3757 	if (zero_used) {
3758 		/* backtracking doesn't work for STACK_ZERO yet. */
3759 		err = mark_chain_precision(env, value_regno);
3760 		if (err)
3761 			return err;
3762 	}
3763 	bpf_diag_record_scrub_stack(env, state, min_off, max_off,
3764 				    BPF_DIAG_MOD_VAR_WRITE);
3765 	return 0;
3766 }
3767 
3768 /* When register 'dst_regno' is assigned some values from stack[min_off,
3769  * max_off), we set the register's type according to the types of the
3770  * respective stack slots. If all the stack values are known to be zeros, then
3771  * so is the destination reg. Otherwise, the register is considered to be
3772  * SCALAR. This function does not deal with register filling; the caller must
3773  * ensure that all spilled registers in the stack range have been marked as
3774  * read.
3775  *
3776  * STACK_SPILL bytes backed by spilled scalar const zeroes are also considered
3777  * zero bytes. In that case, mark the contributing stack slots precise so
3778  * pruning cannot reuse a zero-spill state for a later non-zero spill state.
3779  *
3780  * Returns an error if precision backtracking fails.
3781  */
mark_reg_stack_read(struct bpf_verifier_env * env,struct bpf_func_state * ptr_state,int min_off,int max_off,int dst_regno)3782 static int mark_reg_stack_read(struct bpf_verifier_env *env,
3783 			       /* func where src register points to */
3784 			       struct bpf_func_state *ptr_state,
3785 			       int min_off, int max_off, int dst_regno)
3786 {
3787 	struct bpf_verifier_state *vstate = env->cur_state;
3788 	struct bpf_func_state *state = vstate->frame[vstate->curframe];
3789 	u64 zero_spill_mask = 0;
3790 	int i, slot, spi;
3791 	u8 *stype;
3792 	int zeros = 0;
3793 
3794 	for (i = min_off; i < max_off; i++) {
3795 		slot = -i - 1;
3796 		spi = slot / BPF_REG_SIZE;
3797 		mark_stack_slot_scratched(env, spi);
3798 		stype = ptr_state->stack[spi].slot_type;
3799 		if (stype[slot % BPF_REG_SIZE] == STACK_ZERO) {
3800 			zeros++;
3801 			continue;
3802 		}
3803 		if (stype[slot % BPF_REG_SIZE] == STACK_SPILL &&
3804 		    bpf_register_is_null(&ptr_state->stack[spi].spilled_ptr)) {
3805 			zero_spill_mask |= 1ull << spi;
3806 			zeros++;
3807 			continue;
3808 		}
3809 		break;
3810 	}
3811 	if (zeros == max_off - min_off) {
3812 		/* Any access_size read into register is zero extended,
3813 		 * so the whole register == const_zero.
3814 		 */
3815 		__mark_reg_const_zero(env, &state->regs[dst_regno]);
3816 		if (zero_spill_mask) {
3817 			bpf_bt_set_frame_slot_mask(&env->bt, ptr_state->frameno, zero_spill_mask);
3818 			return mark_chain_precision_batch(env, env->cur_state);
3819 		}
3820 	} else {
3821 		/* have read misc data from the stack */
3822 		mark_reg_unknown(env, state->regs, dst_regno);
3823 	}
3824 
3825 	return 0;
3826 }
3827 
bpf_diag_stack_read_uninit(struct bpf_verifier_env * env,int off,int i,int size)3828 static void bpf_diag_stack_read_uninit(struct bpf_verifier_env *env, int off, int i,
3829 				       int size)
3830 {
3831 	const char *reason;
3832 
3833 	reason = bpf_diag_fmt(env,
3834 			      "This rejected read uses %d bytes at stack offset %d, but byte %d in that range is uninitialized on this path. "
3835 		"Programs loaded with CAP_PERFMON can be allowed to read uninitialized stack bytes, but this program is being rejected without that allowance.",
3836 		size, off, i);
3837 	bpf_diag_memory(
3838 		env, env->insn_idx, "uninitialized stack read", reason,
3839 		"Initialize every byte in the stack range before reading it, adjust the offset and size so the read covers only initialized bytes, "
3840 		"or load with CAP_PERFMON if uninitialized stack reads are intended.");
3841 }
3842 
3843 /* Read the stack at 'off' and put the results into the register indicated by
3844  * 'dst_regno'. It handles reg filling if the addressed stack slot is a
3845  * spilled reg.
3846  *
3847  * 'dst_regno' can be -1, meaning that the read value is not going to a
3848  * register.
3849  *
3850  * The access is assumed to be within the current stack bounds.
3851  */
check_stack_read_fixed_off(struct bpf_verifier_env * env,struct bpf_func_state * reg_state,int off,int size,int dst_regno)3852 static int check_stack_read_fixed_off(struct bpf_verifier_env *env,
3853 				      /* func where src register points to */
3854 				      struct bpf_func_state *reg_state,
3855 				      int off, int size, int dst_regno)
3856 {
3857 	struct bpf_verifier_state *vstate = env->cur_state;
3858 	struct bpf_func_state *state = vstate->frame[vstate->curframe];
3859 	int i, slot = -off - 1, spi = slot / BPF_REG_SIZE;
3860 	struct bpf_reg_state *reg;
3861 	u8 *stype, type;
3862 	int err;
3863 	int insn_flags = INSN_F_STACK_ACCESS;
3864 	int hist_spi = spi, hist_frame = reg_state->frameno;
3865 
3866 	stype = reg_state->stack[spi].slot_type;
3867 	reg = &reg_state->stack[spi].spilled_ptr;
3868 
3869 	mark_stack_slot_scratched(env, spi);
3870 	check_fastcall_stack_contract(env, state, env->insn_idx, off);
3871 
3872 	/*
3873 	 * Refine the in-progress load record's origin to the source stack slot.
3874 	 */
3875 	if (dst_regno >= 0)
3876 		bpf_diag_mod_begin(env, &state->regs[dst_regno], reg, BPF_DIAG_MOD_WRITE);
3877 
3878 	if (bpf_is_spilled_reg(&reg_state->stack[spi])) {
3879 		u8 spill_size = 1;
3880 
3881 		for (i = BPF_REG_SIZE - 1; i > 0 && stype[i - 1] == STACK_SPILL; i--)
3882 			spill_size++;
3883 
3884 		if (size != BPF_REG_SIZE || spill_size != BPF_REG_SIZE) {
3885 			if (reg->type != SCALAR_VALUE) {
3886 				verbose_linfo(env, env->insn_idx, "; ");
3887 				verbose(env, "invalid size of register fill\n");
3888 				return -EACCES;
3889 			}
3890 
3891 			if (dst_regno < 0)
3892 				return 0;
3893 
3894 			if (size <= spill_size &&
3895 			    bpf_stack_narrow_access_ok(off, size, spill_size)) {
3896 				if (env->bpf_capable && size == 4 && spill_size == 4 &&
3897 				    get_reg_width(reg) <= 32)
3898 					/* Ensure stack slot has an ID to build a relation
3899 					 * with the destination register on fill.
3900 					 */
3901 					assign_scalar_id_before_mov(env, reg);
3902 				state->regs[dst_regno] = *reg;
3903 
3904 				/* Break the relation on a narrowing fill.
3905 				 * coerce_reg_to_size will adjust the boundaries.
3906 				 */
3907 				if (get_reg_width(reg) > size * BITS_PER_BYTE)
3908 					clear_scalar_id(&state->regs[dst_regno]);
3909 			} else {
3910 				int spill_cnt = 0, zero_cnt = 0;
3911 
3912 				for (i = 0; i < size; i++) {
3913 					type = stype[(slot - i) % BPF_REG_SIZE];
3914 					if (type == STACK_SPILL) {
3915 						spill_cnt++;
3916 						continue;
3917 					}
3918 					if (type == STACK_MISC)
3919 						continue;
3920 					if (type == STACK_ZERO) {
3921 						zero_cnt++;
3922 						continue;
3923 					}
3924 					if (type == STACK_INVALID && env->allow_uninit_stack)
3925 						continue;
3926 					if (type == STACK_POISON) {
3927 						verbose(env, "reading from stack off %d+%d size %d, slot poisoned by dead code elimination\n",
3928 							off, i, size);
3929 					} else {
3930 						verbose(env, "invalid read from stack off %d+%d size %d\n",
3931 							off, i, size);
3932 						bpf_diag_stack_read_uninit(env, off, i, size);
3933 					}
3934 					return -EACCES;
3935 				}
3936 
3937 				if (spill_cnt == size &&
3938 				    tnum_is_const(reg->var_off) && reg->var_off.value == 0) {
3939 					__mark_reg_const_zero(env, &state->regs[dst_regno]);
3940 					/* this IS register fill, so keep insn_flags */
3941 				} else if (zero_cnt == size) {
3942 					/* similarly to mark_reg_stack_read(), preserve zeroes */
3943 					__mark_reg_const_zero(env, &state->regs[dst_regno]);
3944 					insn_flags = 0; /* not restoring original register state */
3945 				} else {
3946 					err = mark_reg_stack_read(env, reg_state, off, off + size,
3947 								  dst_regno);
3948 					if (err)
3949 						return err;
3950 					insn_flags = 0; /* not restoring original register state */
3951 				}
3952 			}
3953 		} else if (dst_regno >= 0) {
3954 			/* restore register state from stack */
3955 			if (env->bpf_capable)
3956 				/* Ensure stack slot has an ID to build a relation
3957 				 * with the destination register on fill.
3958 				 */
3959 				assign_scalar_id_before_mov(env, reg);
3960 			state->regs[dst_regno] = *reg;
3961 			/* mark reg as written since spilled pointer state likely
3962 			 * has its liveness marks cleared by is_state_visited()
3963 			 * which resets stack/reg liveness for state transitions
3964 			 */
3965 		} else if (__is_pointer_value(env->allow_ptr_leaks, reg)) {
3966 			/* If dst_regno==-1, the caller is asking us whether
3967 			 * it is acceptable to use this value as a SCALAR_VALUE
3968 			 * (e.g. for XADD).
3969 			 * We must not allow unprivileged callers to do that
3970 			 * with spilled pointers.
3971 			 */
3972 			verbose(env, "leaking pointer from stack off %d\n",
3973 				off);
3974 			return -EACCES;
3975 		}
3976 	} else {
3977 		for (i = 0; i < size; i++) {
3978 			type = stype[(slot - i) % BPF_REG_SIZE];
3979 			if (type == STACK_MISC)
3980 				continue;
3981 			if (type == STACK_ZERO)
3982 				continue;
3983 			if (type == STACK_INVALID && env->allow_uninit_stack)
3984 				continue;
3985 			if (type == STACK_POISON) {
3986 				verbose(env, "reading from stack off %d+%d size %d, slot poisoned by dead code elimination\n",
3987 					off, i, size);
3988 			} else {
3989 				verbose(env, "invalid read from stack off %d+%d size %d\n",
3990 					off, i, size);
3991 				bpf_diag_stack_read_uninit(env, off, i, size);
3992 			}
3993 			return -EACCES;
3994 		}
3995 		if (dst_regno >= 0) {
3996 			err = mark_reg_stack_read(env, reg_state, off, off + size, dst_regno);
3997 			if (err)
3998 				return err;
3999 		}
4000 		insn_flags = 0; /* we are not restoring spilled register */
4001 	}
4002 	if (insn_flags)
4003 		return bpf_push_jmp_history(env, env->cur_state, insn_flags,
4004 					    hist_spi, hist_frame, 0);
4005 	return 0;
4006 }
4007 
4008 enum bpf_access_src {
4009 	ACCESS_DIRECT = 1,  /* the access is performed by an instruction */
4010 	ACCESS_HELPER = 2,  /* the access is performed by a helper */
4011 };
4012 
4013 static int check_stack_range_initialized(struct bpf_verifier_env *env, struct bpf_reg_state *reg,
4014 					 argno_t argno, int off, int access_size,
4015 					 bool zero_size_allowed,
4016 					 enum bpf_access_type type,
4017 					 struct bpf_call_arg_meta *meta);
4018 
reg_state(struct bpf_verifier_env * env,int regno)4019 static struct bpf_reg_state *reg_state(struct bpf_verifier_env *env, int regno)
4020 {
4021 	return cur_regs(env) + regno;
4022 }
4023 
4024 /* Read the stack at 'reg + off' and put the result into the register
4025  * 'dst_regno'.
4026  * 'off' includes the pointer register's fixed offset(i.e. 'reg->off'),
4027  * but not its variable offset.
4028  * 'size' is assumed to be <= reg size and the access is assumed to be aligned.
4029  *
4030  * As opposed to check_stack_read_fixed_off, this function doesn't deal with
4031  * filling registers (i.e. reads of spilled register cannot be detected when
4032  * the offset is not fixed). We conservatively mark 'dst_regno' as containing
4033  * SCALAR_VALUE. That's why we assert that the 'reg' has a variable
4034  * offset; for a fixed offset check_stack_read_fixed_off should be used
4035  * instead.
4036  */
check_stack_read_var_off(struct bpf_verifier_env * env,struct bpf_reg_state * reg,argno_t ptr_argno,int off,int size,int dst_regno)4037 static int check_stack_read_var_off(struct bpf_verifier_env *env, struct bpf_reg_state *reg,
4038 				    argno_t ptr_argno, int off, int size, int dst_regno)
4039 {
4040 	struct bpf_func_state *ptr_state = bpf_func(env, reg);
4041 	int err;
4042 	int min_off, max_off;
4043 
4044 	/* Note that we pass a NULL meta, so raw access will not be permitted.
4045 	 */
4046 	err = check_stack_range_initialized(env, reg, ptr_argno, off, size,
4047 					    false, BPF_READ, NULL);
4048 	if (err)
4049 		return err;
4050 
4051 	min_off = reg_smin(reg) + off;
4052 	max_off = reg_smax(reg) + off;
4053 	err = mark_reg_stack_read(env, ptr_state, min_off, max_off + size,
4054 				  dst_regno);
4055 	if (err)
4056 		return err;
4057 	check_fastcall_stack_contract(env, ptr_state, env->insn_idx, min_off);
4058 	return 0;
4059 }
4060 
4061 /* check_stack_read dispatches to check_stack_read_fixed_off or
4062  * check_stack_read_var_off.
4063  *
4064  * The caller must ensure that the offset falls within the allocated stack
4065  * bounds.
4066  *
4067  * 'dst_regno' is a register which will receive the value from the stack. It
4068  * can be -1, meaning that the read value is not going to a register.
4069  */
check_stack_read(struct bpf_verifier_env * env,struct bpf_reg_state * reg,argno_t ptr_argno,int off,int size,int dst_regno)4070 static int check_stack_read(struct bpf_verifier_env *env,
4071 			    struct bpf_reg_state *reg, argno_t ptr_argno, int off, int size,
4072 			    int dst_regno)
4073 {
4074 	struct bpf_func_state *state = bpf_func(env, reg);
4075 	int err;
4076 	/* Some accesses are only permitted with a static offset. */
4077 	bool var_off = !tnum_is_const(reg->var_off);
4078 
4079 	/* The offset is required to be static when reads don't go to a
4080 	 * register, in order to not leak pointers (see
4081 	 * check_stack_read_fixed_off).
4082 	 */
4083 	if (dst_regno < 0 && var_off) {
4084 		const char *reason;
4085 		char tn_buf[48];
4086 
4087 		tnum_strn(tn_buf, sizeof(tn_buf), reg->var_off);
4088 		verbose(env, "variable offset stack pointer cannot be passed into helper function; var_off=%s off=%d size=%d\n",
4089 			tn_buf, off, size);
4090 		reason = bpf_diag_fmt(env,
4091 				      "The helper would access the stack through variable offset %s plus fixed offset %d and size %d. "
4092 			"Helper stack memory arguments require a constant stack offset and a precise initialized range.",
4093 			tn_buf, off, size);
4094 		bpf_diag_memory(
4095 			env, env->insn_idx, "variable stack access", reason,
4096 			"Use a fixed stack offset for helper memory arguments, or copy the needed bytes into a fixed stack slot first.");
4097 		return -EACCES;
4098 	}
4099 	/* Variable offset is prohibited for unprivileged mode for simplicity
4100 	 * since it requires corresponding support in Spectre masking for stack
4101 	 * ALU. See also retrieve_ptr_limit(). The check in
4102 	 * check_stack_access_for_ptr_arithmetic() called by
4103 	 * adjust_ptr_min_max_vals() prevents users from creating stack pointers
4104 	 * with variable offsets, therefore no check is required here. Further,
4105 	 * just checking it here would be insufficient as speculative stack
4106 	 * writes could still lead to unsafe speculative behaviour.
4107 	 */
4108 	if (!var_off) {
4109 		off += reg->var_off.value;
4110 		err = check_stack_read_fixed_off(env, state, off, size,
4111 						 dst_regno);
4112 	} else {
4113 		/* Variable offset stack reads need more conservative handling
4114 		 * than fixed offset ones. Note that dst_regno >= 0 on this
4115 		 * branch.
4116 		 */
4117 		err = check_stack_read_var_off(env, reg, ptr_argno, off, size,
4118 					       dst_regno);
4119 	}
4120 	return err;
4121 }
4122 
4123 /* check_stack_write dispatches to check_stack_write_fixed_off or
4124  * check_stack_write_var_off.
4125  *
4126  * 'reg' is the register used as a pointer into the stack.
4127  * 'value_regno' is the register whose value we're writing to the stack. It can
4128  * be -1, meaning that we're not writing from a register.
4129  *
4130  * The caller must ensure that the offset falls within the maximum stack size.
4131  */
check_stack_write(struct bpf_verifier_env * env,struct bpf_reg_state * reg,int off,int size,int value_regno,int insn_idx)4132 static int check_stack_write(struct bpf_verifier_env *env,
4133 			     struct bpf_reg_state *reg, int off, int size,
4134 			     int value_regno, int insn_idx)
4135 {
4136 	struct bpf_func_state *state = bpf_func(env, reg);
4137 	int err;
4138 
4139 	if (tnum_is_const(reg->var_off)) {
4140 		off += reg->var_off.value;
4141 		err = check_stack_write_fixed_off(env, state, off, size,
4142 						  value_regno, insn_idx);
4143 	} else {
4144 		/* Variable offset stack reads need more conservative handling
4145 		 * than fixed offset ones.
4146 		 */
4147 		err = check_stack_write_var_off(env, state,
4148 						reg, off, size,
4149 						value_regno, insn_idx);
4150 	}
4151 	return err;
4152 }
4153 
4154 /*
4155  * Write a value to the outgoing stack arg area.
4156  * off is a negative offset from r11 (e.g. -8 for arg6, -16 for arg7).
4157  */
check_stack_arg_write(struct bpf_verifier_env * env,struct bpf_func_state * state,int off,struct bpf_reg_state * value_reg)4158 static int check_stack_arg_write(struct bpf_verifier_env *env, struct bpf_func_state *state,
4159 				 int off, struct bpf_reg_state *value_reg)
4160 {
4161 	int max_stack_arg_regs = MAX_BPF_FUNC_ARGS - MAX_BPF_FUNC_REG_ARGS;
4162 	struct bpf_subprog_info *subprog = &env->subprog_info[state->subprogno];
4163 	int spi = -off / BPF_REG_SIZE - 1;
4164 	struct bpf_reg_state *arg;
4165 	int err;
4166 
4167 	if (spi >= max_stack_arg_regs) {
4168 		verbose(env, "stack arg write offset %d exceeds max %d stack args\n",
4169 			off, max_stack_arg_regs);
4170 		return -EINVAL;
4171 	}
4172 
4173 	err = grow_stack_arg_slots(env, state, spi + 1);
4174 	if (err)
4175 		return err;
4176 
4177 	/* Track the max outgoing stack arg slot count. */
4178 	if (spi + 1 > subprog->max_out_stack_arg_cnt)
4179 		subprog->max_out_stack_arg_cnt = spi + 1;
4180 
4181 	arg = &state->stack_arg_regs[spi];
4182 	bpf_diag_mod_begin(env, arg, value_reg, BPF_DIAG_MOD_WRITE);
4183 
4184 	if (value_reg) {
4185 		state->stack_arg_regs[spi] = *value_reg;
4186 	} else {
4187 		/* BPF_ST: store immediate, treat as scalar */
4188 		arg->type = SCALAR_VALUE;
4189 		__mark_reg_known(arg, env->prog->insnsi[env->insn_idx].imm);
4190 	}
4191 	bpf_diag_mod_end(env);
4192 	state->no_stack_arg_load = true;
4193 	return bpf_push_jmp_history(env, env->cur_state,
4194 				    INSN_F_STACK_ARG_ACCESS, spi, 0, 0);
4195 }
4196 
4197 /*
4198  * Read a value from the incoming stack arg area.
4199  * off is a positive offset from r11 (e.g. +8 for arg6, +16 for arg7).
4200  */
check_stack_arg_read(struct bpf_verifier_env * env,struct bpf_func_state * state,int off,int dst_regno)4201 static int check_stack_arg_read(struct bpf_verifier_env *env, struct bpf_func_state *state,
4202 				int off, int dst_regno)
4203 {
4204 	struct bpf_subprog_info *subprog = &env->subprog_info[state->subprogno];
4205 	struct bpf_verifier_state *vstate = env->cur_state;
4206 	int spi = off / BPF_REG_SIZE - 1;
4207 	struct bpf_func_state *caller, *cur;
4208 	struct bpf_reg_state *arg;
4209 
4210 	if (state->no_stack_arg_load) {
4211 		verbose(env, "r11 load must be before any r11 store or call insn\n");
4212 		return -EINVAL;
4213 	}
4214 
4215 	if (spi + 1 > bpf_in_stack_arg_cnt(subprog)) {
4216 		verbose(env, "invalid read from stack arg off %d depth %d\n",
4217 			off, bpf_in_stack_arg_cnt(subprog) * BPF_REG_SIZE);
4218 		return -EACCES;
4219 	}
4220 
4221 	caller = vstate->frame[vstate->curframe - 1];
4222 	arg = &caller->stack_arg_regs[spi];
4223 	cur = vstate->frame[vstate->curframe];
4224 	bpf_diag_mod_begin(env, &cur->regs[dst_regno], arg, BPF_DIAG_MOD_WRITE);
4225 	cur->regs[dst_regno] = *arg;
4226 	bpf_diag_mod_end(env);
4227 	return bpf_push_jmp_history(env, env->cur_state,
4228 				    INSN_F_STACK_ARG_ACCESS, spi, 0, 0);
4229 }
4230 
mark_stack_arg_precision(struct bpf_verifier_env * env,int arg_idx)4231 static int mark_stack_arg_precision(struct bpf_verifier_env *env, int arg_idx)
4232 {
4233 	struct bpf_func_state *caller = cur_func(env);
4234 	int spi = arg_idx - MAX_BPF_FUNC_REG_ARGS;
4235 
4236 	bt_set_frame_stack_arg_slot(&env->bt, caller->frameno, spi);
4237 	return mark_chain_precision_batch(env, env->cur_state);
4238 }
4239 
check_outgoing_stack_args(struct bpf_verifier_env * env,struct bpf_func_state * caller,int nargs,const char * callee_name,const struct btf * btf,const struct btf_param * args)4240 static int check_outgoing_stack_args(struct bpf_verifier_env *env, struct bpf_func_state *caller,
4241 				     int nargs, const char *callee_name, const struct btf *btf,
4242 				     const struct btf_param *args)
4243 {
4244 	int i, spi;
4245 
4246 	for (i = MAX_BPF_FUNC_REG_ARGS; i < nargs; i++) {
4247 		spi = i - MAX_BPF_FUNC_REG_ARGS;
4248 		if (spi >= caller->out_stack_arg_cnt ||
4249 		    caller->stack_arg_regs[spi].type == NOT_INIT) {
4250 			const char *arg_name = NULL;
4251 
4252 			if (args && args[i].name_off)
4253 				arg_name = btf_name_by_offset(btf, args[i].name_off);
4254 			verbose(env, "callee expects %d args, stack arg%d is not initialized\n",
4255 				nargs, spi + 1);
4256 			bpf_diag_stack_arg_uninit(env, env->insn_idx, nargs, spi,
4257 						  callee_name, arg_name);
4258 			return -EFAULT;
4259 		}
4260 	}
4261 
4262 	return 0;
4263 }
4264 
get_func_arg_reg(struct bpf_func_state * caller,struct bpf_reg_state * regs,int arg)4265 static struct bpf_reg_state *get_func_arg_reg(struct bpf_func_state *caller,
4266 					      struct bpf_reg_state *regs, int arg)
4267 {
4268 	if (arg < MAX_BPF_FUNC_REG_ARGS)
4269 		return &regs[arg + 1];
4270 
4271 	return &caller->stack_arg_regs[arg - MAX_BPF_FUNC_REG_ARGS];
4272 }
4273 
check_map_access_type(struct bpf_verifier_env * env,struct bpf_reg_state * reg,int off,int size,enum bpf_access_type type)4274 static int check_map_access_type(struct bpf_verifier_env *env, struct bpf_reg_state *reg,
4275 				 int off, int size, enum bpf_access_type type)
4276 {
4277 	struct bpf_map *map = reg->map_ptr;
4278 	u32 cap = bpf_map_flags_to_cap(map);
4279 
4280 	if (type == BPF_WRITE && !(cap & BPF_MAP_CAN_WRITE)) {
4281 		verbose(env, "write into map forbidden, value_size=%d off=%lld size=%d\n",
4282 			map->value_size, reg_smin(reg) + off, size);
4283 		return -EACCES;
4284 	}
4285 
4286 	if (type == BPF_READ && !(cap & BPF_MAP_CAN_READ)) {
4287 		verbose(env, "read from map forbidden, value_size=%d off=%lld size=%d\n",
4288 			map->value_size, reg_smin(reg) + off, size);
4289 		return -EACCES;
4290 	}
4291 
4292 	return 0;
4293 }
4294 
4295 /* check read/write into memory region (e.g., map value, ringbuf sample, etc) */
__check_mem_access(struct bpf_verifier_env * env,struct bpf_reg_state * reg,argno_t argno,int off,int size,u32 mem_size,bool zero_size_allowed)4296 static int __check_mem_access(struct bpf_verifier_env *env, struct bpf_reg_state *reg, argno_t argno,
4297 			      int off, int size, u32 mem_size,
4298 			      bool zero_size_allowed)
4299 {
4300 	bool size_ok = size > 0 || (size == 0 && zero_size_allowed);
4301 
4302 	if (off >= 0 && size_ok && (u64)off + size <= mem_size)
4303 		return 0;
4304 
4305 	switch (reg->type) {
4306 	case PTR_TO_MAP_KEY:
4307 		verbose(env, "invalid access to map key, key_size=%d off=%d size=%d\n",
4308 			mem_size, off, size);
4309 		break;
4310 	case PTR_TO_MAP_VALUE:
4311 		verbose(env, "invalid access to map value, value_size=%d off=%d size=%d\n",
4312 			mem_size, off, size);
4313 		break;
4314 	case PTR_TO_PACKET:
4315 	case PTR_TO_PACKET_META:
4316 	case PTR_TO_PACKET_END:
4317 		verbose(env, "invalid access to packet, off=%d size=%d, %s(id=%d,off=%d,r=%d)\n",
4318 			off, size, reg_arg_name(env, argno), reg->id, off, mem_size);
4319 		break;
4320 	case PTR_TO_CTX:
4321 		verbose(env, "invalid access to context, ctx_size=%d off=%d size=%d\n",
4322 			mem_size, off, size);
4323 		break;
4324 	case PTR_TO_MEM:
4325 	default:
4326 		verbose(env, "invalid access to memory, mem_size=%u off=%d size=%d\n",
4327 			mem_size, off, size);
4328 	}
4329 
4330 	return -EACCES;
4331 }
4332 
4333 /* check read/write into a memory region with possible variable offset */
check_mem_region_access(struct bpf_verifier_env * env,struct bpf_reg_state * reg,argno_t argno,int off,int size,u32 mem_size,bool zero_size_allowed)4334 static int check_mem_region_access(struct bpf_verifier_env *env, struct bpf_reg_state *reg, argno_t argno,
4335 				   int off, int size, u32 mem_size,
4336 				   bool zero_size_allowed)
4337 {
4338 	const char *proof = "";
4339 	const char *start;
4340 	s64 max_start, max_end;
4341 	int err;
4342 
4343 	/* We may have adjusted the register pointing to memory region, so we
4344 	 * need to try adding each of min_value and max_value to off
4345 	 * to make sure our theoretical access will be safe.
4346 	 *
4347 	 * The minimum value is only important with signed
4348 	 * comparisons where we can't assume the floor of a
4349 	 * value is 0.  If we are using signed variables for our
4350 	 * index'es we need to make sure that whatever we use
4351 	 * will have a set floor within our range.
4352 	 */
4353 	if (reg_smin(reg) < 0 &&
4354 	    (reg_smin(reg) == S64_MIN ||
4355 	     (off + reg_smin(reg) != (s64)(s32)(off + reg_smin(reg))) ||
4356 	      reg_smin(reg) + off < 0)) {
4357 		verbose(env, "%s min value is negative, either use unsigned index or do a if (index >=0) check.\n",
4358 			reg_arg_name(env, argno));
4359 		err = -EACCES;
4360 		if (bpf_diag_enabled(env)) {
4361 			start = bpf_diag_fmt_s64_sum(env, reg_smin(reg), off);
4362 			proof = bpf_diag_fmt(
4363 				env, "the minimal bound for a memory access is a negative value: %s",
4364 				start);
4365 		}
4366 		goto report_error;
4367 	}
4368 
4369 	err = __check_mem_access(env, reg, argno, reg_smin(reg) + off, size,
4370 				 mem_size, zero_size_allowed);
4371 	if (err) {
4372 		verbose(env, "%s min value is outside of the allowed memory range\n",
4373 			reg_arg_name(env, argno));
4374 		if (bpf_diag_enabled(env)) {
4375 			start = bpf_diag_fmt_s64_sum(env, reg_smin(reg), off);
4376 			proof = bpf_diag_fmt(
4377 				env, "the minimal bound for a memory access is %s and is outside of the object of size %u",
4378 				start, mem_size);
4379 		}
4380 		goto report_error;
4381 	}
4382 
4383 	/* If we haven't set a max value then we need to bail since we can't be
4384 	 * sure we won't do bad things.
4385 	 * If reg_umax(reg) + off could overflow, treat that as unbounded too.
4386 	 */
4387 	if (reg_umax(reg) >= BPF_MAX_VAR_OFF) {
4388 		verbose(env, "%s unbounded memory access, make sure to bounds check any such access\n",
4389 			reg_arg_name(env, argno));
4390 		err = -EACCES;
4391 		if (bpf_diag_enabled(env))
4392 			proof = bpf_diag_fmt(
4393 				env, "the maximal bound for a memory access is %llu and exceeds maximum allowed offset of %u",
4394 				reg_umax(reg), BPF_MAX_VAR_OFF);
4395 		goto report_error;
4396 	}
4397 
4398 	err = __check_mem_access(env, reg, argno, reg_umax(reg) + off, size,
4399 				 mem_size, zero_size_allowed);
4400 	if (err) {
4401 		verbose(env, "%s max value is outside of the allowed memory range\n",
4402 			reg_arg_name(env, argno));
4403 		if (bpf_diag_enabled(env)) {
4404 			max_start = (s64)reg_umax(reg) + off;
4405 			max_end = max_start + size;
4406 			proof = bpf_diag_fmt(
4407 				env, "the maximal bound for a memory access is %lld: start %lld + access_size %d, beyond object_size %u",
4408 				max_end, max_start, size, mem_size);
4409 		}
4410 		goto report_error;
4411 	}
4412 
4413 	return 0;
4414 
4415 report_error:
4416 	bpf_diag_mem_bounds(env, env->insn_idx, reg_from_argno(argno),
4417 			    reg_arg_name(env, argno), reg_type_str(env, reg->type), proof,
4418 			    off, size, mem_size, reg);
4419 	return err;
4420 }
4421 
__check_ptr_off_reg(struct bpf_verifier_env * env,const struct bpf_reg_state * reg,argno_t argno,bool fixed_off_ok)4422 static int __check_ptr_off_reg(struct bpf_verifier_env *env,
4423 			       const struct bpf_reg_state *reg, argno_t argno,
4424 			       bool fixed_off_ok)
4425 {
4426 	/* Access to this pointer-typed register or passing it to a helper
4427 	 * is only allowed in its original, unmodified form.
4428 	 */
4429 
4430 	if (!tnum_is_const(reg->var_off)) {
4431 		char tn_buf[48];
4432 
4433 		tnum_strn(tn_buf, sizeof(tn_buf), reg->var_off);
4434 		verbose(env, "variable %s access var_off=%s disallowed\n",
4435 			reg_type_str(env, reg->type), tn_buf);
4436 		return -EACCES;
4437 	}
4438 
4439 	if (reg_smin(reg) < 0) {
4440 		verbose(env, "negative offset %s ptr %s off=%lld disallowed\n",
4441 			reg_type_str(env, reg->type), reg_arg_name(env, argno), reg->var_off.value);
4442 		return -EACCES;
4443 	}
4444 
4445 	if (!fixed_off_ok && reg->var_off.value != 0) {
4446 		verbose(env, "dereference of modified %s ptr %s off=%lld disallowed\n",
4447 			reg_type_str(env, reg->type), reg_arg_name(env, argno), reg->var_off.value);
4448 		bpf_diag_invalid_deref(env, env->insn_idx, reg_from_argno(argno),
4449 				       reg_arg_name(env, argno), reg,
4450 					      BPF_DIAG_DEREF_MODIFIED_PTR, reg->var_off.value);
4451 		return -EACCES;
4452 	}
4453 
4454 	return 0;
4455 }
4456 
check_ptr_off_reg(struct bpf_verifier_env * env,const struct bpf_reg_state * reg,int regno)4457 static int check_ptr_off_reg(struct bpf_verifier_env *env,
4458 		             const struct bpf_reg_state *reg, int regno)
4459 {
4460 	return __check_ptr_off_reg(env, reg, argno_from_reg(regno), false);
4461 }
4462 
map_kptr_match_type(struct bpf_verifier_env * env,struct btf_field * kptr_field,struct bpf_reg_state * reg,u32 regno)4463 static int map_kptr_match_type(struct bpf_verifier_env *env,
4464 			       struct btf_field *kptr_field,
4465 			       struct bpf_reg_state *reg, u32 regno)
4466 {
4467 	const char *targ_name = btf_type_name(kptr_field->kptr.btf, kptr_field->kptr.btf_id);
4468 	int perm_flags;
4469 	const char *reg_name = "";
4470 
4471 	if (base_type(reg->type) != PTR_TO_BTF_ID)
4472 		goto bad_type;
4473 
4474 	if (btf_is_kernel(reg->btf)) {
4475 		perm_flags = PTR_MAYBE_NULL | PTR_TRUSTED | MEM_RCU;
4476 
4477 		/* Only unreferenced case accepts untrusted pointers */
4478 		if (kptr_field->type == BPF_KPTR_UNREF)
4479 			perm_flags |= PTR_UNTRUSTED;
4480 	} else {
4481 		perm_flags = PTR_MAYBE_NULL | MEM_ALLOC;
4482 		if (kptr_field->type == BPF_KPTR_PERCPU)
4483 			perm_flags |= MEM_PERCPU;
4484 	}
4485 
4486 	if (type_flag(reg->type) & ~perm_flags)
4487 		goto bad_type;
4488 
4489 	/* We need to verify reg->type and reg->btf, before accessing reg->btf */
4490 	reg_name = btf_type_name(reg->btf, reg->btf_id);
4491 
4492 	/* For ref_ptr case, release function check should ensure we get one
4493 	 * referenced PTR_TO_BTF_ID, and that its fixed offset is 0. For the
4494 	 * normal store of unreferenced kptr, we must ensure var_off is zero.
4495 	 * Since ref_ptr cannot be accessed directly by BPF insns, check for
4496 	 * reg->id is not needed here.
4497 	 */
4498 	if (__check_ptr_off_reg(env, reg, argno_from_reg(regno), true))
4499 		return -EACCES;
4500 
4501 	/* A full type match is needed, as BTF can be vmlinux, module or prog BTF, and
4502 	 * we also need to take into account the reg->var_off.
4503 	 *
4504 	 * We want to support cases like:
4505 	 *
4506 	 * struct foo {
4507 	 *         struct bar br;
4508 	 *         struct baz bz;
4509 	 * };
4510 	 *
4511 	 * struct foo *v;
4512 	 * v = func();	      // PTR_TO_BTF_ID
4513 	 * val->foo = v;      // reg->var_off is zero, btf and btf_id match type
4514 	 * val->bar = &v->br; // reg->var_off is still zero, but we need to retry with
4515 	 *                    // first member type of struct after comparison fails
4516 	 * val->baz = &v->bz; // reg->var_off is non-zero, so struct needs to be walked
4517 	 *                    // to match type
4518 	 *
4519 	 * In the kptr_ref case, check_func_arg_reg_off already ensures reg->var_off
4520 	 * is zero. We must also ensure that btf_struct_ids_match does not walk
4521 	 * the struct to match type against first member of struct, i.e. reject
4522 	 * second case from above. Hence, when type is BPF_KPTR_REF, we set
4523 	 * strict mode to true for type match.
4524 	 */
4525 	if (!btf_struct_ids_match(&env->log, reg->btf, reg->btf_id, reg->var_off.value,
4526 				  kptr_field->kptr.btf, kptr_field->kptr.btf_id,
4527 				  kptr_field->type != BPF_KPTR_UNREF,
4528 				  !type_is_alloc(reg->type)))
4529 		goto bad_type;
4530 	return 0;
4531 bad_type:
4532 	verbose(env, "invalid kptr access, R%d type=%s%s ", regno,
4533 		reg_type_str(env, reg->type), reg_name);
4534 	verbose(env, "expected=%s%s", reg_type_str(env, PTR_TO_BTF_ID), targ_name);
4535 	if (kptr_field->type == BPF_KPTR_UNREF)
4536 		verbose(env, " or %s%s\n", reg_type_str(env, PTR_TO_BTF_ID | PTR_UNTRUSTED),
4537 			targ_name);
4538 	else
4539 		verbose(env, "\n");
4540 	return -EINVAL;
4541 }
4542 
in_sleepable(struct bpf_verifier_env * env)4543 static bool in_sleepable(struct bpf_verifier_env *env)
4544 {
4545 	return env->cur_state->in_sleepable;
4546 }
4547 
4548 /* The non-sleepable programs and sleepable programs with explicit bpf_rcu_read_lock()
4549  * can dereference RCU protected pointers and result is PTR_TRUSTED.
4550  */
in_rcu_cs(struct bpf_verifier_env * env)4551 static bool in_rcu_cs(struct bpf_verifier_env *env)
4552 {
4553 	return env->cur_state->active_rcu_locks ||
4554 	       env->cur_state->active_preempt_locks ||
4555 	       env->cur_state->active_locks ||
4556 	       env->cur_state->active_irq_id ||
4557 	       !in_sleepable(env);
4558 }
4559 
4560 /* Once GCC supports btf_type_tag the following mechanism will be replaced with tag check */
4561 BTF_SET_START(rcu_protected_types)
4562 #ifdef CONFIG_NET
BTF_ID(struct,prog_test_ref_kfunc)4563 BTF_ID(struct, prog_test_ref_kfunc)
4564 #endif
4565 #ifdef CONFIG_CGROUPS
4566 BTF_ID(struct, cgroup)
4567 #endif
4568 #ifdef CONFIG_BPF_JIT
4569 BTF_ID(struct, bpf_cpumask)
4570 #endif
4571 BTF_ID(struct, task_struct)
4572 #ifdef CONFIG_CRYPTO
4573 BTF_ID(struct, bpf_crypto_ctx)
4574 #endif
4575 #ifdef CONFIG_INET
4576 BTF_ID(struct, bpf_ksock)
4577 #endif
4578 BTF_SET_END(rcu_protected_types)
4579 
4580 static bool rcu_protected_object(const struct btf *btf, u32 btf_id)
4581 {
4582 	if (!btf_is_kernel(btf))
4583 		return true;
4584 	return btf_id_set_contains(&rcu_protected_types, btf_id);
4585 }
4586 
kptr_pointee_btf_record(struct btf_field * kptr_field)4587 static struct btf_record *kptr_pointee_btf_record(struct btf_field *kptr_field)
4588 {
4589 	struct btf_struct_meta *meta;
4590 
4591 	if (btf_is_kernel(kptr_field->kptr.btf))
4592 		return NULL;
4593 
4594 	meta = btf_find_struct_meta(kptr_field->kptr.btf,
4595 				    kptr_field->kptr.btf_id);
4596 
4597 	return meta ? meta->record : NULL;
4598 }
4599 
rcu_safe_kptr(const struct btf_field * field)4600 static bool rcu_safe_kptr(const struct btf_field *field)
4601 {
4602 	const struct btf_field_kptr *kptr = &field->kptr;
4603 
4604 	return field->type == BPF_KPTR_PERCPU ||
4605 	       (field->type == BPF_KPTR_REF && rcu_protected_object(kptr->btf, kptr->btf_id));
4606 }
4607 
btf_ld_kptr_type(struct bpf_verifier_env * env,struct btf_field * kptr_field)4608 static u32 btf_ld_kptr_type(struct bpf_verifier_env *env, struct btf_field *kptr_field)
4609 {
4610 	struct btf_record *rec;
4611 	u32 ret;
4612 
4613 	ret = PTR_MAYBE_NULL;
4614 	if (rcu_safe_kptr(kptr_field) && in_rcu_cs(env)) {
4615 		ret |= MEM_RCU;
4616 		if (kptr_field->type == BPF_KPTR_PERCPU)
4617 			ret |= MEM_PERCPU;
4618 		else if (!btf_is_kernel(kptr_field->kptr.btf))
4619 			ret |= MEM_ALLOC;
4620 
4621 		rec = kptr_pointee_btf_record(kptr_field);
4622 		if (rec && btf_record_has_field(rec, BPF_GRAPH_NODE))
4623 			ret |= NON_OWN_REF;
4624 	} else {
4625 		ret |= PTR_UNTRUSTED;
4626 	}
4627 
4628 	return ret;
4629 }
4630 
mark_uptr_ld_reg(struct bpf_verifier_env * env,u32 regno,struct btf_field * field)4631 static int mark_uptr_ld_reg(struct bpf_verifier_env *env, u32 regno,
4632 			    struct btf_field *field)
4633 {
4634 	struct bpf_reg_state *reg;
4635 	const struct btf_type *t;
4636 
4637 	t = btf_type_by_id(field->kptr.btf, field->kptr.btf_id);
4638 	mark_reg_known_zero(env, cur_regs(env), regno);
4639 	reg = reg_state(env, regno);
4640 	reg->type = PTR_TO_MEM | PTR_MAYBE_NULL;
4641 	reg->mem_size = t->size;
4642 	reg->id = ++env->id_gen;
4643 
4644 	return 0;
4645 }
4646 
check_map_kptr_access(struct bpf_verifier_env * env,int value_regno,int insn_idx,struct btf_field * kptr_field)4647 static int check_map_kptr_access(struct bpf_verifier_env *env,
4648 				 int value_regno, int insn_idx,
4649 				 struct btf_field *kptr_field)
4650 {
4651 	struct bpf_insn *insn = &env->prog->insnsi[insn_idx];
4652 	int class = BPF_CLASS(insn->code);
4653 	struct bpf_reg_state *val_reg;
4654 	int ret;
4655 
4656 	/* Things we already checked for in check_map_access and caller:
4657 	 *  - Reject cases where variable offset may touch kptr
4658 	 *  - size of access (must be BPF_DW)
4659 	 *  - tnum_is_const(reg->var_off)
4660 	 *  - kptr_field->offset == off + reg->var_off.value
4661 	 */
4662 	/* Only BPF_[LDX,STX,ST] | BPF_MEM | BPF_DW is supported */
4663 	if (BPF_MODE(insn->code) != BPF_MEM) {
4664 		verbose(env, "kptr in map can only be accessed using BPF_MEM instruction mode\n");
4665 		return -EACCES;
4666 	}
4667 
4668 	/* We only allow loading referenced kptr, since it will be marked as
4669 	 * untrusted, similar to unreferenced kptr.
4670 	 */
4671 	if (class != BPF_LDX &&
4672 	    (kptr_field->type == BPF_KPTR_REF || kptr_field->type == BPF_KPTR_PERCPU)) {
4673 		verbose(env, "store to referenced kptr disallowed\n");
4674 		return -EACCES;
4675 	}
4676 	if (class != BPF_LDX && kptr_field->type == BPF_UPTR) {
4677 		verbose(env, "store to uptr disallowed\n");
4678 		return -EACCES;
4679 	}
4680 
4681 	if (class == BPF_LDX) {
4682 		if (kptr_field->type == BPF_UPTR)
4683 			return mark_uptr_ld_reg(env, value_regno, kptr_field);
4684 
4685 		/* We can simply mark the value_regno receiving the pointer
4686 		 * value from map as PTR_TO_BTF_ID, with the correct type.
4687 		 */
4688 		ret = mark_btf_ld_reg(env, cur_regs(env), value_regno, PTR_TO_BTF_ID,
4689 				      kptr_field->kptr.btf, kptr_field->kptr.btf_id,
4690 				      btf_ld_kptr_type(env, kptr_field));
4691 		if (ret < 0)
4692 			return ret;
4693 	} else if (class == BPF_STX) {
4694 		val_reg = reg_state(env, value_regno);
4695 		if (!bpf_register_is_null(val_reg) &&
4696 		    map_kptr_match_type(env, kptr_field, val_reg, value_regno))
4697 			return -EACCES;
4698 	} else if (class == BPF_ST) {
4699 		if (insn->imm) {
4700 			verbose(env, "BPF_ST imm must be 0 when storing to kptr at off=%u\n",
4701 				kptr_field->offset);
4702 			return -EACCES;
4703 		}
4704 	} else {
4705 		verbose(env, "kptr in map can only be accessed using BPF_LDX/BPF_STX/BPF_ST\n");
4706 		return -EACCES;
4707 	}
4708 	return 0;
4709 }
4710 
4711 /*
4712  * Return the size of the memory region accessible from a pointer to map value.
4713  * For INSN_ARRAY maps whole bpf_insn_array->ips array is accessible.
4714  */
map_mem_size(const struct bpf_map * map)4715 static u32 map_mem_size(const struct bpf_map *map)
4716 {
4717 	if (map->map_type == BPF_MAP_TYPE_INSN_ARRAY)
4718 		return map->max_entries * sizeof(long);
4719 
4720 	return map->value_size;
4721 }
4722 
4723 /* check read/write into a map element with possible variable offset */
check_map_access(struct bpf_verifier_env * env,struct bpf_reg_state * reg,argno_t argno,int off,int size,bool zero_size_allowed,enum bpf_access_src src)4724 static int check_map_access(struct bpf_verifier_env *env, struct bpf_reg_state *reg, argno_t argno,
4725 			    int off, int size, bool zero_size_allowed,
4726 			    enum bpf_access_src src)
4727 {
4728 	struct bpf_map *map = reg->map_ptr;
4729 	u32 mem_size = map_mem_size(map);
4730 	struct btf_record *rec;
4731 	int err, i;
4732 
4733 	err = check_mem_region_access(env, reg, argno, off, size, mem_size, zero_size_allowed);
4734 	if (err)
4735 		return err;
4736 
4737 	if (IS_ERR_OR_NULL(map->record))
4738 		return 0;
4739 	rec = map->record;
4740 	for (i = 0; i < rec->cnt; i++) {
4741 		struct btf_field *field = &rec->fields[i];
4742 		u32 p = field->offset;
4743 
4744 		/* If any part of a field  can be touched by load/store, reject
4745 		 * this program. To check that [x1, x2) overlaps with [y1, y2),
4746 		 * it is sufficient to check x1 < y2 && y1 < x2.
4747 		 */
4748 		if (reg_smin(reg) + off < p + field->size &&
4749 		    p < reg_umax(reg) + off + size) {
4750 			switch (field->type) {
4751 			case BPF_KPTR_UNREF:
4752 			case BPF_KPTR_REF:
4753 			case BPF_KPTR_PERCPU:
4754 			case BPF_UPTR:
4755 				if (src != ACCESS_DIRECT) {
4756 					verbose(env, "%s cannot be accessed indirectly by helper\n",
4757 						btf_field_type_name(field->type));
4758 					return -EACCES;
4759 				}
4760 				if (!tnum_is_const(reg->var_off)) {
4761 					verbose(env, "%s access cannot have variable offset\n",
4762 						btf_field_type_name(field->type));
4763 					return -EACCES;
4764 				}
4765 				if (p != off + reg->var_off.value) {
4766 					verbose(env, "%s access misaligned expected=%u off=%llu\n",
4767 						btf_field_type_name(field->type),
4768 						p, off + reg->var_off.value);
4769 					return -EACCES;
4770 				}
4771 				if (size != bpf_size_to_bytes(BPF_DW)) {
4772 					verbose(env, "%s access size must be BPF_DW\n",
4773 						btf_field_type_name(field->type));
4774 					return -EACCES;
4775 				}
4776 				break;
4777 			default:
4778 				verbose(env, "%s cannot be accessed directly by load/store\n",
4779 					btf_field_type_name(field->type));
4780 				return -EACCES;
4781 			}
4782 		}
4783 	}
4784 	return 0;
4785 }
4786 
may_access_direct_pkt_data(struct bpf_verifier_env * env,const struct bpf_func_proto * fn,enum bpf_access_type t)4787 static bool may_access_direct_pkt_data(struct bpf_verifier_env *env,
4788 			       const struct bpf_func_proto *fn,
4789 			       enum bpf_access_type t)
4790 {
4791 	enum bpf_prog_type prog_type = resolve_prog_type(env->prog);
4792 
4793 	switch (prog_type) {
4794 	/* Program types only with direct read access go here! */
4795 	case BPF_PROG_TYPE_LWT_IN:
4796 	case BPF_PROG_TYPE_LWT_OUT:
4797 	case BPF_PROG_TYPE_LWT_SEG6LOCAL:
4798 	case BPF_PROG_TYPE_SK_REUSEPORT:
4799 	case BPF_PROG_TYPE_FLOW_DISSECTOR:
4800 	case BPF_PROG_TYPE_CGROUP_SKB:
4801 		if (t == BPF_WRITE)
4802 			return false;
4803 		fallthrough;
4804 
4805 	/* Program types with direct read + write access go here! */
4806 	case BPF_PROG_TYPE_SCHED_CLS:
4807 	case BPF_PROG_TYPE_SCHED_ACT:
4808 	case BPF_PROG_TYPE_XDP:
4809 	case BPF_PROG_TYPE_LWT_XMIT:
4810 	case BPF_PROG_TYPE_SK_SKB:
4811 	case BPF_PROG_TYPE_SK_MSG:
4812 		if (fn)
4813 			return fn->pkt_access;
4814 
4815 		env->seen_direct_write = true;
4816 		return true;
4817 
4818 	case BPF_PROG_TYPE_CGROUP_SOCKOPT:
4819 		if (t == BPF_WRITE)
4820 			env->seen_direct_write = true;
4821 
4822 		return true;
4823 
4824 	default:
4825 		return false;
4826 	}
4827 }
4828 
check_packet_access(struct bpf_verifier_env * env,struct bpf_reg_state * reg,argno_t argno,int off,int size,bool zero_size_allowed)4829 static int check_packet_access(struct bpf_verifier_env *env, struct bpf_reg_state *reg, argno_t argno, int off,
4830 			       int size, bool zero_size_allowed)
4831 {
4832 	int err;
4833 
4834 	if (reg->range < 0) {
4835 		verbose(env, "%s offset is outside of the packet\n", reg_arg_name(env, argno));
4836 		return -EINVAL;
4837 	}
4838 
4839 	err = check_mem_region_access(env, reg, argno, off, size, reg->range, zero_size_allowed);
4840 	if (err)
4841 		return err;
4842 
4843 	/* __check_mem_access has made sure "off + size - 1" is within u16.
4844 	 * reg_umax(reg) can't be bigger than MAX_PACKET_OFF which is 0xffff,
4845 	 * otherwise find_good_pkt_pointers would have refused to set range info
4846 	 * that __check_mem_access would have rejected this pkt access.
4847 	 * Therefore, "off + reg_umax(reg) + size - 1" won't overflow u32.
4848 	 */
4849 	env->prog->aux->max_pkt_offset =
4850 		max_t(u32, env->prog->aux->max_pkt_offset,
4851 		      off + reg_umax(reg) + size - 1);
4852 
4853 	return 0;
4854 }
4855 
is_var_ctx_off_allowed(struct bpf_prog * prog)4856 static bool is_var_ctx_off_allowed(struct bpf_prog *prog)
4857 {
4858 	return resolve_prog_type(prog) == BPF_PROG_TYPE_SYSCALL;
4859 }
4860 
4861 /* check access to 'struct bpf_context' fields.  Supports fixed offsets only */
__check_ctx_access(struct bpf_verifier_env * env,int insn_idx,int off,int size,enum bpf_access_type t,struct bpf_insn_access_aux * info)4862 static int __check_ctx_access(struct bpf_verifier_env *env, int insn_idx, int off, int size,
4863 			      enum bpf_access_type t, struct bpf_insn_access_aux *info)
4864 {
4865 	if (env->ops->is_valid_access &&
4866 	    env->ops->is_valid_access(off, size, t, env->prog, info)) {
4867 		/* A non zero info.ctx_field_size indicates that this field is a
4868 		 * candidate for later verifier transformation to load the whole
4869 		 * field and then apply a mask when accessed with a narrower
4870 		 * access than actual ctx access size. A zero info.ctx_field_size
4871 		 * will only allow for whole field access and rejects any other
4872 		 * type of narrower access.
4873 		 */
4874 		if (base_type(info->reg_type) == PTR_TO_BTF_ID) {
4875 			if (info->ref_id &&
4876 			    !find_reference_state(env->cur_state, info->ref_id)) {
4877 				verbose(env, "invalid bpf_context access off=%d. Reference may already be released\n",
4878 					off);
4879 				return -EACCES;
4880 			}
4881 		} else {
4882 			env->insn_aux_data[insn_idx].ctx_field_size = info->ctx_field_size;
4883 		}
4884 		/* remember the offset of last byte accessed in ctx */
4885 		if (env->prog->aux->max_ctx_offset < off + size)
4886 			env->prog->aux->max_ctx_offset = off + size;
4887 		return 0;
4888 	}
4889 
4890 	verbose(env, "invalid bpf_context access off=%d size=%d\n", off, size);
4891 	return -EACCES;
4892 }
4893 
check_ctx_access(struct bpf_verifier_env * env,int insn_idx,struct bpf_reg_state * reg,argno_t argno,int off,int access_size,enum bpf_access_type t,struct bpf_insn_access_aux * info)4894 static int check_ctx_access(struct bpf_verifier_env *env, int insn_idx, struct bpf_reg_state *reg, argno_t argno,
4895 			    int off, int access_size, enum bpf_access_type t,
4896 			    struct bpf_insn_access_aux *info)
4897 {
4898 	/*
4899 	 * Program types that don't rewrite ctx accesses can safely
4900 	 * dereference ctx pointers with fixed offsets.
4901 	 */
4902 	bool var_off_ok = is_var_ctx_off_allowed(env->prog);
4903 	bool fixed_off_ok = !env->ops->convert_ctx_access;
4904 	int err;
4905 
4906 	if (var_off_ok)
4907 		err = check_mem_region_access(env, reg, argno, off, access_size, U16_MAX, false);
4908 	else
4909 		err = __check_ptr_off_reg(env, reg, argno, fixed_off_ok);
4910 	if (err)
4911 		return err;
4912 	off += reg_umax(reg);
4913 
4914 	err = __check_ctx_access(env, insn_idx, off, access_size, t, info);
4915 	if (err)
4916 		verbose_linfo(env, insn_idx, "; ");
4917 	return err;
4918 }
4919 
check_flow_keys_access(struct bpf_verifier_env * env,struct bpf_reg_state * reg,argno_t argno,int off,int size)4920 static int check_flow_keys_access(struct bpf_verifier_env *env,
4921 				  struct bpf_reg_state *reg, argno_t argno,
4922 				  int off, int size)
4923 {
4924 	/* Only a constant offset is allowed here; fold it into off. */
4925 	if (!tnum_is_const(reg->var_off)) {
4926 		char tn_buf[48];
4927 
4928 		tnum_strn(tn_buf, sizeof(tn_buf), reg->var_off);
4929 		verbose(env, "%s invalid variable offset to flow keys: off=%d, var_off=%s\n",
4930 			reg_arg_name(env, argno), off, tn_buf);
4931 		return -EACCES;
4932 	}
4933 	off += reg->var_off.value;
4934 
4935 	if (size < 0 || off < 0 ||
4936 	    (u64)off + size > sizeof(struct bpf_flow_keys)) {
4937 		verbose(env, "invalid access to flow keys off=%d size=%d\n",
4938 			off, size);
4939 		return -EACCES;
4940 	}
4941 	return 0;
4942 }
4943 
check_sock_access(struct bpf_verifier_env * env,int insn_idx,struct bpf_reg_state * reg,argno_t argno,int off,int size,enum bpf_access_type t)4944 static int check_sock_access(struct bpf_verifier_env *env, int insn_idx,
4945 			     struct bpf_reg_state *reg, argno_t argno, int off, int size,
4946 			     enum bpf_access_type t)
4947 {
4948 	struct bpf_insn_access_aux info = {};
4949 	bool valid;
4950 
4951 	if (reg_smin(reg) < 0) {
4952 		verbose(env, "%s min value is negative, either use unsigned index or do a if (index >=0) check.\n",
4953 			reg_arg_name(env, argno));
4954 		return -EACCES;
4955 	}
4956 
4957 	switch (reg->type) {
4958 	case PTR_TO_SOCK_COMMON:
4959 		valid = bpf_sock_common_is_valid_access(off, size, t, &info);
4960 		break;
4961 	case PTR_TO_SOCKET:
4962 		valid = bpf_sock_is_valid_access(off, size, t, &info);
4963 		break;
4964 	case PTR_TO_TCP_SOCK:
4965 		valid = bpf_tcp_sock_is_valid_access(off, size, t, &info);
4966 		break;
4967 	case PTR_TO_XDP_SOCK:
4968 		valid = bpf_xdp_sock_is_valid_access(off, size, t, &info);
4969 		break;
4970 	default:
4971 		valid = false;
4972 	}
4973 
4974 	if (valid) {
4975 		env->insn_aux_data[insn_idx].ctx_field_size =
4976 			info.ctx_field_size;
4977 		return 0;
4978 	}
4979 
4980 	verbose(env, "%s invalid %s access off=%d size=%d\n",
4981 		reg_arg_name(env, argno), reg_type_str(env, reg->type), off, size);
4982 
4983 	return -EACCES;
4984 }
4985 
is_pointer_value(struct bpf_verifier_env * env,int regno)4986 static bool is_pointer_value(struct bpf_verifier_env *env, int regno)
4987 {
4988 	return __is_pointer_value(env->allow_ptr_leaks, reg_state(env, regno));
4989 }
4990 
is_ctx_reg(struct bpf_verifier_env * env,int regno)4991 static bool is_ctx_reg(struct bpf_verifier_env *env, int regno)
4992 {
4993 	const struct bpf_reg_state *reg = reg_state(env, regno);
4994 
4995 	return reg->type == PTR_TO_CTX;
4996 }
4997 
is_sk_reg(struct bpf_verifier_env * env,int regno)4998 static bool is_sk_reg(struct bpf_verifier_env *env, int regno)
4999 {
5000 	const struct bpf_reg_state *reg = reg_state(env, regno);
5001 
5002 	return type_is_sk_pointer(reg->type);
5003 }
5004 
is_pkt_reg(struct bpf_verifier_env * env,int regno)5005 static bool is_pkt_reg(struct bpf_verifier_env *env, int regno)
5006 {
5007 	const struct bpf_reg_state *reg = reg_state(env, regno);
5008 
5009 	return type_is_pkt_pointer(reg->type);
5010 }
5011 
is_flow_key_reg(struct bpf_verifier_env * env,int regno)5012 static bool is_flow_key_reg(struct bpf_verifier_env *env, int regno)
5013 {
5014 	const struct bpf_reg_state *reg = reg_state(env, regno);
5015 
5016 	/* Separate to is_ctx_reg() since we still want to allow BPF_ST here. */
5017 	return reg->type == PTR_TO_FLOW_KEYS;
5018 }
5019 
is_arena_reg(struct bpf_verifier_env * env,int regno)5020 static bool is_arena_reg(struct bpf_verifier_env *env, int regno)
5021 {
5022 	const struct bpf_reg_state *reg = reg_state(env, regno);
5023 
5024 	return reg->type == PTR_TO_ARENA;
5025 }
5026 
is_load_acq_unsafe(struct bpf_verifier_env * env,int regno,struct bpf_insn * insn)5027 static bool is_load_acq_unsafe(struct bpf_verifier_env *env, int regno,
5028 			       struct bpf_insn *insn)
5029 {
5030 	const struct bpf_reg_state *reg = reg_state(env, regno);
5031 
5032 	/*
5033 	 * A BPF_LOAD_ACQ is not rewritten to a BPF_PROBE_MEM load by the
5034 	 * verifier, unlike a regular BPF_LDX. The JIT would emit a plain load
5035 	 * with no exception table entry, so a fault (e.g. NULL deref) crashes
5036 	 * the kernel instead of being handled. Reject the source pointer types
5037 	 * that would have needed that protection, the remaining ones stay
5038 	 * allowed.
5039 	 */
5040 	return insn->imm == BPF_LOAD_ACQ && bpf_may_fault_on_deref(reg->type);
5041 }
5042 
5043 /* Return false if @regno contains a pointer whose type isn't supported for
5044  * atomic instruction @insn.
5045  */
atomic_ptr_type_ok(struct bpf_verifier_env * env,int regno,struct bpf_insn * insn)5046 static bool atomic_ptr_type_ok(struct bpf_verifier_env *env, int regno,
5047 			       struct bpf_insn *insn)
5048 {
5049 	if (is_ctx_reg(env, regno))
5050 		return false;
5051 	if (is_pkt_reg(env, regno))
5052 		return false;
5053 	if (is_flow_key_reg(env, regno))
5054 		return false;
5055 	if (is_sk_reg(env, regno))
5056 		return false;
5057 	if (is_arena_reg(env, regno))
5058 		return bpf_jit_supports_insn(insn, true);
5059 	if (is_load_acq_unsafe(env, regno, insn))
5060 		return false;
5061 	return true;
5062 }
5063 
5064 static u32 *reg2btf_ids[__BPF_REG_TYPE_MAX] = {
5065 #ifdef CONFIG_NET
5066 	[PTR_TO_SOCKET] = &btf_sock_ids[BTF_SOCK_TYPE_SOCK],
5067 	[PTR_TO_SOCK_COMMON] = &btf_sock_ids[BTF_SOCK_TYPE_SOCK_COMMON],
5068 	[PTR_TO_TCP_SOCK] = &btf_sock_ids[BTF_SOCK_TYPE_TCP],
5069 #endif
5070 	[CONST_PTR_TO_MAP] = btf_bpf_map_id,
5071 };
5072 
lookup_reg2btf_ids(u32 ref_id)5073 static enum bpf_reg_type lookup_reg2btf_ids(u32 ref_id)
5074 {
5075 	enum bpf_reg_type type;
5076 
5077 	for (type = 0; type < __BPF_REG_TYPE_MAX; type++) {
5078 		if (reg2btf_ids[type] && *reg2btf_ids[type] == ref_id)
5079 			return type;
5080 	}
5081 
5082 	return NOT_INIT;
5083 }
5084 
is_trusted_reg(struct bpf_verifier_env * env,const struct bpf_reg_state * reg)5085 static bool is_trusted_reg(struct bpf_verifier_env *env, const struct bpf_reg_state *reg)
5086 {
5087 	/* A referenced register is always trusted. */
5088 	if (reg_is_referenced(env, reg))
5089 		return true;
5090 
5091 	/* Types listed in the reg2btf_ids are always trusted */
5092 	if (reg2btf_ids[base_type(reg->type)] &&
5093 	    !bpf_type_has_unsafe_modifiers(reg->type))
5094 		return true;
5095 
5096 	/* If a register is not referenced, it is trusted if it has the
5097 	 * MEM_ALLOC or PTR_TRUSTED type modifiers, and no others. Some of the
5098 	 * other type modifiers may be safe, but we elect to take an opt-in
5099 	 * approach here as some (e.g. PTR_UNTRUSTED and PTR_MAYBE_NULL) are
5100 	 * not.
5101 	 *
5102 	 * Eventually, we should make PTR_TRUSTED the single source of truth
5103 	 * for whether a register is trusted.
5104 	 */
5105 	return type_flag(reg->type) & BPF_REG_TRUSTED_MODIFIERS &&
5106 	       !bpf_type_has_unsafe_modifiers(reg->type);
5107 }
5108 
is_rcu_reg(const struct bpf_reg_state * reg)5109 static bool is_rcu_reg(const struct bpf_reg_state *reg)
5110 {
5111 	return reg->type & MEM_RCU;
5112 }
5113 
clear_trusted_flags(enum bpf_type_flag * flag)5114 static void clear_trusted_flags(enum bpf_type_flag *flag)
5115 {
5116 	*flag &= ~(BPF_REG_TRUSTED_MODIFIERS | MEM_RCU);
5117 }
5118 
check_pkt_ptr_alignment(struct bpf_verifier_env * env,const struct bpf_reg_state * reg,int off,int size,bool strict)5119 static int check_pkt_ptr_alignment(struct bpf_verifier_env *env,
5120 				   const struct bpf_reg_state *reg,
5121 				   int off, int size, bool strict)
5122 {
5123 	struct tnum reg_off;
5124 	int ip_align;
5125 
5126 	/* Byte size accesses are always allowed. */
5127 	if (!strict || size == 1)
5128 		return 0;
5129 
5130 	/* For platforms that do not have a Kconfig enabling
5131 	 * CONFIG_HAVE_EFFICIENT_UNALIGNED_ACCESS the value of
5132 	 * NET_IP_ALIGN is universally set to '2'.  And on platforms
5133 	 * that do set CONFIG_HAVE_EFFICIENT_UNALIGNED_ACCESS, we get
5134 	 * to this code only in strict mode where we want to emulate
5135 	 * the NET_IP_ALIGN==2 checking.  Therefore use an
5136 	 * unconditional IP align value of '2'.
5137 	 */
5138 	ip_align = 2;
5139 
5140 	reg_off = tnum_add(reg->var_off, tnum_const(ip_align + off));
5141 	if (!tnum_is_aligned(reg_off, size)) {
5142 		char tn_buf[48];
5143 
5144 		tnum_strn(tn_buf, sizeof(tn_buf), reg->var_off);
5145 		verbose(env,
5146 			"misaligned packet access off %d+%s+%d size %d\n",
5147 			ip_align, tn_buf, off, size);
5148 		return -EACCES;
5149 	}
5150 
5151 	return 0;
5152 }
5153 
check_generic_ptr_alignment(struct bpf_verifier_env * env,const struct bpf_reg_state * reg,const char * pointer_desc,int off,int size,bool strict)5154 static int check_generic_ptr_alignment(struct bpf_verifier_env *env,
5155 				       const struct bpf_reg_state *reg,
5156 				       const char *pointer_desc,
5157 				       int off, int size, bool strict)
5158 {
5159 	struct tnum reg_off;
5160 
5161 	/* Byte size accesses are always allowed. */
5162 	if (!strict || size == 1)
5163 		return 0;
5164 
5165 	reg_off = tnum_add(reg->var_off, tnum_const(off));
5166 	if (!tnum_is_aligned(reg_off, size)) {
5167 		char tn_buf[48];
5168 
5169 		tnum_strn(tn_buf, sizeof(tn_buf), reg->var_off);
5170 		verbose(env, "misaligned %saccess off %s+%d size %d\n",
5171 			pointer_desc, tn_buf, off, size);
5172 		return -EACCES;
5173 	}
5174 
5175 	return 0;
5176 }
5177 
check_ptr_alignment(struct bpf_verifier_env * env,const struct bpf_reg_state * reg,int off,int size,bool strict_alignment_once)5178 static int check_ptr_alignment(struct bpf_verifier_env *env,
5179 			       const struct bpf_reg_state *reg, int off,
5180 			       int size, bool strict_alignment_once)
5181 {
5182 	bool strict = env->strict_alignment || strict_alignment_once;
5183 	const char *pointer_desc = "";
5184 
5185 	switch (reg->type) {
5186 	case PTR_TO_PACKET:
5187 	case PTR_TO_PACKET_META:
5188 		/* Special case, because of NET_IP_ALIGN. Given metadata sits
5189 		 * right in front, treat it the very same way.
5190 		 */
5191 		return check_pkt_ptr_alignment(env, reg, off, size, strict);
5192 	case PTR_TO_FLOW_KEYS:
5193 		pointer_desc = "flow keys ";
5194 		break;
5195 	case PTR_TO_MAP_KEY:
5196 		pointer_desc = "key ";
5197 		break;
5198 	case PTR_TO_MAP_VALUE:
5199 		pointer_desc = "value ";
5200 		if (reg->map_ptr->map_type == BPF_MAP_TYPE_INSN_ARRAY)
5201 			strict = true;
5202 		break;
5203 	case PTR_TO_CTX:
5204 		pointer_desc = "context ";
5205 		break;
5206 	case PTR_TO_STACK:
5207 		pointer_desc = "stack ";
5208 		/* The stack spill tracking logic in check_stack_write_fixed_off()
5209 		 * and check_stack_read_fixed_off() relies on stack accesses being
5210 		 * aligned.
5211 		 */
5212 		strict = true;
5213 		break;
5214 	case PTR_TO_SOCKET:
5215 		pointer_desc = "sock ";
5216 		break;
5217 	case PTR_TO_SOCK_COMMON:
5218 		pointer_desc = "sock_common ";
5219 		break;
5220 	case PTR_TO_TCP_SOCK:
5221 		pointer_desc = "tcp_sock ";
5222 		break;
5223 	case PTR_TO_XDP_SOCK:
5224 		pointer_desc = "xdp_sock ";
5225 		break;
5226 	case PTR_TO_ARENA:
5227 		return 0;
5228 	default:
5229 		break;
5230 	}
5231 	return check_generic_ptr_alignment(env, reg, pointer_desc, off, size,
5232 					   strict);
5233 }
5234 
bpf_enable_priv_stack(struct bpf_prog * prog)5235 static enum priv_stack_mode bpf_enable_priv_stack(struct bpf_prog *prog)
5236 {
5237 	if (!bpf_jit_supports_private_stack())
5238 		return NO_PRIV_STACK;
5239 
5240 	/* bpf_prog_check_recur() checks all prog types that use bpf trampoline
5241 	 * while kprobe/tp/perf_event/raw_tp don't use trampoline hence checked
5242 	 * explicitly.
5243 	 */
5244 	switch (prog->type) {
5245 	case BPF_PROG_TYPE_KPROBE:
5246 	case BPF_PROG_TYPE_TRACEPOINT:
5247 	case BPF_PROG_TYPE_PERF_EVENT:
5248 	case BPF_PROG_TYPE_RAW_TRACEPOINT:
5249 		return PRIV_STACK_ADAPTIVE;
5250 	case BPF_PROG_TYPE_TRACING:
5251 	case BPF_PROG_TYPE_LSM:
5252 	case BPF_PROG_TYPE_STRUCT_OPS:
5253 		if (prog->aux->priv_stack_requested || bpf_prog_check_recur(prog))
5254 			return PRIV_STACK_ADAPTIVE;
5255 		fallthrough;
5256 	default:
5257 		break;
5258 	}
5259 
5260 	return NO_PRIV_STACK;
5261 }
5262 
round_up_stack_depth(struct bpf_verifier_env * env,int stack_depth)5263 static int round_up_stack_depth(struct bpf_verifier_env *env, int stack_depth)
5264 {
5265 	if (env->prog->jit_requested)
5266 		return round_up(stack_depth, 16);
5267 
5268 	/* round up to 32-bytes, since this is granularity
5269 	 * of interpreter stack size
5270 	 */
5271 	return round_up(max_t(u32, stack_depth, 1), 32);
5272 }
5273 
5274 /* temporary state used for call frame depth calculation */
5275 struct bpf_subprog_call_depth_info {
5276 	int ret_insn; /* caller instruction where we return to. */
5277 	int caller; /* caller subprogram idx */
5278 	int frame; /* # of consecutive static call stack frames on top of stack */
5279 };
5280 
5281 /* starting from main bpf function walk all instructions of the function
5282  * and recursively walk all callees that given function can call.
5283  * Ignore jump and exit insns.
5284  */
check_max_stack_depth_subprog(struct bpf_verifier_env * env,int idx,struct bpf_subprog_call_depth_info * dinfo,bool priv_stack_supported)5285 static int check_max_stack_depth_subprog(struct bpf_verifier_env *env, int idx,
5286 					 struct bpf_subprog_call_depth_info *dinfo,
5287 					 bool priv_stack_supported)
5288 {
5289 	struct bpf_subprog_info *subprog = env->subprog_info;
5290 	struct bpf_insn *insn = env->prog->insnsi;
5291 	int depth = 0, frame = 0, i, subprog_end, subprog_depth;
5292 	bool tail_call_reachable = false;
5293 	int total;
5294 	int tmp;
5295 
5296 	/* no caller idx */
5297 	dinfo[idx].caller = -1;
5298 
5299 	i = subprog[idx].start;
5300 	if (!priv_stack_supported)
5301 		subprog[idx].priv_stack_mode = NO_PRIV_STACK;
5302 process_func:
5303 	/* protect against potential stack overflow that might happen when
5304 	 * bpf2bpf calls get combined with tailcalls. Limit the caller's stack
5305 	 * depth for such case down to 256 so that the worst case scenario
5306 	 * would result in 8k stack size (32 which is tailcall limit * 256 =
5307 	 * 8k).
5308 	 *
5309 	 * To get the idea what might happen, see an example:
5310 	 * func1 -> sub rsp, 128
5311 	 *  subfunc1 -> sub rsp, 256
5312 	 *  tailcall1 -> add rsp, 256
5313 	 *   func2 -> sub rsp, 192 (total stack size = 128 + 192 = 320)
5314 	 *   subfunc2 -> sub rsp, 64
5315 	 *   subfunc22 -> sub rsp, 128
5316 	 *   tailcall2 -> add rsp, 128
5317 	 *    func3 -> sub rsp, 32 (total stack size 128 + 192 + 64 + 32 = 416)
5318 	 *
5319 	 * tailcall will unwind the current stack frame but it will not get rid
5320 	 * of caller's stack as shown on the example above.
5321 	 */
5322 	if (idx && subprog[idx].has_tail_call && depth >= 256) {
5323 		verbose(env,
5324 			"tail_calls are not allowed when call stack of previous frames is %d bytes. Too large\n",
5325 			depth);
5326 		return -EACCES;
5327 	}
5328 
5329 	subprog_depth = round_up_stack_depth(env, subprog[idx].stack_depth);
5330 	if (IS_ENABLED(CONFIG_X86_64) && subprog[idx].stack_arg_cnt) {
5331 		/* x86-64 uses R9 for both private stack frame pointer and arg6. */
5332 		subprog[idx].priv_stack_mode = NO_PRIV_STACK;
5333 	} else if (priv_stack_supported) {
5334 		/* Request private stack support only if the subprog stack
5335 		 * depth is no less than BPF_PRIV_STACK_MIN_SIZE. This is to
5336 		 * avoid jit penalty if the stack usage is small.
5337 		 */
5338 		if (subprog[idx].priv_stack_mode == PRIV_STACK_UNKNOWN &&
5339 		    subprog_depth >= BPF_PRIV_STACK_MIN_SIZE)
5340 			subprog[idx].priv_stack_mode = PRIV_STACK_ADAPTIVE;
5341 	}
5342 
5343 	if (subprog[idx].priv_stack_mode == PRIV_STACK_ADAPTIVE) {
5344 		if (subprog_depth > env->max_stack_depth)
5345 			env->max_stack_depth = subprog_depth;
5346 		if (subprog_depth > MAX_BPF_STACK) {
5347 			verbose(env, "stack size of subprog %d is %d. Too large\n",
5348 				idx, subprog_depth);
5349 			return -EACCES;
5350 		}
5351 	} else {
5352 		depth += subprog_depth;
5353 		if (depth > env->max_stack_depth)
5354 			env->max_stack_depth = depth;
5355 		if (depth > MAX_BPF_STACK) {
5356 			total = 0;
5357 			for (tmp = idx; tmp >= 0; tmp = dinfo[tmp].caller)
5358 				total++;
5359 
5360 			verbose(env, "combined stack size of %d calls is %d. Too large\n",
5361 				total, depth);
5362 			return -EACCES;
5363 		}
5364 	}
5365 continue_func:
5366 	subprog_end = subprog[idx + 1].start;
5367 	for (; i < subprog_end; i++) {
5368 		int next_insn, sidx;
5369 
5370 		if (bpf_pseudo_kfunc_call(insn + i) && !insn[i].off) {
5371 			bool err = false;
5372 
5373 			if (!bpf_is_throw_kfunc(insn + i))
5374 				continue;
5375 			for (tmp = idx; tmp >= 0 && !err; tmp = dinfo[tmp].caller) {
5376 				if (subprog[tmp].is_cb) {
5377 					err = true;
5378 					break;
5379 				}
5380 			}
5381 			if (!err)
5382 				continue;
5383 			verbose(env,
5384 				"bpf_throw kfunc (insn %d) cannot be called from callback subprog %d\n",
5385 				i, idx);
5386 			return -EINVAL;
5387 		}
5388 
5389 		if (!bpf_pseudo_call(insn + i) && !bpf_pseudo_func(insn + i))
5390 			continue;
5391 		/* remember insn and function to return to */
5392 
5393 		/* find the callee */
5394 		next_insn = i + insn[i].imm + 1;
5395 		sidx = bpf_find_subprog(env, next_insn);
5396 		if (verifier_bug_if(sidx < 0, env, "callee not found at insn %d", next_insn))
5397 			return -EFAULT;
5398 		if (subprog[sidx].is_async_cb) {
5399 			/* async callbacks don't increase bpf prog stack size unless called directly */
5400 			if (!bpf_pseudo_call(insn + i))
5401 				continue;
5402 			if (subprog[sidx].is_exception_cb) {
5403 				verbose(env, "insn %d cannot call exception cb directly", i);
5404 				return -EINVAL;
5405 			}
5406 		}
5407 
5408 		/* store caller info for after we return from callee */
5409 		dinfo[idx].frame = frame;
5410 		dinfo[idx].ret_insn = i + 1;
5411 
5412 		/* push caller idx into callee's dinfo */
5413 		dinfo[sidx].caller = idx;
5414 
5415 		i = next_insn;
5416 
5417 		idx = sidx;
5418 		if (!priv_stack_supported)
5419 			subprog[idx].priv_stack_mode = NO_PRIV_STACK;
5420 
5421 		/* sync tail_call_reachable with callee state on entry */
5422 		tail_call_reachable = subprog[idx].has_tail_call;
5423 
5424 		frame = bpf_subprog_is_global(env, idx) ? 0 : frame + 1;
5425 		if (frame >= MAX_CALL_FRAMES) {
5426 			verbose(env, "the call stack of %d frames is too deep !\n",
5427 				frame);
5428 			return -E2BIG;
5429 		}
5430 		goto process_func;
5431 	}
5432 	/* if tail call got detected across bpf2bpf calls then mark each of the
5433 	 * currently present subprog frames as tail call reachable subprogs;
5434 	 * this info will be utilized by JIT so that we will be preserving the
5435 	 * tail call counter throughout bpf2bpf calls combined with tailcalls
5436 	 */
5437 	if (tail_call_reachable) {
5438 		for (tmp = idx; tmp >= 0; tmp = dinfo[tmp].caller) {
5439 			if (subprog[tmp].is_cb) {
5440 				verbose(env, "cannot tail call within callback\n");
5441 				return -EINVAL;
5442 			}
5443 			if (subprog[tmp].stack_arg_cnt) {
5444 				verbose(env, "tail_calls are not allowed in programs with stack args\n");
5445 				return -EINVAL;
5446 			}
5447 			subprog[tmp].tail_call_reachable = true;
5448 		}
5449 	} else if (!idx && subprog[0].has_tail_call && subprog[0].stack_arg_cnt) {
5450 		verbose(env, "tail_calls are not allowed in programs with stack args\n");
5451 		return -EINVAL;
5452 	}
5453 
5454 	if (subprog[0].tail_call_reachable)
5455 		env->prog->aux->tail_call_reachable = true;
5456 
5457 	/* end of for() loop means the last insn of the 'subprog'
5458 	 * was reached. Doesn't matter whether it was JA or EXIT
5459 	 */
5460 	if (frame == 0 && dinfo[idx].caller < 0)
5461 		return 0;
5462 	if (subprog[idx].priv_stack_mode != PRIV_STACK_ADAPTIVE)
5463 		depth -= round_up_stack_depth(env, subprog[idx].stack_depth);
5464 
5465 	/* pop caller idx from callee */
5466 	idx = dinfo[idx].caller;
5467 
5468 	/* retrieve caller state from its frame */
5469 	frame = dinfo[idx].frame;
5470 	i = dinfo[idx].ret_insn;
5471 
5472 	/* reset tail_call_reachable to the parent's actual state */
5473 	tail_call_reachable = subprog[idx].tail_call_reachable;
5474 
5475 	goto continue_func;
5476 }
5477 
check_max_stack_depth(struct bpf_verifier_env * env)5478 static int check_max_stack_depth(struct bpf_verifier_env *env)
5479 {
5480 	enum priv_stack_mode priv_stack_mode = PRIV_STACK_UNKNOWN;
5481 	struct bpf_subprog_call_depth_info *dinfo;
5482 	struct bpf_subprog_info *si = env->subprog_info;
5483 	bool priv_stack_supported;
5484 	int ret;
5485 
5486 	dinfo = kvcalloc(env->subprog_cnt, sizeof(*dinfo), GFP_KERNEL_ACCOUNT);
5487 	if (!dinfo)
5488 		return -ENOMEM;
5489 
5490 	for (int i = 0; i < env->subprog_cnt; i++) {
5491 		if (si[i].has_tail_call) {
5492 			priv_stack_mode = NO_PRIV_STACK;
5493 			break;
5494 		}
5495 	}
5496 
5497 	if (priv_stack_mode == PRIV_STACK_UNKNOWN)
5498 		priv_stack_mode = bpf_enable_priv_stack(env->prog);
5499 
5500 	/* All async_cb subprogs use normal kernel stack. If a particular
5501 	 * subprog appears in both main prog and async_cb subtree, that
5502 	 * subprog will use normal kernel stack to avoid potential nesting.
5503 	 * The reverse subprog traversal ensures when main prog subtree is
5504 	 * checked, the subprogs appearing in async_cb subtrees are already
5505 	 * marked as using normal kernel stack, so stack size checking can
5506 	 * be done properly.
5507 	 */
5508 	for (int i = env->subprog_cnt - 1; i >= 0; i--) {
5509 		if (!i || si[i].is_async_cb) {
5510 			priv_stack_supported = !i && priv_stack_mode == PRIV_STACK_ADAPTIVE;
5511 			ret = check_max_stack_depth_subprog(env, i, dinfo,
5512 					priv_stack_supported);
5513 			if (ret < 0) {
5514 				kvfree(dinfo);
5515 				return ret;
5516 			}
5517 		}
5518 	}
5519 
5520 	for (int i = 0; i < env->subprog_cnt; i++) {
5521 		if (si[i].priv_stack_mode == PRIV_STACK_ADAPTIVE) {
5522 			env->prog->aux->jits_use_priv_stack = true;
5523 			break;
5524 		}
5525 	}
5526 
5527 	kvfree(dinfo);
5528 
5529 	return 0;
5530 }
5531 
__check_buffer_access(struct bpf_verifier_env * env,const char * buf_info,const struct bpf_reg_state * reg,argno_t argno,int off,int size,u32 * access_end)5532 static int __check_buffer_access(struct bpf_verifier_env *env,
5533 				 const char *buf_info,
5534 				 const struct bpf_reg_state *reg,
5535 				 argno_t argno, int off, int size,
5536 				 u32 *access_end)
5537 {
5538 	s64 start;
5539 
5540 	if (!tnum_is_const(reg->var_off)) {
5541 		char tn_buf[48];
5542 
5543 		tnum_strn(tn_buf, sizeof(tn_buf), reg->var_off);
5544 		verbose(env,
5545 			"%s invalid variable buffer offset: off=%d, var_off=%s\n",
5546 			reg_arg_name(env, argno), off, tn_buf);
5547 		return -EACCES;
5548 	}
5549 
5550 	start = (s64)reg->var_off.value + off;
5551 	if (start < 0) {
5552 		verbose(env,
5553 			"%s invalid negative %s buffer offset: off=%d, var_off=%lld\n",
5554 			reg_arg_name(env, argno), buf_info, off, (s64)reg->var_off.value);
5555 		return -EACCES;
5556 	}
5557 
5558 	*access_end = start + size;
5559 	return 0;
5560 }
5561 
check_tp_buffer_access(struct bpf_verifier_env * env,const struct bpf_reg_state * reg,argno_t argno,int off,int size)5562 static int check_tp_buffer_access(struct bpf_verifier_env *env,
5563 				  const struct bpf_reg_state *reg,
5564 				  argno_t argno, int off, int size)
5565 {
5566 	u32 access_end;
5567 	int err;
5568 
5569 	err = __check_buffer_access(env, "tracepoint", reg, argno, off, size, &access_end);
5570 	if (err)
5571 		return err;
5572 
5573 	env->prog->aux->max_tp_access = max(access_end, env->prog->aux->max_tp_access);
5574 
5575 	return 0;
5576 }
5577 
check_buffer_access(struct bpf_verifier_env * env,const struct bpf_reg_state * reg,argno_t argno,int off,int size,bool zero_size_allowed,u32 * max_access)5578 static int check_buffer_access(struct bpf_verifier_env *env,
5579 			       const struct bpf_reg_state *reg,
5580 			       argno_t argno, int off, int size,
5581 			       bool zero_size_allowed,
5582 			       u32 *max_access)
5583 {
5584 	const char *buf_info = type_is_rdonly_mem(reg->type) ? "rdonly" : "rdwr";
5585 	u32 access_end;
5586 	int err;
5587 
5588 	err = __check_buffer_access(env, buf_info, reg, argno, off, size, &access_end);
5589 	if (err)
5590 		return err;
5591 
5592 	*max_access = max(access_end, *max_access);
5593 
5594 	return 0;
5595 }
5596 
5597 /* BPF architecture zero extends alu32 ops into 64-bit registesr */
zext_32_to_64(struct bpf_reg_state * reg)5598 static void zext_32_to_64(struct bpf_reg_state *reg)
5599 {
5600 	reg->var_off = tnum_subreg(reg->var_off);
5601 	reg_set_urange64(reg, reg_u32_min(reg), reg_u32_max(reg));
5602 }
5603 
5604 /* truncate register to smaller size (in bytes)
5605  * must be called with size < BPF_REG_SIZE
5606  */
coerce_reg_to_size(struct bpf_reg_state * reg,int size)5607 static void coerce_reg_to_size(struct bpf_reg_state *reg, int size)
5608 {
5609 	u64 mask;
5610 
5611 	/* clear high bits in bit representation */
5612 	reg->var_off = tnum_cast(reg->var_off, size);
5613 
5614 	/* fix arithmetic bounds */
5615 	mask = ((u64)1 << (size * 8)) - 1;
5616 	if ((reg_umin(reg) & ~mask) == (reg_umax(reg) & ~mask))
5617 		reg_set_urange64(reg, reg_umin(reg) & mask, reg_umax(reg) & mask);
5618 	else
5619 		reg_set_urange64(reg, 0, mask);
5620 
5621 	/* If size is smaller than 32bit register the 32bit register
5622 	 * values are also truncated so we push 64-bit bounds into
5623 	 * 32-bit bounds. Above were truncated < 32-bits already.
5624 	 */
5625 	if (size < 4)
5626 		__mark_reg32_unbounded(reg);
5627 
5628 	reg_bounds_sync(reg);
5629 }
5630 
set_sext64_default_val(struct bpf_reg_state * reg,int size)5631 static void set_sext64_default_val(struct bpf_reg_state *reg, int size)
5632 {
5633 	if (size == 1) {
5634 		reg_set_srange64(reg, S8_MIN, S8_MAX);
5635 		reg_set_srange32(reg, S8_MIN, S8_MAX);
5636 	} else if (size == 2) {
5637 		reg_set_srange64(reg, S16_MIN, S16_MAX);
5638 		reg_set_srange32(reg, S16_MIN, S16_MAX);
5639 	} else {
5640 		/* size == 4 */
5641 		reg_set_srange64(reg, S32_MIN, S32_MAX);
5642 		reg_set_srange32(reg, S32_MIN, S32_MAX);
5643 	}
5644 	reg->var_off = tnum_unknown;
5645 }
5646 
coerce_reg_to_size_sx(struct bpf_reg_state * reg,int size)5647 static void coerce_reg_to_size_sx(struct bpf_reg_state *reg, int size)
5648 {
5649 	s64 init_s64_max, init_s64_min, s64_max, s64_min, u64_cval;
5650 	u64 top_smax_value, top_smin_value;
5651 	u64 num_bits = size * 8;
5652 
5653 	if (tnum_is_const(reg->var_off)) {
5654 		u64_cval = reg->var_off.value;
5655 		if (size == 1)
5656 			reg->var_off = tnum_const((s8)u64_cval);
5657 		else if (size == 2)
5658 			reg->var_off = tnum_const((s16)u64_cval);
5659 		else
5660 			/* size == 4 */
5661 			reg->var_off = tnum_const((s32)u64_cval);
5662 
5663 		u64_cval = reg->var_off.value;
5664 		reg->r64 = cnum64_from_urange(u64_cval, u64_cval);
5665 		reg->r32 = cnum32_from_urange((u32)u64_cval, (u32)u64_cval);
5666 		return;
5667 	}
5668 
5669 	top_smax_value = ((u64)reg_smax(reg) >> num_bits) << num_bits;
5670 	top_smin_value = ((u64)reg_smin(reg) >> num_bits) << num_bits;
5671 
5672 	if (top_smax_value != top_smin_value)
5673 		goto out;
5674 
5675 	/* find the s64_min and s64_min after sign extension */
5676 	if (size == 1) {
5677 		init_s64_max = (s8)reg_smax(reg);
5678 		init_s64_min = (s8)reg_smin(reg);
5679 	} else if (size == 2) {
5680 		init_s64_max = (s16)reg_smax(reg);
5681 		init_s64_min = (s16)reg_smin(reg);
5682 	} else {
5683 		init_s64_max = (s32)reg_smax(reg);
5684 		init_s64_min = (s32)reg_smin(reg);
5685 	}
5686 
5687 	s64_max = max(init_s64_max, init_s64_min);
5688 	s64_min = min(init_s64_max, init_s64_min);
5689 
5690 	/* both of s64_max/s64_min positive or negative */
5691 	if ((s64_max >= 0) == (s64_min >= 0)) {
5692 		reg_set_srange64(reg, s64_min, s64_max);
5693 		reg_set_srange32(reg, s64_min, s64_max);
5694 		reg->var_off = tnum_range(s64_min, s64_max);
5695 		return;
5696 	}
5697 
5698 out:
5699 	set_sext64_default_val(reg, size);
5700 }
5701 
set_sext32_default_val(struct bpf_reg_state * reg,int size)5702 static void set_sext32_default_val(struct bpf_reg_state *reg, int size)
5703 {
5704 	if (size == 1)
5705 		reg_set_srange32(reg, S8_MIN, S8_MAX);
5706 	else
5707 		/* size == 2 */
5708 		reg_set_srange32(reg, S16_MIN, S16_MAX);
5709 	reg->var_off = tnum_subreg(tnum_unknown);
5710 }
5711 
coerce_subreg_to_size_sx(struct bpf_reg_state * reg,int size)5712 static void coerce_subreg_to_size_sx(struct bpf_reg_state *reg, int size)
5713 {
5714 	s32 init_s32_max, init_s32_min, s32_max, s32_min, u32_val;
5715 	u32 top_smax_value, top_smin_value;
5716 	u32 num_bits = size * 8;
5717 
5718 	if (tnum_is_const(reg->var_off)) {
5719 		u32_val = reg->var_off.value;
5720 		if (size == 1)
5721 			reg->var_off = tnum_const((s8)u32_val);
5722 		else
5723 			reg->var_off = tnum_const((s16)u32_val);
5724 
5725 		u32_val = reg->var_off.value;
5726 		reg_set_srange32(reg, u32_val, u32_val);
5727 		return;
5728 	}
5729 
5730 	top_smax_value = ((u32)reg_s32_max(reg) >> num_bits) << num_bits;
5731 	top_smin_value = ((u32)reg_s32_min(reg) >> num_bits) << num_bits;
5732 
5733 	if (top_smax_value != top_smin_value)
5734 		goto out;
5735 
5736 	/* find the s32_min and s32_min after sign extension */
5737 	if (size == 1) {
5738 		init_s32_max = (s8)reg_s32_max(reg);
5739 		init_s32_min = (s8)reg_s32_min(reg);
5740 	} else {
5741 		/* size == 2 */
5742 		init_s32_max = (s16)reg_s32_max(reg);
5743 		init_s32_min = (s16)reg_s32_min(reg);
5744 	}
5745 	s32_max = max(init_s32_max, init_s32_min);
5746 	s32_min = min(init_s32_max, init_s32_min);
5747 
5748 	if ((s32_min >= 0) == (s32_max >= 0)) {
5749 		reg_set_srange32(reg, s32_min, s32_max);
5750 		reg->var_off = tnum_subreg(tnum_range(s32_min, s32_max));
5751 		return;
5752 	}
5753 
5754 out:
5755 	set_sext32_default_val(reg, size);
5756 }
5757 
bpf_map_is_rdonly(const struct bpf_map * map)5758 bool bpf_map_is_rdonly(const struct bpf_map *map)
5759 {
5760 	/* A map is considered read-only if the following condition are true:
5761 	 *
5762 	 * 1) BPF program side cannot change any of the map content. The
5763 	 *    BPF_F_RDONLY_PROG flag is throughout the lifetime of a map
5764 	 *    and was set at map creation time.
5765 	 * 2) The map value(s) have been initialized from user space by a
5766 	 *    loader and then "frozen", such that no new map update/delete
5767 	 *    operations from syscall side are possible for the rest of
5768 	 *    the map's lifetime from that point onwards.
5769 	 * 3) Any parallel/pending map update/delete operations from syscall
5770 	 *    side have been completed. Only after that point, it's safe to
5771 	 *    assume that map value(s) are immutable.
5772 	 */
5773 	return (map->map_flags & BPF_F_RDONLY_PROG) &&
5774 	       READ_ONCE(map->frozen) &&
5775 	       !bpf_map_write_active(map);
5776 }
5777 
bpf_map_direct_read(struct bpf_map * map,int off,int size,u64 * val,bool is_ldsx)5778 int bpf_map_direct_read(struct bpf_map *map, int off, int size, u64 *val,
5779 			bool is_ldsx)
5780 {
5781 	void *ptr;
5782 	u64 addr;
5783 	int err;
5784 
5785 	if (map->map_type == BPF_MAP_TYPE_INSN_ARRAY || map->map_type == BPF_MAP_TYPE_PERCPU_ARRAY)
5786 		return -EINVAL;
5787 	err = map->ops->map_direct_value_addr(map, &addr, off);
5788 	if (err)
5789 		return err;
5790 	ptr = (void *)(long)addr + off;
5791 
5792 	switch (size) {
5793 	case sizeof(u8):
5794 		*val = is_ldsx ? (s64)*(s8 *)ptr : (u64)*(u8 *)ptr;
5795 		break;
5796 	case sizeof(u16):
5797 		*val = is_ldsx ? (s64)*(s16 *)ptr : (u64)*(u16 *)ptr;
5798 		break;
5799 	case sizeof(u32):
5800 		*val = is_ldsx ? (s64)*(s32 *)ptr : (u64)*(u32 *)ptr;
5801 		break;
5802 	case sizeof(u64):
5803 		*val = *(u64 *)ptr;
5804 		break;
5805 	default:
5806 		return -EINVAL;
5807 	}
5808 	return 0;
5809 }
5810 
5811 #define BTF_TYPE_SAFE_RCU(__type)  __PASTE(__type, __safe_rcu)
5812 #define BTF_TYPE_SAFE_RCU_OR_NULL(__type)  __PASTE(__type, __safe_rcu_or_null)
5813 #define BTF_TYPE_SAFE_TRUSTED(__type)  __PASTE(__type, __safe_trusted)
5814 #define BTF_TYPE_SAFE_TRUSTED_OR_NULL(__type)  __PASTE(__type, __safe_trusted_or_null)
5815 
5816 /*
5817  * Allow list few fields as RCU trusted or full trusted.
5818  * This logic doesn't allow mix tagging and will be removed once GCC supports
5819  * btf_type_tag.
5820  */
5821 
5822 /* RCU trusted: these fields are trusted in RCU CS and never NULL */
BTF_TYPE_SAFE_RCU(struct task_struct)5823 BTF_TYPE_SAFE_RCU(struct task_struct) {
5824 	const cpumask_t *cpus_ptr;
5825 	struct css_set __rcu *cgroups;
5826 	struct task_struct __rcu *real_parent;
5827 	struct task_struct *group_leader;
5828 };
5829 
BTF_TYPE_SAFE_RCU(struct cgroup)5830 BTF_TYPE_SAFE_RCU(struct cgroup) {
5831 	/* cgrp->kn is always accessible as documented in kernel/cgroup/cgroup.c */
5832 	struct kernfs_node *kn;
5833 };
5834 
BTF_TYPE_SAFE_RCU(struct css_set)5835 BTF_TYPE_SAFE_RCU(struct css_set) {
5836 	struct cgroup *dfl_cgrp;
5837 };
5838 
BTF_TYPE_SAFE_RCU(struct cgroup_subsys_state)5839 BTF_TYPE_SAFE_RCU(struct cgroup_subsys_state) {
5840 	struct cgroup *cgroup;
5841 };
5842 
5843 /* RCU trusted: these fields are trusted in RCU CS and can be NULL */
BTF_TYPE_SAFE_RCU_OR_NULL(struct mm_struct)5844 BTF_TYPE_SAFE_RCU_OR_NULL(struct mm_struct) {
5845 	struct file __rcu *exe_file;
5846 #ifdef CONFIG_MEMCG
5847 	struct task_struct __rcu *owner;
5848 #endif
5849 };
5850 
5851 /* skb->sk, req->sk are not RCU protected, but we mark them as such
5852  * because bpf prog accessible sockets are SOCK_RCU_FREE.
5853  */
BTF_TYPE_SAFE_RCU_OR_NULL(struct sk_buff)5854 BTF_TYPE_SAFE_RCU_OR_NULL(struct sk_buff) {
5855 	struct sock *sk;
5856 };
5857 
BTF_TYPE_SAFE_RCU_OR_NULL(struct request_sock)5858 BTF_TYPE_SAFE_RCU_OR_NULL(struct request_sock) {
5859 	struct sock *sk;
5860 };
5861 
5862 /* full trusted: these fields are trusted even outside of RCU CS and never NULL */
BTF_TYPE_SAFE_TRUSTED(struct bpf_iter_meta)5863 BTF_TYPE_SAFE_TRUSTED(struct bpf_iter_meta) {
5864 	struct seq_file *seq;
5865 };
5866 
BTF_TYPE_SAFE_TRUSTED(struct bpf_iter__task)5867 BTF_TYPE_SAFE_TRUSTED(struct bpf_iter__task) {
5868 	struct bpf_iter_meta *meta;
5869 	struct task_struct *task;
5870 };
5871 
BTF_TYPE_SAFE_TRUSTED(struct linux_binprm)5872 BTF_TYPE_SAFE_TRUSTED(struct linux_binprm) {
5873 	struct file *file;
5874 };
5875 
BTF_TYPE_SAFE_TRUSTED(struct file)5876 BTF_TYPE_SAFE_TRUSTED(struct file) {
5877 	struct inode *f_inode;
5878 };
5879 
BTF_TYPE_SAFE_TRUSTED_OR_NULL(struct dentry)5880 BTF_TYPE_SAFE_TRUSTED_OR_NULL(struct dentry) {
5881 	struct inode *d_inode;
5882 };
5883 
BTF_TYPE_SAFE_TRUSTED_OR_NULL(struct socket)5884 BTF_TYPE_SAFE_TRUSTED_OR_NULL(struct socket) {
5885 	struct sock *sk;
5886 };
5887 
BTF_TYPE_SAFE_TRUSTED_OR_NULL(struct vm_area_struct)5888 BTF_TYPE_SAFE_TRUSTED_OR_NULL(struct vm_area_struct) {
5889 	struct mm_struct *vm_mm;
5890 	struct file *vm_file;
5891 };
5892 
type_is_rcu(struct bpf_verifier_env * env,struct bpf_reg_state * reg,const char * field_name,u32 btf_id)5893 static bool type_is_rcu(struct bpf_verifier_env *env,
5894 			struct bpf_reg_state *reg,
5895 			const char *field_name, u32 btf_id)
5896 {
5897 	BTF_TYPE_EMIT(BTF_TYPE_SAFE_RCU(struct task_struct));
5898 	BTF_TYPE_EMIT(BTF_TYPE_SAFE_RCU(struct cgroup));
5899 	BTF_TYPE_EMIT(BTF_TYPE_SAFE_RCU(struct css_set));
5900 	BTF_TYPE_EMIT(BTF_TYPE_SAFE_RCU(struct cgroup_subsys_state));
5901 
5902 	return btf_nested_type_is_trusted(&env->log, reg, field_name, btf_id, "__safe_rcu");
5903 }
5904 
type_is_rcu_or_null(struct bpf_verifier_env * env,struct bpf_reg_state * reg,const char * field_name,u32 btf_id)5905 static bool type_is_rcu_or_null(struct bpf_verifier_env *env,
5906 				struct bpf_reg_state *reg,
5907 				const char *field_name, u32 btf_id)
5908 {
5909 	BTF_TYPE_EMIT(BTF_TYPE_SAFE_RCU_OR_NULL(struct mm_struct));
5910 	BTF_TYPE_EMIT(BTF_TYPE_SAFE_RCU_OR_NULL(struct sk_buff));
5911 	BTF_TYPE_EMIT(BTF_TYPE_SAFE_RCU_OR_NULL(struct request_sock));
5912 
5913 	return btf_nested_type_is_trusted(&env->log, reg, field_name, btf_id, "__safe_rcu_or_null");
5914 }
5915 
type_is_trusted(struct bpf_verifier_env * env,struct bpf_reg_state * reg,const char * field_name,u32 btf_id)5916 static bool type_is_trusted(struct bpf_verifier_env *env,
5917 			    struct bpf_reg_state *reg,
5918 			    const char *field_name, u32 btf_id)
5919 {
5920 	BTF_TYPE_EMIT(BTF_TYPE_SAFE_TRUSTED(struct bpf_iter_meta));
5921 	BTF_TYPE_EMIT(BTF_TYPE_SAFE_TRUSTED(struct bpf_iter__task));
5922 	BTF_TYPE_EMIT(BTF_TYPE_SAFE_TRUSTED(struct linux_binprm));
5923 	BTF_TYPE_EMIT(BTF_TYPE_SAFE_TRUSTED(struct file));
5924 
5925 	return btf_nested_type_is_trusted(&env->log, reg, field_name, btf_id, "__safe_trusted");
5926 }
5927 
type_is_trusted_or_null(struct bpf_verifier_env * env,struct bpf_reg_state * reg,const char * field_name,u32 btf_id)5928 static bool type_is_trusted_or_null(struct bpf_verifier_env *env,
5929 				    struct bpf_reg_state *reg,
5930 				    const char *field_name, u32 btf_id)
5931 {
5932 	BTF_TYPE_EMIT(BTF_TYPE_SAFE_TRUSTED_OR_NULL(struct socket));
5933 	BTF_TYPE_EMIT(BTF_TYPE_SAFE_TRUSTED_OR_NULL(struct dentry));
5934 	BTF_TYPE_EMIT(BTF_TYPE_SAFE_TRUSTED_OR_NULL(struct vm_area_struct));
5935 
5936 	return btf_nested_type_is_trusted(&env->log, reg, field_name, btf_id,
5937 					  "__safe_trusted_or_null");
5938 }
5939 
check_ptr_to_btf_access(struct bpf_verifier_env * env,struct bpf_reg_state * regs,struct bpf_reg_state * reg,argno_t argno,int off,int size,enum bpf_access_type atype,int value_regno)5940 static int check_ptr_to_btf_access(struct bpf_verifier_env *env,
5941 				   struct bpf_reg_state *regs, struct bpf_reg_state *reg,
5942 				   argno_t argno, int off, int size,
5943 				   enum bpf_access_type atype,
5944 				   int value_regno)
5945 {
5946 	const struct btf_type *t = btf_type_by_id(reg->btf, reg->btf_id);
5947 	const char *tname = btf_name_by_offset(reg->btf, t->name_off);
5948 	const char *field_name = NULL;
5949 	enum bpf_type_flag flag = 0;
5950 	u32 btf_id = 0;
5951 	int ret;
5952 
5953 	if (!env->allow_ptr_leaks) {
5954 		verbose(env,
5955 			"'struct %s' access is allowed only to CAP_PERFMON and CAP_SYS_ADMIN\n",
5956 			tname);
5957 		return -EPERM;
5958 	}
5959 	if (!env->prog->gpl_compatible && btf_is_kernel(reg->btf)) {
5960 		verbose(env,
5961 			"Cannot access kernel 'struct %s' from non-GPL compatible program\n",
5962 			tname);
5963 		return -EINVAL;
5964 	}
5965 
5966 	if (!tnum_is_const(reg->var_off)) {
5967 		char tn_buf[48];
5968 
5969 		tnum_strn(tn_buf, sizeof(tn_buf), reg->var_off);
5970 		verbose(env,
5971 			"%s is ptr_%s invalid variable offset: off=%d, var_off=%s\n",
5972 			reg_arg_name(env, argno), tname, off, tn_buf);
5973 		return -EACCES;
5974 	}
5975 
5976 	off += reg->var_off.value;
5977 
5978 	if (off < 0) {
5979 		verbose(env,
5980 			"%s is ptr_%s invalid negative access: off=%d\n",
5981 			reg_arg_name(env, argno), tname, off);
5982 		return -EACCES;
5983 	}
5984 
5985 	if (reg->type & MEM_USER) {
5986 		verbose(env,
5987 			"%s is ptr_%s access user memory: off=%d\n",
5988 			reg_arg_name(env, argno), tname, off);
5989 		return -EACCES;
5990 	}
5991 
5992 	if (reg->type & MEM_PERCPU) {
5993 		verbose(env,
5994 			"%s is ptr_%s access percpu memory: off=%d\n",
5995 			reg_arg_name(env, argno), tname, off);
5996 		return -EACCES;
5997 	}
5998 
5999 	if (atype != BPF_READ && bpf_may_fault_on_deref(reg->type)) {
6000 		verbose(env, "only read is supported\n");
6001 		return -EACCES;
6002 	}
6003 
6004 	if (env->ops->btf_struct_access && !type_is_alloc(reg->type) && atype == BPF_WRITE) {
6005 		if (!btf_is_kernel(reg->btf)) {
6006 			verifier_bug(env, "reg->btf must be kernel btf");
6007 			return -EFAULT;
6008 		}
6009 		ret = env->ops->btf_struct_access(&env->log, reg, off, size);
6010 		if (ret < 0)
6011 			verbose(env,
6012 				"%s cannot write into ptr_%s at off=%d size=%d\n",
6013 				reg_arg_name(env, argno), tname, off, size);
6014 	} else {
6015 		/* Writes are permitted with default btf_struct_access for
6016 		 * program allocated objects (which always have id > 0).
6017 		 */
6018 		if (atype != BPF_READ && !type_is_ptr_alloc_obj(reg->type)) {
6019 			verbose(env, "only read is supported\n");
6020 			return -EACCES;
6021 		}
6022 
6023 		if (type_is_alloc(reg->type) && !type_is_non_owning_ref(reg->type) &&
6024 		    !(reg->type & MEM_RCU) && !reg_is_referenced(env, reg)) {
6025 			verifier_bug(env, "allocated object must have a referenced id");
6026 			return -EFAULT;
6027 		}
6028 
6029 		ret = btf_struct_access(&env->log, reg, off, size, atype, &btf_id, &flag, &field_name);
6030 	}
6031 
6032 	if (ret < 0)
6033 		return ret;
6034 
6035 	if (ret != PTR_TO_BTF_ID) {
6036 		/* just mark; */
6037 
6038 	} else if (type_flag(reg->type) & PTR_UNTRUSTED) {
6039 		/* If this is an untrusted pointer, all pointers formed by walking it
6040 		 * also inherit the untrusted flag.
6041 		 */
6042 		flag = PTR_UNTRUSTED;
6043 
6044 	} else if (is_trusted_reg(env, reg) || is_rcu_reg(reg)) {
6045 		/* By default any pointer obtained from walking a trusted pointer is no
6046 		 * longer trusted, unless the field being accessed has explicitly been
6047 		 * marked as inheriting its parent's state of trust (either full or RCU).
6048 		 * For example:
6049 		 * 'cgroups' pointer is untrusted if task->cgroups dereference
6050 		 * happened in a sleepable program outside of bpf_rcu_read_lock()
6051 		 * section. In a non-sleepable program it's trusted while in RCU CS (aka MEM_RCU).
6052 		 * Note bpf_rcu_read_unlock() converts MEM_RCU pointers to PTR_UNTRUSTED.
6053 		 *
6054 		 * A regular RCU-protected pointer with __rcu tag can also be deemed
6055 		 * trusted if we are in an RCU CS. Such pointer can be NULL.
6056 		 */
6057 		if (type_is_trusted(env, reg, field_name, btf_id)) {
6058 			flag |= PTR_TRUSTED;
6059 		} else if (type_is_trusted_or_null(env, reg, field_name, btf_id)) {
6060 			flag |= PTR_TRUSTED | PTR_MAYBE_NULL;
6061 		} else if (in_rcu_cs(env) && !type_may_be_null(reg->type)) {
6062 			if (type_is_rcu(env, reg, field_name, btf_id)) {
6063 				/* ignore __rcu tag and mark it MEM_RCU */
6064 				flag |= MEM_RCU;
6065 			} else if (flag & MEM_RCU ||
6066 				   type_is_rcu_or_null(env, reg, field_name, btf_id)) {
6067 				/* __rcu tagged pointers can be NULL */
6068 				flag |= MEM_RCU | PTR_MAYBE_NULL;
6069 
6070 				/* We always trust them */
6071 				if (type_is_rcu_or_null(env, reg, field_name, btf_id) &&
6072 				    flag & PTR_UNTRUSTED)
6073 					flag &= ~PTR_UNTRUSTED;
6074 			} else if (flag & (MEM_PERCPU | MEM_USER)) {
6075 				/* keep as-is */
6076 			} else {
6077 				/* walking unknown pointers yields old deprecated PTR_TO_BTF_ID */
6078 				clear_trusted_flags(&flag);
6079 			}
6080 		} else {
6081 			/*
6082 			 * If not in RCU CS or MEM_RCU pointer can be NULL then
6083 			 * aggressively mark as untrusted otherwise such
6084 			 * pointers will be plain PTR_TO_BTF_ID without flags
6085 			 * and will be allowed to be passed into helpers for
6086 			 * compat reasons.
6087 			 */
6088 			flag = PTR_UNTRUSTED;
6089 		}
6090 	} else {
6091 		/* Old compat. Deprecated */
6092 		clear_trusted_flags(&flag);
6093 	}
6094 
6095 	if (atype == BPF_READ && value_regno >= 0) {
6096 		ret = mark_btf_ld_reg(env, regs, value_regno, ret, reg->btf, btf_id, flag);
6097 		if (ret < 0)
6098 			return ret;
6099 	}
6100 
6101 	return 0;
6102 }
6103 
check_ptr_to_map_access(struct bpf_verifier_env * env,struct bpf_reg_state * regs,struct bpf_reg_state * reg,argno_t argno,int off,int size,enum bpf_access_type atype,int value_regno)6104 static int check_ptr_to_map_access(struct bpf_verifier_env *env,
6105 				   struct bpf_reg_state *regs, struct bpf_reg_state *reg,
6106 				   argno_t argno, int off, int size,
6107 				   enum bpf_access_type atype,
6108 				   int value_regno)
6109 {
6110 	struct bpf_map *map = reg->map_ptr;
6111 	struct bpf_reg_state map_reg;
6112 	enum bpf_type_flag flag = 0;
6113 	const struct btf_type *t;
6114 	const char *tname;
6115 	u32 btf_id;
6116 	int ret;
6117 
6118 	if (!btf_vmlinux) {
6119 		verbose(env, "map_ptr access not supported without CONFIG_DEBUG_INFO_BTF\n");
6120 		return -ENOTSUPP;
6121 	}
6122 
6123 	if (!map->ops->map_btf_id || !*map->ops->map_btf_id) {
6124 		verbose(env, "map_ptr access not supported for map type %d\n",
6125 			map->map_type);
6126 		return -ENOTSUPP;
6127 	}
6128 
6129 	t = btf_type_by_id(btf_vmlinux, *map->ops->map_btf_id);
6130 	tname = btf_name_by_offset(btf_vmlinux, t->name_off);
6131 
6132 	if (!env->allow_ptr_leaks) {
6133 		verbose(env,
6134 			"'struct %s' access is allowed only to CAP_PERFMON and CAP_SYS_ADMIN\n",
6135 			tname);
6136 		return -EPERM;
6137 	}
6138 
6139 	if (off < 0) {
6140 		verbose(env, "%s is %s invalid negative access: off=%d\n",
6141 			reg_arg_name(env, argno), tname, off);
6142 		return -EACCES;
6143 	}
6144 
6145 	if (atype != BPF_READ) {
6146 		verbose(env, "only read from %s is supported\n", tname);
6147 		return -EACCES;
6148 	}
6149 
6150 	/* Simulate access to a PTR_TO_BTF_ID */
6151 	memset(&map_reg, 0, sizeof(map_reg));
6152 	ret = mark_btf_ld_reg(env, &map_reg, 0, PTR_TO_BTF_ID,
6153 			      btf_vmlinux, *map->ops->map_btf_id, 0);
6154 	if (ret < 0)
6155 		return ret;
6156 	ret = btf_struct_access(&env->log, &map_reg, off, size, atype, &btf_id, &flag, NULL);
6157 	if (ret < 0)
6158 		return ret;
6159 
6160 	if (value_regno >= 0) {
6161 		ret = mark_btf_ld_reg(env, regs, value_regno, ret, btf_vmlinux, btf_id, flag);
6162 		if (ret < 0)
6163 			return ret;
6164 	}
6165 
6166 	return 0;
6167 }
6168 
6169 /* Check that the stack access at the given offset is within bounds. The
6170  * maximum valid offset is -1.
6171  *
6172  * The minimum valid offset is -MAX_BPF_STACK for writes, and
6173  * -state->allocated_stack for reads.
6174  */
check_stack_slot_within_bounds(struct bpf_verifier_env * env,s64 off,struct bpf_func_state * state,enum bpf_access_type t)6175 static int check_stack_slot_within_bounds(struct bpf_verifier_env *env,
6176                                           s64 off,
6177                                           struct bpf_func_state *state,
6178                                           enum bpf_access_type t)
6179 {
6180 	int min_valid_off;
6181 
6182 	if (t == BPF_WRITE || env->allow_uninit_stack)
6183 		min_valid_off = -MAX_BPF_STACK;
6184 	else
6185 		min_valid_off = -state->allocated_stack;
6186 
6187 	if (off < min_valid_off || off > -1)
6188 		return -EACCES;
6189 	return 0;
6190 }
6191 
6192 /* Check that the stack access at 'regno + off' falls within the maximum stack
6193  * bounds.
6194  *
6195  * 'off' includes `regno->offset`, but not its dynamic part (if any).
6196  */
check_stack_access_within_bounds(struct bpf_verifier_env * env,struct bpf_reg_state * reg,argno_t argno,int off,int access_size,enum bpf_access_type type)6197 static int check_stack_access_within_bounds(
6198 		struct bpf_verifier_env *env, struct bpf_reg_state *reg,
6199 		argno_t argno, int off, int access_size,
6200 		enum bpf_access_type type)
6201 {
6202 	struct bpf_func_state *state = bpf_func(env, reg);
6203 	s64 min_off, max_off;
6204 	int err;
6205 	char *err_extra;
6206 
6207 	if (type == BPF_READ)
6208 		err_extra = " read from";
6209 	else
6210 		err_extra = " write to";
6211 
6212 	if (tnum_is_const(reg->var_off)) {
6213 		min_off = (s64)reg->var_off.value + off;
6214 		max_off = min_off + access_size;
6215 	} else {
6216 		if (reg_smax(reg) >= BPF_MAX_VAR_OFF ||
6217 		    reg_smin(reg) <= -BPF_MAX_VAR_OFF) {
6218 			verbose(env, "invalid unbounded variable-offset%s stack %s\n",
6219 				err_extra, reg_arg_name(env, argno));
6220 			return -EACCES;
6221 		}
6222 		min_off = reg_smin(reg) + off;
6223 		max_off = reg_smax(reg) + off + access_size;
6224 	}
6225 
6226 	err = check_stack_slot_within_bounds(env, min_off, state, type);
6227 	if (!err && max_off > 0)
6228 		err = -EINVAL; /* out of stack access into non-negative offsets */
6229 	if (!err && access_size < 0)
6230 		/* access_size should not be negative (or overflow an int); others checks
6231 		 * along the way should have prevented such an access.
6232 		 */
6233 		err = -EFAULT; /* invalid negative access size; integer overflow? */
6234 
6235 	if (err) {
6236 		if (tnum_is_const(reg->var_off)) {
6237 			verbose(env, "invalid%s stack %s off=%lld size=%d\n",
6238 				err_extra, reg_arg_name(env, argno), min_off, access_size);
6239 		} else {
6240 			char tn_buf[48];
6241 
6242 			tnum_strn(tn_buf, sizeof(tn_buf), reg->var_off);
6243 			verbose(env, "invalid variable-offset%s stack %s var_off=%s off=%d size=%d\n",
6244 				err_extra, reg_arg_name(env, argno), tn_buf, off, access_size);
6245 		}
6246 		return err;
6247 	}
6248 
6249 	/* Note that there is no stack access with offset zero, so the needed stack
6250 	 * size is -min_off, not -min_off+1.
6251 	 */
6252 	return grow_stack_state(env, state, -min_off /* size */);
6253 }
6254 
get_func_retval_range(struct bpf_prog * prog,struct bpf_retval_range * range)6255 static bool get_func_retval_range(struct bpf_prog *prog,
6256 				  struct bpf_retval_range *range)
6257 {
6258 	if (prog->type == BPF_PROG_TYPE_LSM &&
6259 		prog->expected_attach_type == BPF_LSM_MAC &&
6260 		!bpf_lsm_get_retval_range(prog, range)) {
6261 		return true;
6262 	}
6263 	return false;
6264 }
6265 
add_scalar_to_reg(struct bpf_reg_state * dst_reg,s64 val)6266 static void add_scalar_to_reg(struct bpf_reg_state *dst_reg, s64 val)
6267 {
6268 	struct bpf_reg_state fake_reg;
6269 
6270 	if (!val)
6271 		return;
6272 
6273 	fake_reg.type = SCALAR_VALUE;
6274 	__mark_reg_known(&fake_reg, val);
6275 
6276 	scalar32_min_max_add(dst_reg, &fake_reg);
6277 	scalar_min_max_add(dst_reg, &fake_reg);
6278 	dst_reg->var_off = tnum_add(dst_reg->var_off, fake_reg.var_off);
6279 
6280 	reg_bounds_sync(dst_reg);
6281 }
6282 
check_map_mem_read(struct bpf_verifier_env * env,struct bpf_reg_state * reg,int off,int bpf_size,int value_regno,bool is_ldsx)6283 static int check_map_mem_read(struct bpf_verifier_env *env, struct bpf_reg_state *reg, int off,
6284 			      int bpf_size, int value_regno, bool is_ldsx)
6285 {
6286 	struct bpf_reg_state *regs = cur_regs(env);
6287 	int size = bpf_size_to_bytes(bpf_size);
6288 	struct bpf_map *map = reg->map_ptr;
6289 
6290 	switch (map->map_type) {
6291 	case BPF_MAP_TYPE_INSN_ARRAY:
6292 		if (bpf_size != BPF_DW) {
6293 			verbose(env, "Invalid read of %d bytes from insn_array\n", size);
6294 			return -EACCES;
6295 		}
6296 		regs[value_regno] = *reg;
6297 		add_scalar_to_reg(&regs[value_regno], off);
6298 		regs[value_regno].type = PTR_TO_INSN;
6299 		return 0;
6300 	case BPF_MAP_TYPE_PERCPU_ARRAY:
6301 		goto reg_unknown;
6302 	default:
6303 		break;
6304 	}
6305 
6306 	/* If map is read-only, track its contents as scalars. */
6307 	if (tnum_is_const(reg->var_off) &&
6308 	    bpf_map_is_rdonly(map) &&
6309 	    map->ops->map_direct_value_addr) {
6310 		int map_off = off + reg->var_off.value;
6311 		u64 val = 0;
6312 		int err;
6313 
6314 		err = bpf_map_direct_read(map, map_off, size, &val, is_ldsx);
6315 		if (err)
6316 			return err;
6317 
6318 		regs[value_regno].type = SCALAR_VALUE;
6319 		__mark_reg_known(&regs[value_regno], val);
6320 		return 0;
6321 	}
6322 
6323 reg_unknown:
6324 	mark_reg_unknown(env, regs, value_regno);
6325 	return 0;
6326 }
6327 
6328 /* check whether memory at (regno + off) is accessible for t = (read | write)
6329  * if t==write, value_regno is a register which value is stored into memory
6330  * if t==read, value_regno is a register which will receive the value from memory
6331  * if t==write && value_regno==-1, some unknown value is stored into memory
6332  * if t==read && value_regno==-1, don't care what we read from memory
6333  */
check_mem_access(struct bpf_verifier_env * env,int insn_idx,struct bpf_reg_state * reg,argno_t argno,int off,int bpf_size,enum bpf_access_type t,int value_regno,bool strict_alignment_once,bool is_ldsx)6334 static int check_mem_access(struct bpf_verifier_env *env, int insn_idx, struct bpf_reg_state *reg, argno_t argno,
6335 			    int off, int bpf_size, enum bpf_access_type t,
6336 			    int value_regno, bool strict_alignment_once, bool is_ldsx)
6337 {
6338 	struct bpf_reg_state *regs = cur_regs(env);
6339 	int size, err = 0;
6340 
6341 	size = bpf_size_to_bytes(bpf_size);
6342 	if (size < 0)
6343 		return size;
6344 
6345 	err = check_ptr_alignment(env, reg, off, size, strict_alignment_once);
6346 	if (err)
6347 		return err;
6348 
6349 	if (reg->type == PTR_TO_MAP_KEY) {
6350 		if (t == BPF_WRITE) {
6351 			verbose(env, "write to change key %s not allowed\n",
6352 				reg_arg_name(env, argno));
6353 			return -EACCES;
6354 		}
6355 
6356 		err = check_mem_region_access(env, reg, argno, off, size,
6357 					      reg->map_ptr->key_size, false);
6358 		if (err)
6359 			return err;
6360 		if (value_regno >= 0)
6361 			mark_reg_unknown(env, regs, value_regno);
6362 	} else if (reg->type == PTR_TO_MAP_VALUE) {
6363 		struct btf_field *kptr_field = NULL;
6364 
6365 		if (t == BPF_WRITE && value_regno >= 0 &&
6366 		    is_pointer_value(env, value_regno)) {
6367 			verbose(env, "R%d leaks addr into map\n", value_regno);
6368 			return -EACCES;
6369 		}
6370 		err = check_map_access_type(env, reg, off, size, t);
6371 		if (err)
6372 			return err;
6373 		err = check_map_access(env, reg, argno, off, size, false, ACCESS_DIRECT);
6374 		if (err)
6375 			return err;
6376 		if (tnum_is_const(reg->var_off))
6377 			kptr_field = btf_record_find(reg->map_ptr->record,
6378 						     off + reg->var_off.value, BPF_KPTR | BPF_UPTR);
6379 		if (kptr_field) {
6380 			err = check_map_kptr_access(env, value_regno, insn_idx, kptr_field);
6381 		} else if (t == BPF_READ && value_regno >= 0) {
6382 			err = check_map_mem_read(env, reg, off, bpf_size, value_regno, is_ldsx);
6383 		}
6384 	} else if (base_type(reg->type) == PTR_TO_MEM) {
6385 		bool rdonly_mem = type_is_rdonly_mem(reg->type);
6386 		bool rdonly_untrusted = rdonly_mem && (reg->type & PTR_UNTRUSTED);
6387 
6388 		if (type_may_be_null(reg->type)) {
6389 			verbose(env, "%s invalid mem access '%s'\n", reg_arg_name(env, argno),
6390 				reg_type_str(env, reg->type));
6391 			bpf_diag_invalid_deref(env, insn_idx, reg_from_argno(argno),
6392 					       reg_arg_name(env, argno), reg,
6393 						      BPF_DIAG_DEREF_NULLABLE_PTR, 0);
6394 			return -EACCES;
6395 		}
6396 
6397 		if (t == BPF_WRITE && rdonly_mem) {
6398 			verbose(env, "%s cannot write into %s\n",
6399 				reg_arg_name(env, argno), reg_type_str(env, reg->type));
6400 			return -EACCES;
6401 		}
6402 
6403 		if (t == BPF_WRITE && value_regno >= 0 &&
6404 		    is_pointer_value(env, value_regno)) {
6405 			verbose(env, "R%d leaks addr into mem\n", value_regno);
6406 			return -EACCES;
6407 		}
6408 
6409 		/*
6410 		 * Accesses to untrusted PTR_TO_MEM are done through probe
6411 		 * instructions, hence no need to check bounds in that case.
6412 		 */
6413 		if (!rdonly_untrusted)
6414 			err = check_mem_region_access(env, reg, argno, off, size,
6415 						      reg->mem_size, false);
6416 		if (!err && value_regno >= 0 && (t == BPF_READ || rdonly_mem))
6417 			mark_reg_unknown(env, regs, value_regno);
6418 	} else if (reg->type == PTR_TO_CTX) {
6419 		struct bpf_insn_access_aux info = {
6420 			.reg_type = SCALAR_VALUE,
6421 			.is_ldsx = is_ldsx,
6422 			.log = &env->log,
6423 		};
6424 		struct bpf_retval_range range;
6425 
6426 		if (t == BPF_WRITE && value_regno >= 0 &&
6427 		    is_pointer_value(env, value_regno)) {
6428 			verbose(env, "R%d leaks addr into ctx\n", value_regno);
6429 			return -EACCES;
6430 		}
6431 
6432 		err = check_ctx_access(env, insn_idx, reg, argno, off, size, t, &info);
6433 		if (!err && t == BPF_READ && value_regno >= 0) {
6434 			/* ctx access returns either a scalar, or a
6435 			 * PTR_TO_PACKET[_META,_END]. In the latter
6436 			 * case, we know the offset is zero.
6437 			 */
6438 			if (info.reg_type == SCALAR_VALUE) {
6439 				if (info.is_retval && get_func_retval_range(env->prog, &range)) {
6440 					mark_reg_unknown(env, regs, value_regno);
6441 					err = __mark_reg_s32_range(env, regs, value_regno,
6442 								   range.minval, range.maxval);
6443 					if (err)
6444 						return err;
6445 				} else {
6446 					mark_reg_unknown(env, regs, value_regno);
6447 				}
6448 			} else {
6449 				mark_reg_known_zero(env, regs,
6450 						    value_regno);
6451 				if (base_type(info.reg_type) == PTR_TO_BTF_ID) {
6452 					regs[value_regno].btf = info.btf;
6453 					regs[value_regno].btf_id = info.btf_id;
6454 					regs[value_regno].id = info.ref_id;
6455 				}
6456 				if (type_may_be_null(info.reg_type) && !regs[value_regno].id)
6457 					regs[value_regno].id = ++env->id_gen;
6458 			}
6459 			regs[value_regno].type = info.reg_type;
6460 		}
6461 
6462 	} else if (reg->type == PTR_TO_STACK) {
6463 		/* Basic bounds checks. */
6464 		err = check_stack_access_within_bounds(env, reg, argno, off, size, t);
6465 		if (err)
6466 			return err;
6467 
6468 		if (t == BPF_READ)
6469 			err = check_stack_read(env, reg, argno, off, size,
6470 					       value_regno);
6471 		else
6472 			err = check_stack_write(env, reg, off, size,
6473 						value_regno, insn_idx);
6474 	} else if (reg_is_pkt_pointer(reg)) {
6475 		if (t == BPF_WRITE && !may_access_direct_pkt_data(env, NULL, t)) {
6476 			verbose(env, "cannot write into packet\n");
6477 			return -EACCES;
6478 		}
6479 		if (t == BPF_WRITE && value_regno >= 0 &&
6480 		    is_pointer_value(env, value_regno)) {
6481 			verbose(env, "R%d leaks addr into packet\n",
6482 				value_regno);
6483 			return -EACCES;
6484 		}
6485 		err = check_packet_access(env, reg, argno, off, size, false);
6486 		if (!err && t == BPF_READ && value_regno >= 0)
6487 			mark_reg_unknown(env, regs, value_regno);
6488 	} else if (reg->type == PTR_TO_FLOW_KEYS) {
6489 		if (t == BPF_WRITE && value_regno >= 0 &&
6490 		    is_pointer_value(env, value_regno)) {
6491 			verbose(env, "R%d leaks addr into flow keys\n",
6492 				value_regno);
6493 			return -EACCES;
6494 		}
6495 
6496 		err = check_flow_keys_access(env, reg, argno, off, size);
6497 		if (!err && t == BPF_READ && value_regno >= 0)
6498 			mark_reg_unknown(env, regs, value_regno);
6499 	} else if (type_is_sk_pointer(reg->type)) {
6500 		if (t == BPF_WRITE) {
6501 			verbose(env, "%s cannot write into %s\n",
6502 				reg_arg_name(env, argno), reg_type_str(env, reg->type));
6503 			return -EACCES;
6504 		}
6505 		err = check_sock_access(env, insn_idx, reg, argno, off, size, t);
6506 		if (!err && value_regno >= 0)
6507 			mark_reg_unknown(env, regs, value_regno);
6508 	} else if (reg->type == PTR_TO_TP_BUFFER) {
6509 		err = check_tp_buffer_access(env, reg, argno, off, size);
6510 		if (!err && t == BPF_READ && value_regno >= 0)
6511 			mark_reg_unknown(env, regs, value_regno);
6512 	} else if (base_type(reg->type) == PTR_TO_BTF_ID &&
6513 		   !type_may_be_null(reg->type)) {
6514 		err = check_ptr_to_btf_access(env, regs, reg, argno, off, size, t,
6515 					      value_regno);
6516 	} else if (reg->type == CONST_PTR_TO_MAP) {
6517 		err = check_ptr_to_map_access(env, regs, reg, argno, off, size, t,
6518 					      value_regno);
6519 	} else if (base_type(reg->type) == PTR_TO_BUF &&
6520 		   !type_may_be_null(reg->type)) {
6521 		bool rdonly_mem = type_is_rdonly_mem(reg->type);
6522 		u32 *max_access;
6523 
6524 		if (rdonly_mem) {
6525 			if (t == BPF_WRITE) {
6526 				verbose(env, "%s cannot write into %s\n",
6527 					reg_arg_name(env, argno), reg_type_str(env, reg->type));
6528 				return -EACCES;
6529 			}
6530 			max_access = &env->prog->aux->max_rdonly_access;
6531 		} else {
6532 			max_access = &env->prog->aux->max_rdwr_access;
6533 		}
6534 
6535 		err = check_buffer_access(env, reg, argno, off, size, false,
6536 					  max_access);
6537 
6538 		if (!err && value_regno >= 0 && (rdonly_mem || t == BPF_READ))
6539 			mark_reg_unknown(env, regs, value_regno);
6540 	} else if (reg->type == PTR_TO_ARENA) {
6541 		if (t == BPF_READ && value_regno >= 0)
6542 			mark_reg_unknown(env, regs, value_regno);
6543 	} else {
6544 		enum bpf_diag_invalid_deref_kind kind = BPF_DIAG_DEREF_INVALID_PTR;
6545 
6546 		verbose(env, "%s invalid mem access '%s'\n", reg_arg_name(env, argno),
6547 			reg_type_str(env, reg->type));
6548 		if (reg->type == SCALAR_VALUE)
6549 			kind = BPF_DIAG_DEREF_SCALAR;
6550 		else if (type_may_be_null(reg->type))
6551 			kind = BPF_DIAG_DEREF_NULLABLE_PTR;
6552 		bpf_diag_invalid_deref(env, insn_idx, reg_from_argno(argno),
6553 				       reg_arg_name(env, argno), reg, kind, 0);
6554 		return -EACCES;
6555 	}
6556 
6557 	if (!err && size < BPF_REG_SIZE && value_regno >= 0 && t == BPF_READ &&
6558 	    regs[value_regno].type == SCALAR_VALUE) {
6559 		if (!is_ldsx) {
6560 			/* b/h/w load zero-extends, mark upper bits as known 0 */
6561 			coerce_reg_to_size(&regs[value_regno], size);
6562 		} else {
6563 			/*
6564 			 * Sign-extension can change the register value relative
6565 			 * to a scalar it is linked with by id (e.g. a zero-
6566 			 * extending fill of the same spilled stack slot), thus
6567 			 * drop the shared id in that case.
6568 			 */
6569 			bool no_sext = reg_umax(&regs[value_regno]) <
6570 					(1ULL << (size * BITS_PER_BYTE - 1));
6571 
6572 			coerce_reg_to_size_sx(&regs[value_regno], size);
6573 			if (!no_sext)
6574 				clear_scalar_id(&regs[value_regno]);
6575 		}
6576 	}
6577 	return err;
6578 }
6579 
6580 static int save_aux_ptr_type(struct bpf_verifier_env *env, enum bpf_reg_type type,
6581 			     bool allow_trust_mismatch);
6582 
check_load_mem(struct bpf_verifier_env * env,struct bpf_insn * insn,bool strict_alignment_once,bool is_ldsx,bool allow_trust_mismatch,const char * ctx)6583 static int check_load_mem(struct bpf_verifier_env *env, struct bpf_insn *insn,
6584 			  bool strict_alignment_once, bool is_ldsx,
6585 			  bool allow_trust_mismatch, const char *ctx)
6586 {
6587 	struct bpf_verifier_state *vstate = env->cur_state;
6588 	struct bpf_func_state *state = vstate->frame[vstate->curframe];
6589 	struct bpf_reg_state *regs = cur_regs(env);
6590 	enum bpf_reg_type src_reg_type;
6591 	int err;
6592 
6593 	/* Handle stack arg read */
6594 	if (is_stack_arg_ldx(insn)) {
6595 		err = check_reg_arg(env, insn->dst_reg, DST_OP_NO_MARK);
6596 		if (err)
6597 			return err;
6598 		return check_stack_arg_read(env, state, insn->off, insn->dst_reg);
6599 	}
6600 
6601 	/* check src operand */
6602 	err = check_reg_arg(env, insn->src_reg, SRC_OP);
6603 	if (err)
6604 		return err;
6605 
6606 	/* check dst operand */
6607 	err = check_reg_arg(env, insn->dst_reg, DST_OP_NO_MARK);
6608 	if (err)
6609 		return err;
6610 
6611 	src_reg_type = regs[insn->src_reg].type;
6612 
6613 	/*
6614 	 * check_stack_read_fixed_off() may refine the modification's origin to
6615 	 * the source stack slot.
6616 	 */
6617 	bpf_diag_mod_begin(env, &regs[insn->dst_reg], NULL, BPF_DIAG_MOD_WRITE);
6618 	err = check_mem_access(env, env->insn_idx, regs + insn->src_reg, argno_from_reg(insn->src_reg), insn->off,
6619 			       BPF_SIZE(insn->code), BPF_READ, insn->dst_reg,
6620 			       strict_alignment_once, is_ldsx);
6621 	err = err ?: save_aux_ptr_type(env, src_reg_type,
6622 				       allow_trust_mismatch);
6623 	err = err ?: reg_bounds_sanity_check(env, &regs[insn->dst_reg], ctx);
6624 	if (!err)
6625 		bpf_diag_mod_end(env);
6626 
6627 	return err;
6628 }
6629 
check_store_reg(struct bpf_verifier_env * env,struct bpf_insn * insn,bool strict_alignment_once)6630 static int check_store_reg(struct bpf_verifier_env *env, struct bpf_insn *insn,
6631 			   bool strict_alignment_once)
6632 {
6633 	struct bpf_verifier_state *vstate = env->cur_state;
6634 	struct bpf_func_state *state = vstate->frame[vstate->curframe];
6635 	struct bpf_reg_state *regs = cur_regs(env);
6636 	enum bpf_reg_type dst_reg_type;
6637 	int err;
6638 
6639 	/* Handle stack arg write */
6640 	if (is_stack_arg_stx(insn)) {
6641 		err = check_reg_arg(env, insn->src_reg, SRC_OP);
6642 		if (err)
6643 			return err;
6644 		return check_stack_arg_write(env, state, insn->off, regs + insn->src_reg);
6645 	}
6646 
6647 	/* check src1 operand */
6648 	err = check_reg_arg(env, insn->src_reg, SRC_OP);
6649 	if (err)
6650 		return err;
6651 
6652 	/* check src2 operand */
6653 	err = check_reg_arg(env, insn->dst_reg, SRC_OP);
6654 	if (err)
6655 		return err;
6656 
6657 	dst_reg_type = regs[insn->dst_reg].type;
6658 
6659 	/* Check if (dst_reg + off) is writeable. */
6660 	err = check_mem_access(env, env->insn_idx, regs + insn->dst_reg, argno_from_reg(insn->dst_reg), insn->off,
6661 			       BPF_SIZE(insn->code), BPF_WRITE, insn->src_reg,
6662 			       strict_alignment_once, false);
6663 	err = err ?: save_aux_ptr_type(env, dst_reg_type, false);
6664 
6665 	return err;
6666 }
6667 
check_atomic_rmw(struct bpf_verifier_env * env,struct bpf_insn * insn)6668 static int check_atomic_rmw(struct bpf_verifier_env *env,
6669 			    struct bpf_insn *insn)
6670 {
6671 	struct bpf_reg_state *dst_reg;
6672 	int load_reg;
6673 	int err;
6674 
6675 	if (BPF_SIZE(insn->code) != BPF_W && BPF_SIZE(insn->code) != BPF_DW) {
6676 		verbose(env, "invalid atomic operand size\n");
6677 		return -EINVAL;
6678 	}
6679 
6680 	/* check src1 operand */
6681 	err = check_reg_arg(env, insn->src_reg, SRC_OP);
6682 	if (err)
6683 		return err;
6684 
6685 	/* check src2 operand */
6686 	err = check_reg_arg(env, insn->dst_reg, SRC_OP);
6687 	if (err)
6688 		return err;
6689 
6690 	if (insn->imm == BPF_CMPXCHG) {
6691 		/* Check comparison of R0 with memory location */
6692 		const u32 aux_reg = BPF_REG_0;
6693 
6694 		err = check_reg_arg(env, aux_reg, SRC_OP);
6695 		if (err)
6696 			return err;
6697 
6698 		if (is_pointer_value(env, aux_reg)) {
6699 			verbose(env, "R%d leaks addr into mem\n", aux_reg);
6700 			return -EACCES;
6701 		}
6702 	}
6703 
6704 	if (is_pointer_value(env, insn->src_reg)) {
6705 		verbose(env, "R%d leaks addr into mem\n", insn->src_reg);
6706 		return -EACCES;
6707 	}
6708 
6709 	if (!atomic_ptr_type_ok(env, insn->dst_reg, insn)) {
6710 		verbose(env, "BPF_ATOMIC stores into R%d %s is not allowed\n",
6711 			insn->dst_reg,
6712 			reg_type_str(env, reg_state(env, insn->dst_reg)->type));
6713 		return -EACCES;
6714 	}
6715 
6716 	load_reg = bpf_atomic_load_reg(insn);
6717 	if (load_reg >= 0) {
6718 		/* check and record load of old value */
6719 		err = check_reg_arg(env, load_reg, DST_OP);
6720 		if (err)
6721 			return err;
6722 	}
6723 
6724 	dst_reg = cur_regs(env) + insn->dst_reg;
6725 
6726 	/* Check whether we can read the memory, with second call for fetch
6727 	 * case to simulate the register fill.
6728 	 */
6729 	err = check_mem_access(env, env->insn_idx, dst_reg, argno_from_reg(insn->dst_reg), insn->off,
6730 			       BPF_SIZE(insn->code), BPF_READ, -1, true, false);
6731 	if (!err && load_reg >= 0) {
6732 		bpf_diag_mod_begin(env, cur_regs(env) + load_reg, NULL, BPF_DIAG_MOD_WRITE);
6733 		err = check_mem_access(env, env->insn_idx, dst_reg, argno_from_reg(insn->dst_reg),
6734 				       insn->off, BPF_SIZE(insn->code),
6735 				       BPF_READ, load_reg, true, false);
6736 		if (!err)
6737 			bpf_diag_mod_end(env);
6738 	}
6739 	if (err)
6740 		return err;
6741 
6742 	err = save_aux_ptr_type(env, dst_reg->type, false);
6743 	if (err)
6744 		return err;
6745 	/* Check whether we can write into the same memory. */
6746 	err = check_mem_access(env, env->insn_idx, dst_reg, argno_from_reg(insn->dst_reg), insn->off,
6747 			       BPF_SIZE(insn->code), BPF_WRITE, -1, true, false);
6748 	if (err)
6749 		return err;
6750 	return 0;
6751 }
6752 
check_atomic_load(struct bpf_verifier_env * env,struct bpf_insn * insn)6753 static int check_atomic_load(struct bpf_verifier_env *env,
6754 			     struct bpf_insn *insn)
6755 {
6756 	int err;
6757 
6758 	err = check_reg_arg(env, insn->src_reg, SRC_OP);
6759 	if (err)
6760 		return err;
6761 
6762 	if (!atomic_ptr_type_ok(env, insn->src_reg, insn)) {
6763 		verbose(env, "BPF_ATOMIC loads from R%d %s is not allowed\n",
6764 			insn->src_reg,
6765 			reg_type_str(env, reg_state(env, insn->src_reg)->type));
6766 		return -EACCES;
6767 	}
6768 
6769 	return check_load_mem(env, insn, true, false, false, "atomic_load");
6770 }
6771 
check_atomic_store(struct bpf_verifier_env * env,struct bpf_insn * insn)6772 static int check_atomic_store(struct bpf_verifier_env *env,
6773 			      struct bpf_insn *insn)
6774 {
6775 	int err;
6776 
6777 	err = check_store_reg(env, insn, true);
6778 	if (err)
6779 		return err;
6780 
6781 	if (!atomic_ptr_type_ok(env, insn->dst_reg, insn)) {
6782 		verbose(env, "BPF_ATOMIC stores into R%d %s is not allowed\n",
6783 			insn->dst_reg,
6784 			reg_type_str(env, reg_state(env, insn->dst_reg)->type));
6785 		return -EACCES;
6786 	}
6787 
6788 	return 0;
6789 }
6790 
check_atomic(struct bpf_verifier_env * env,struct bpf_insn * insn)6791 static int check_atomic(struct bpf_verifier_env *env, struct bpf_insn *insn)
6792 {
6793 	switch (insn->imm) {
6794 	case BPF_ADD:
6795 	case BPF_ADD | BPF_FETCH:
6796 	case BPF_AND:
6797 	case BPF_AND | BPF_FETCH:
6798 	case BPF_OR:
6799 	case BPF_OR | BPF_FETCH:
6800 	case BPF_XOR:
6801 	case BPF_XOR | BPF_FETCH:
6802 	case BPF_XCHG:
6803 	case BPF_CMPXCHG:
6804 		return check_atomic_rmw(env, insn);
6805 	case BPF_LOAD_ACQ:
6806 		if (BPF_SIZE(insn->code) == BPF_DW && BITS_PER_LONG != 64) {
6807 			verbose(env,
6808 				"64-bit load-acquires are only supported on 64-bit arches\n");
6809 			return -EOPNOTSUPP;
6810 		}
6811 		return check_atomic_load(env, insn);
6812 	case BPF_STORE_REL:
6813 		if (BPF_SIZE(insn->code) == BPF_DW && BITS_PER_LONG != 64) {
6814 			verbose(env,
6815 				"64-bit store-releases are only supported on 64-bit arches\n");
6816 			return -EOPNOTSUPP;
6817 		}
6818 		return check_atomic_store(env, insn);
6819 	default:
6820 		verbose(env, "BPF_ATOMIC uses invalid atomic opcode %02x\n",
6821 			insn->imm);
6822 		return -EINVAL;
6823 	}
6824 }
6825 
6826 /* When register 'regno' is used to read the stack (either directly or through
6827  * a helper function) make sure that it's within stack boundary and, depending
6828  * on the access type and privileges, that all elements of the stack are
6829  * initialized.
6830  *
6831  * All registers that have been spilled on the stack in the slots within the
6832  * read offsets are marked as read.
6833  */
check_stack_range_initialized(struct bpf_verifier_env * env,struct bpf_reg_state * reg,argno_t argno,int off,int access_size,bool zero_size_allowed,enum bpf_access_type type,struct bpf_call_arg_meta * meta)6834 static int check_stack_range_initialized(
6835 		struct bpf_verifier_env *env, struct bpf_reg_state *reg, argno_t argno, int off,
6836 		int access_size, bool zero_size_allowed,
6837 		enum bpf_access_type type, struct bpf_call_arg_meta *meta)
6838 {
6839 	struct bpf_func_state *state = bpf_func(env, reg);
6840 	int err, min_off, max_off, i, j, slot, spi;
6841 	/* Some accesses can write anything into the stack, others are
6842 	 * read-only.
6843 	 */
6844 	bool clobber = type == BPF_WRITE;
6845 	/*
6846 	 * Negative access_size signals global subprog arg check where
6847 	 * STACK_POISON slots are acceptable. static stack liveness
6848 	 * might have determined that subprog doesn't read them,
6849 	 * but BTF based global subprog validation isn't accurate enough.
6850 	 */
6851 	bool allow_poison = access_size < 0 || clobber;
6852 	/* The call will initialize the memory; uninitialized stack allowed */
6853 	bool raw_mode = meta && meta->arg_raw_mem.regno == reg_from_argno(argno);
6854 
6855 	access_size = abs(access_size);
6856 
6857 	if (access_size == 0 && !zero_size_allowed) {
6858 		verbose(env, "invalid zero-sized read\n");
6859 		return -EACCES;
6860 	}
6861 
6862 	err = check_stack_access_within_bounds(env, reg, argno, off, access_size, type);
6863 	if (err)
6864 		return err;
6865 
6866 	if (tnum_is_const(reg->var_off)) {
6867 		min_off = max_off = reg->var_off.value + off;
6868 	} else {
6869 		/* Variable offset is prohibited for unprivileged mode for
6870 		 * simplicity since it requires corresponding support in
6871 		 * Spectre masking for stack ALU.
6872 		 * See also retrieve_ptr_limit().
6873 		 */
6874 		if (!env->bypass_spec_v1) {
6875 			char tn_buf[48];
6876 
6877 			tnum_strn(tn_buf, sizeof(tn_buf), reg->var_off);
6878 			verbose(env, "%s variable offset stack access prohibited for !root, var_off=%s\n",
6879 				reg_arg_name(env, argno), tn_buf);
6880 			return -EACCES;
6881 		}
6882 		/* Only initialized buffer on stack is allowed to be accessed
6883 		 * with variable offset. With uninitialized buffer it's hard to
6884 		 * guarantee that whole memory is marked as initialized on
6885 		 * helper return since specific bounds are unknown what may
6886 		 * cause uninitialized stack leaking.
6887 		 */
6888 		raw_mode = false;
6889 
6890 		min_off = reg_smin(reg) + off;
6891 		max_off = reg_smax(reg) + off;
6892 	}
6893 
6894 	if (raw_mode) {
6895 		meta->arg_raw_mem.size = access_size;
6896 		return 0;
6897 	}
6898 
6899 	for (i = min_off; i < max_off + access_size; i++) {
6900 		u8 *stype;
6901 
6902 		slot = -i - 1;
6903 		spi = slot / BPF_REG_SIZE;
6904 		if (state->allocated_stack <= slot) {
6905 			verbose(env, "allocated_stack too small\n");
6906 			return -EFAULT;
6907 		}
6908 
6909 		stype = &state->stack[spi].slot_type[slot % BPF_REG_SIZE];
6910 		if (*stype == STACK_MISC)
6911 			goto mark;
6912 		if ((*stype == STACK_ZERO) ||
6913 		    (*stype == STACK_INVALID && env->allow_uninit_stack)) {
6914 			if (clobber) {
6915 				/* helper can write anything into the stack */
6916 				*stype = STACK_MISC;
6917 			}
6918 			goto mark;
6919 		}
6920 
6921 		if (bpf_is_spilled_reg(&state->stack[spi]) &&
6922 		    (state->stack[spi].spilled_ptr.type == SCALAR_VALUE ||
6923 		     env->allow_ptr_leaks)) {
6924 			if (clobber) {
6925 				__mark_reg_unknown(env, &state->stack[spi].spilled_ptr);
6926 				for (j = 0; j < BPF_REG_SIZE; j++)
6927 					scrub_spilled_slot(&state->stack[spi].slot_type[j]);
6928 			}
6929 			goto mark;
6930 		}
6931 
6932 		if (*stype == STACK_POISON) {
6933 			if (allow_poison)
6934 				goto mark;
6935 			verbose(env, "reading from stack %s off %d+%d size %d, slot poisoned by dead code elimination\n",
6936 				reg_arg_name(env, argno), min_off, i - min_off, access_size);
6937 		} else if (tnum_is_const(reg->var_off)) {
6938 			verbose(env, "invalid read from stack %s off %d+%d size %d\n",
6939 				reg_arg_name(env, argno), min_off, i - min_off, access_size);
6940 		} else {
6941 			char tn_buf[48];
6942 
6943 			tnum_strn(tn_buf, sizeof(tn_buf), reg->var_off);
6944 			verbose(env, "invalid read from stack %s var_off %s+%d size %d\n",
6945 				reg_arg_name(env, argno), tn_buf, i - min_off, access_size);
6946 		}
6947 		return -EACCES;
6948 mark:
6949 		;
6950 	}
6951 	return 0;
6952 }
6953 
check_helper_mem_access(struct bpf_verifier_env * env,struct bpf_reg_state * reg,argno_t argno,int access_size,enum bpf_access_type access_type,bool zero_size_allowed,struct bpf_call_arg_meta * meta,bool * known_memory)6954 static int check_helper_mem_access(struct bpf_verifier_env *env, struct bpf_reg_state *reg,
6955 				   argno_t argno, int access_size,
6956 				   enum bpf_access_type access_type, bool zero_size_allowed,
6957 				   struct bpf_call_arg_meta *meta, bool *known_memory)
6958 {
6959 	struct bpf_reg_state *regs = cur_regs(env);
6960 	u32 *max_access;
6961 
6962 	if (known_memory)
6963 		*known_memory = true;
6964 
6965 	switch (base_type(reg->type)) {
6966 	case PTR_TO_PACKET:
6967 	case PTR_TO_PACKET_META:
6968 		return check_packet_access(env, reg, argno, 0, access_size,
6969 					   zero_size_allowed);
6970 	case PTR_TO_MAP_KEY:
6971 		if (access_type == BPF_WRITE) {
6972 			verbose(env, "%s cannot write into %s\n",
6973 				reg_arg_name(env, argno), reg_type_str(env, reg->type));
6974 			return -EACCES;
6975 		}
6976 		return check_mem_region_access(env, reg, argno, 0, access_size,
6977 					       reg->map_ptr->key_size, false);
6978 	case PTR_TO_MAP_VALUE:
6979 		if (check_map_access_type(env, reg, 0, access_size, access_type))
6980 			return -EACCES;
6981 		return check_map_access(env, reg, argno, 0, access_size,
6982 					zero_size_allowed, ACCESS_HELPER);
6983 	case PTR_TO_MEM:
6984 		if (type_is_rdonly_mem(reg->type)) {
6985 			if (access_type == BPF_WRITE) {
6986 				verbose(env, "%s cannot write into %s\n",
6987 					reg_arg_name(env, argno), reg_type_str(env, reg->type));
6988 				return -EACCES;
6989 			}
6990 		}
6991 		return check_mem_region_access(env, reg, argno, 0,
6992 					       access_size, reg->mem_size,
6993 					       zero_size_allowed);
6994 	case PTR_TO_BUF:
6995 		if (type_is_rdonly_mem(reg->type)) {
6996 			if (access_type == BPF_WRITE) {
6997 				verbose(env, "%s cannot write into %s\n",
6998 					reg_arg_name(env, argno), reg_type_str(env, reg->type));
6999 				return -EACCES;
7000 			}
7001 
7002 			max_access = &env->prog->aux->max_rdonly_access;
7003 		} else {
7004 			max_access = &env->prog->aux->max_rdwr_access;
7005 		}
7006 		return check_buffer_access(env, reg, argno, 0,
7007 					   access_size, zero_size_allowed,
7008 					   max_access);
7009 	case PTR_TO_STACK:
7010 		return check_stack_range_initialized(
7011 				env, reg,
7012 				argno, 0, access_size,
7013 				zero_size_allowed, access_type, meta);
7014 	case PTR_TO_BTF_ID:
7015 		return check_ptr_to_btf_access(env, regs, reg, argno, 0,
7016 					       access_size, access_type, -1);
7017 	case PTR_TO_CTX:
7018 		/* Only permit reading or writing syscall context using helper calls. */
7019 		if (is_var_ctx_off_allowed(env->prog)) {
7020 			int err = check_mem_region_access(env, reg, argno, 0, access_size, U16_MAX,
7021 							  zero_size_allowed);
7022 			if (err)
7023 				return err;
7024 			if (env->prog->aux->max_ctx_offset < reg_umax(reg) + access_size)
7025 				env->prog->aux->max_ctx_offset = reg_umax(reg) + access_size;
7026 			return 0;
7027 		}
7028 		fallthrough;
7029 	default: /* scalar_value or invalid ptr */
7030 		/* Allow zero-byte read from NULL, regardless of pointer type */
7031 		if (zero_size_allowed && access_size == 0 &&
7032 		    bpf_register_is_null(reg))
7033 			return 0;
7034 		if (known_memory && base_type(reg->type) != PTR_TO_CTX)
7035 			*known_memory = false;
7036 
7037 		verbose(env, "%s type=%s ", reg_arg_name(env, argno),
7038 			reg_type_str(env, reg->type));
7039 		verbose(env, "expected=%s\n", reg_type_str(env, PTR_TO_STACK));
7040 		return -EACCES;
7041 	}
7042 }
7043 
7044 enum bpf_mem_size_failure {
7045 	BPF_MEM_SIZE_FAIL_NONE,
7046 	BPF_MEM_SIZE_FAIL_MEMORY,
7047 	BPF_MEM_SIZE_FAIL_SIZE,
7048 };
7049 
7050 /* verify arguments to helpers or kfuncs consisting of a pointer and an access
7051  * size.
7052  *
7053  * @mem_reg contains the pointer, @size_reg contains the access size.
7054  */
check_mem_size_reg(struct bpf_verifier_env * env,struct bpf_reg_state * mem_reg,struct bpf_reg_state * size_reg,argno_t mem_argno,argno_t size_argno,u32 access_type,bool zero_size_allowed,struct bpf_call_arg_meta * meta,enum bpf_mem_size_failure * failure)7055 static int check_mem_size_reg(struct bpf_verifier_env *env,
7056 			      struct bpf_reg_state *mem_reg,
7057 			      struct bpf_reg_state *size_reg, argno_t mem_argno,
7058 			      argno_t size_argno, u32 access_type,
7059 			      bool zero_size_allowed,
7060 			      struct bpf_call_arg_meta *meta,
7061 			      enum bpf_mem_size_failure *failure)
7062 {
7063 	int err = 0;
7064 
7065 	if (failure)
7066 		*failure = BPF_MEM_SIZE_FAIL_NONE;
7067 
7068 	/* This is used to refine r0 return value bounds for helpers
7069 	 * that enforce this value as an upper bound on return values.
7070 	 * See do_refine_retval_range() for helpers that can refine
7071 	 * the return value. C type of helper is u32 so we pull register
7072 	 * bound from umax_value however, if negative verifier errors
7073 	 * out. Only upper bounds can be learned because retval is an
7074 	 * int type and negative retvals are allowed.
7075 	 */
7076 	meta->msize_max_value = reg_umax(size_reg);
7077 
7078 	/* The register is SCALAR_VALUE; the access check happens using
7079 	 * its boundaries. For unprivileged variable accesses, disable
7080 	 * raw mode so that the program is required to initialize all
7081 	 * the memory that the helper could just partially fill up.
7082 	 */
7083 	if (!tnum_is_const(size_reg->var_off))
7084 		meta = NULL;
7085 
7086 	if (reg_smin(size_reg) < 0) {
7087 		verbose(env, "%s min value is negative, either use unsigned or 'var &= const'\n",
7088 			reg_arg_name(env, size_argno));
7089 		err = -EACCES;
7090 		goto size_error;
7091 	}
7092 
7093 	if (reg_umin(size_reg) == 0 && !zero_size_allowed) {
7094 		verbose(env, "%s invalid zero-sized read: u64=[%lld,%lld]\n",
7095 			reg_arg_name(env, size_argno), reg_umin(size_reg), reg_umax(size_reg));
7096 		err = -EACCES;
7097 		goto size_error;
7098 	}
7099 
7100 	if (reg_umax(size_reg) >= BPF_MAX_VAR_SIZ) {
7101 		verbose(env, "%s unbounded memory access, use 'var &= const' or 'if (var < const)'\n",
7102 			reg_arg_name(env, size_argno));
7103 		err = -EACCES;
7104 		goto size_error;
7105 	}
7106 
7107 	if (access_type & BPF_READ)
7108 		err = check_helper_mem_access(env, mem_reg, mem_argno, reg_umax(size_reg),
7109 					      BPF_READ, zero_size_allowed, meta, NULL);
7110 	if (!err && access_type & BPF_WRITE)
7111 		err = check_helper_mem_access(env, mem_reg, mem_argno, reg_umax(size_reg),
7112 					      BPF_WRITE, zero_size_allowed, meta, NULL);
7113 	if (err && failure)
7114 		*failure = BPF_MEM_SIZE_FAIL_MEMORY;
7115 
7116 	if (!err) {
7117 		int regno = reg_from_argno(size_argno);
7118 
7119 		if (regno >= 0)
7120 			err = mark_chain_precision(env, regno);
7121 		else
7122 			err = mark_stack_arg_precision(env, arg_idx_from_argno(size_argno));
7123 	}
7124 
7125 	return err;
7126 
7127 size_error:
7128 	if (failure)
7129 		*failure = BPF_MEM_SIZE_FAIL_SIZE;
7130 	return err;
7131 }
7132 
check_mem_reg(struct bpf_verifier_env * env,struct bpf_reg_state * reg,argno_t argno,u32 mem_size,enum bpf_access_type access_type,struct bpf_call_arg_meta * meta,bool * known_memory)7133 static int check_mem_reg(struct bpf_verifier_env *env, struct bpf_reg_state *reg,
7134 			 argno_t argno, u32 mem_size, enum bpf_access_type access_type,
7135 			 struct bpf_call_arg_meta *meta, bool *known_memory)
7136 {
7137 	int size, err = 0;
7138 
7139 	if (bpf_register_is_null(reg))
7140 		return 0;
7141 	if (known_memory)
7142 		*known_memory = true;
7143 
7144 	if (mem_size > S32_MAX) {
7145 		verbose(env, "%s memory size %u is too large\n",
7146 			reg_arg_name(env, argno), mem_size);
7147 		return -EACCES;
7148 	}
7149 
7150 	/*
7151 	 * Only a global subprog (meta == NULL) may read poisoned stack slots:
7152 	 * its static stack liveness proved the callee body skips them.
7153 	 */
7154 	size = (!meta && base_type(reg->type) == PTR_TO_STACK) ? -(int)mem_size : mem_size;
7155 
7156 	if (access_type & BPF_READ)
7157 		err = check_helper_mem_access(env, reg, argno, size, BPF_READ, true, meta,
7158 					      known_memory);
7159 	if (!err && (access_type & BPF_WRITE))
7160 		err = check_helper_mem_access(env, reg, argno, size, BPF_WRITE, true, meta,
7161 					      known_memory);
7162 
7163 	return err;
7164 }
7165 
process_const_alloc_mem_size(struct bpf_verifier_env * env,struct bpf_reg_state * reg,argno_t argno,struct ret_mem_desc * ret_mem)7166 static int process_const_alloc_mem_size(struct bpf_verifier_env *env, struct bpf_reg_state *reg,
7167 					argno_t argno, struct ret_mem_desc *ret_mem)
7168 {
7169 	int regno = reg_from_argno(argno);
7170 	int err;
7171 
7172 	if (ret_mem->found) {
7173 		verifier_bug(env, "only one allocation size argument permitted");
7174 		return -EFAULT;
7175 	}
7176 
7177 	if (!tnum_is_const(reg->var_off)) {
7178 		verbose(env, "%s is not a const\n", reg_arg_name(env, argno));
7179 		return -EINVAL;
7180 	}
7181 
7182 	if (reg->var_off.value > U32_MAX) {
7183 		verbose(env, "%s allocation size exceeds u32 max\n", reg_arg_name(env, argno));
7184 		return -EINVAL;
7185 	}
7186 
7187 	if (regno >= 0)
7188 		err = mark_chain_precision(env, regno);
7189 	else
7190 		err = mark_stack_arg_precision(env, arg_idx_from_argno(argno));
7191 	if (err)
7192 		return err;
7193 
7194 	ret_mem->size = reg->var_off.value;
7195 	ret_mem->found = true;
7196 
7197 	return 0;
7198 }
7199 
process_const_arg(struct bpf_verifier_env * env,struct bpf_reg_state * reg,argno_t argno,struct bpf_call_arg_meta * meta)7200 static int process_const_arg(struct bpf_verifier_env *env, struct bpf_reg_state *reg,
7201 			     argno_t argno, struct bpf_call_arg_meta *meta)
7202 {
7203 	int regno = reg_from_argno(argno);
7204 	int err;
7205 
7206 	if (meta->arg_constant.found) {
7207 		verifier_bug(env, "only one constant argument permitted");
7208 		return -EFAULT;
7209 	}
7210 
7211 	if (!tnum_is_const(reg->var_off)) {
7212 		verbose(env, "%s must be a known constant\n", reg_arg_name(env, argno));
7213 		return -EINVAL;
7214 	}
7215 
7216 	if (regno >= 0)
7217 		err = mark_chain_precision(env, regno);
7218 	else
7219 		err = mark_stack_arg_precision(env, arg_idx_from_argno(argno));
7220 	if (err < 0)
7221 		return err;
7222 
7223 	meta->arg_constant.found = true;
7224 	meta->arg_constant.value = reg->var_off.value;
7225 
7226 	return 0;
7227 }
7228 
7229 enum {
7230 	PROCESS_SPIN_LOCK = (1 << 0),
7231 	PROCESS_RES_LOCK  = (1 << 1),
7232 	PROCESS_LOCK_IRQ  = (1 << 2),
7233 };
7234 
7235 /* Implementation details:
7236  * bpf_map_lookup returns PTR_TO_MAP_VALUE_OR_NULL.
7237  * bpf_obj_new returns PTR_TO_BTF_ID | MEM_ALLOC | PTR_MAYBE_NULL.
7238  * Two bpf_map_lookups (even with the same key) will have different reg->id.
7239  * Two separate bpf_obj_new will also have different reg->id.
7240  * For traditional PTR_TO_MAP_VALUE or PTR_TO_BTF_ID | MEM_ALLOC, the verifier
7241  * clears reg->id after value_or_null->value transition, since the verifier only
7242  * cares about the range of access to valid map value pointer and doesn't care
7243  * about actual address of the map element.
7244  * For maps with 'struct bpf_spin_lock' inside map value the verifier keeps
7245  * reg->id > 0 after value_or_null->value transition. By doing so
7246  * two bpf_map_lookups will be considered two different pointers that
7247  * point to different bpf_spin_locks. Likewise for pointers to allocated objects
7248  * returned from bpf_obj_new.
7249  * The verifier allows taking only one bpf_spin_lock at a time to avoid
7250  * dead-locks.
7251  * Since only one bpf_spin_lock is allowed the checks are simpler than
7252  * reg_is_refcounted() logic. The verifier needs to remember only
7253  * one spin_lock instead of array of acquired_refs.
7254  * env->cur_state->active_locks remembers which map value element or allocated
7255  * object got locked and clears it after bpf_spin_unlock.
7256  */
process_spin_lock(struct bpf_verifier_env * env,struct bpf_reg_state * reg,argno_t argno,int flags)7257 static int process_spin_lock(struct bpf_verifier_env *env, struct bpf_reg_state *reg, argno_t argno, int flags)
7258 {
7259 	bool is_lock = flags & PROCESS_SPIN_LOCK, is_res_lock = flags & PROCESS_RES_LOCK;
7260 	const char *lock_str = is_res_lock ? "bpf_res_spin" : "bpf_spin";
7261 	struct bpf_verifier_state *cur = env->cur_state;
7262 	struct bpf_reference_state *lock;
7263 	bool is_const = tnum_is_const(reg->var_off);
7264 	bool is_irq = flags & PROCESS_LOCK_IRQ;
7265 	u64 val = reg->var_off.value;
7266 	struct bpf_map *map = NULL;
7267 	struct btf *btf = NULL;
7268 	struct btf_record *rec;
7269 	u32 spin_lock_off;
7270 	int err;
7271 
7272 	if (!is_const) {
7273 		verbose(env,
7274 			"%s doesn't have constant offset. %s_lock has to be at the constant offset\n",
7275 			reg_arg_name(env, argno), lock_str);
7276 		return -EINVAL;
7277 	}
7278 	if (reg->type == PTR_TO_MAP_VALUE) {
7279 		map = reg->map_ptr;
7280 		if (!map->btf) {
7281 			verbose(env,
7282 				"map '%s' has to have BTF in order to use %s_lock\n",
7283 				map->name, lock_str);
7284 			return -EINVAL;
7285 		}
7286 	} else {
7287 		btf = reg->btf;
7288 	}
7289 
7290 	rec = reg_btf_record(reg);
7291 	if (!btf_record_has_field(rec, is_res_lock ? BPF_RES_SPIN_LOCK : BPF_SPIN_LOCK)) {
7292 		verbose(env, "%s '%s' has no valid %s_lock\n", map ? "map" : "local",
7293 			map ? map->name : "kptr", lock_str);
7294 		return -EINVAL;
7295 	}
7296 	spin_lock_off = is_res_lock ? rec->res_spin_lock_off : rec->spin_lock_off;
7297 	if (spin_lock_off != val) {
7298 		verbose(env, "off %lld doesn't point to 'struct %s_lock' that is at %d\n",
7299 			val, lock_str, spin_lock_off);
7300 		return -EINVAL;
7301 	}
7302 	if (is_lock) {
7303 		void *ptr;
7304 		int type;
7305 
7306 		if (map)
7307 			ptr = map;
7308 		else
7309 			ptr = btf;
7310 
7311 		if (!is_res_lock && cur->active_locks) {
7312 			lock = find_lock_state(cur, REF_TYPE_LOCK, 0, NULL);
7313 			if (lock) {
7314 				verbose(env,
7315 					"Locking two bpf_spin_locks are not allowed\n");
7316 				bpf_diag_lock(
7317 					env, env->insn_idx, "nested spin lock",
7318 					"This path already holds a bpf_spin_lock. The verifier allows only one regular BPF spin lock at a time.",
7319 					"Unlock the current bpf_spin_lock before taking another one.", lock);
7320 				return -EINVAL;
7321 			}
7322 		} else if (is_res_lock && cur->active_locks) {
7323 			lock = find_lock_state(cur, REF_TYPE_RES_LOCK | REF_TYPE_RES_LOCK_IRQ,
7324 					       reg->id, ptr);
7325 			if (lock) {
7326 				verbose(env, "Acquiring the same lock again, AA deadlock detected\n");
7327 				bpf_diag_lock(
7328 					env, env->insn_idx, "recursive resource spin lock",
7329 					"This path already holds the same resource spin lock. Taking it again would deadlock.",
7330 					"Avoid reacquiring the same resource spin lock before it is unlocked.", lock);
7331 				return -EINVAL;
7332 			}
7333 		}
7334 
7335 		if (is_res_lock && is_irq)
7336 			type = REF_TYPE_RES_LOCK_IRQ;
7337 		else if (is_res_lock)
7338 			type = REF_TYPE_RES_LOCK;
7339 		else
7340 			type = REF_TYPE_LOCK;
7341 		err = acquire_lock_state(env, env->insn_idx, type, reg->id, ptr);
7342 		if (err < 0) {
7343 			verbose(env, "Failed to acquire lock state\n");
7344 			return err;
7345 		}
7346 	} else {
7347 		void *ptr;
7348 		int type;
7349 
7350 		if (map)
7351 			ptr = map;
7352 		else
7353 			ptr = btf;
7354 
7355 		if (!cur->active_locks) {
7356 			verbose(env, "%s_unlock without taking a lock\n", lock_str);
7357 			bpf_diag_res(
7358 				env, env->insn_idx, "unlock without lock",
7359 				"This unlock operation has no matching active lock on the current path.",
7360 				"Take the matching lock before this unlock, or remove the unmatched unlock path.");
7361 			return -EINVAL;
7362 		}
7363 
7364 		if (is_res_lock && is_irq)
7365 			type = REF_TYPE_RES_LOCK_IRQ;
7366 		else if (is_res_lock)
7367 			type = REF_TYPE_RES_LOCK;
7368 		else
7369 			type = REF_TYPE_LOCK;
7370 
7371 		lock = find_lock_state(cur, type, reg->id, ptr);
7372 		if (!lock) {
7373 			verbose(env, "%s_unlock of different lock\n", lock_str);
7374 			lock = find_lock_state(cur, REF_TYPE_LOCK_MASK, cur->active_lock_id,
7375 					       cur->active_lock_ptr);
7376 			bpf_diag_lock(
7377 				env, env->insn_idx, "unlock of a different lock",
7378 				"This unlock does not match any active lock with the same tracked identity on the current path.",
7379 				"Unlock the same lock object that was most recently acquired.", lock);
7380 			return -EINVAL;
7381 		}
7382 		if (reg->id != cur->active_lock_id || ptr != cur->active_lock_ptr) {
7383 			verbose(env, "%s_unlock cannot be out of order\n", lock_str);
7384 			lock = find_lock_state(cur, REF_TYPE_LOCK_MASK, cur->active_lock_id,
7385 					       cur->active_lock_ptr);
7386 			bpf_diag_lock(
7387 				env, env->insn_idx, "unlock out of order",
7388 				"Locks must be released in last-in, first-out order, but this unlock does not match the currently active lock.",
7389 				"Release nested locks in the reverse order they were acquired.", lock);
7390 			return -EINVAL;
7391 		}
7392 		if (release_lock_state(env, type, reg->id, ptr)) {
7393 			verbose(env, "%s_unlock of different lock\n", lock_str);
7394 			bpf_diag_lock(
7395 				env, env->insn_idx, "unlock of a different lock",
7396 				"The verifier could not release a lock state matching this unlock operation.",
7397 				"Pass the same lock object and lock kind that were used for the matching lock operation.",
7398 				lock);
7399 			return -EINVAL;
7400 		}
7401 		if (!in_rcu_cs(env))
7402 			invalidate_rcu_protected_refs(env);
7403 
7404 		invalidate_non_owning_refs(env);
7405 	}
7406 	return 0;
7407 }
7408 
7409 /* Check if @regno is a pointer to a specific field in a map value */
check_map_field_pointer(struct bpf_verifier_env * env,struct bpf_reg_state * reg,argno_t argno,enum btf_field_type field_type,struct bpf_map_desc * map_desc)7410 static int check_map_field_pointer(struct bpf_verifier_env *env, struct bpf_reg_state *reg, argno_t argno,
7411 				   enum btf_field_type field_type,
7412 				   struct bpf_map_desc *map_desc)
7413 {
7414 	bool is_const = tnum_is_const(reg->var_off);
7415 	struct bpf_map *map = reg->map_ptr;
7416 	u64 val = reg->var_off.value;
7417 	const char *struct_name = btf_field_type_name(field_type);
7418 	int field_off = -1;
7419 
7420 	if (!is_const) {
7421 		verbose(env,
7422 			"%s doesn't have constant offset. %s has to be at the constant offset\n",
7423 			reg_arg_name(env, argno), struct_name);
7424 		return -EINVAL;
7425 	}
7426 	if (!map->btf) {
7427 		verbose(env, "map '%s' has to have BTF in order to use %s\n", map->name,
7428 			struct_name);
7429 		return -EINVAL;
7430 	}
7431 	if (!btf_record_has_field(map->record, field_type)) {
7432 		verbose(env, "map '%s' has no valid %s\n", map->name, struct_name);
7433 		return -EINVAL;
7434 	}
7435 	switch (field_type) {
7436 	case BPF_TIMER:
7437 		field_off = map->record->timer_off;
7438 		break;
7439 	case BPF_TASK_WORK:
7440 		field_off = map->record->task_work_off;
7441 		break;
7442 	case BPF_WORKQUEUE:
7443 		field_off = map->record->wq_off;
7444 		break;
7445 	default:
7446 		verifier_bug(env, "unsupported BTF field type: %s\n", struct_name);
7447 		return -EINVAL;
7448 	}
7449 	if (field_off != val) {
7450 		verbose(env, "off %lld doesn't point to 'struct %s' that is at %d\n",
7451 			val, struct_name, field_off);
7452 		return -EINVAL;
7453 	}
7454 	if (map_desc->ptr) {
7455 		verifier_bug(env, "Two map pointers in a %s helper", struct_name);
7456 		return -EFAULT;
7457 	}
7458 	map_desc->uid = reg->map_uid;
7459 	map_desc->ptr = map;
7460 	return 0;
7461 }
7462 
process_timer_func(struct bpf_verifier_env * env,struct bpf_reg_state * reg,argno_t argno,struct bpf_map_desc * map)7463 static int process_timer_func(struct bpf_verifier_env *env, struct bpf_reg_state *reg, argno_t argno,
7464 			      struct bpf_map_desc *map)
7465 {
7466 	if (IS_ENABLED(CONFIG_PREEMPT_RT)) {
7467 		verbose(env, "bpf_timer cannot be used for PREEMPT_RT.\n");
7468 		return -EOPNOTSUPP;
7469 	}
7470 	return check_map_field_pointer(env, reg, argno, BPF_TIMER, map);
7471 }
7472 
process_kptr_func(struct bpf_verifier_env * env,int regno,struct bpf_call_arg_meta * meta)7473 static int process_kptr_func(struct bpf_verifier_env *env, int regno,
7474 			     struct bpf_call_arg_meta *meta)
7475 {
7476 	struct bpf_reg_state *reg = reg_state(env, regno);
7477 	struct btf_field *kptr_field;
7478 	struct bpf_map *map_ptr;
7479 	struct btf_record *rec;
7480 	u32 kptr_off;
7481 
7482 	if (type_is_ptr_alloc_obj(reg->type)) {
7483 		rec = reg_btf_record(reg);
7484 	} else { /* PTR_TO_MAP_VALUE */
7485 		map_ptr = reg->map_ptr;
7486 		if (!map_ptr->btf) {
7487 			verbose(env, "map '%s' has to have BTF in order to use bpf_kptr_xchg\n",
7488 				map_ptr->name);
7489 			return -EINVAL;
7490 		}
7491 		rec = map_ptr->record;
7492 		meta->map.ptr = map_ptr;
7493 	}
7494 
7495 	if (!tnum_is_const(reg->var_off)) {
7496 		verbose(env,
7497 			"R%d doesn't have constant offset. kptr has to be at the constant offset\n",
7498 			regno);
7499 		return -EINVAL;
7500 	}
7501 
7502 	if (!btf_record_has_field(rec, BPF_KPTR)) {
7503 		verbose(env, "R%d has no valid kptr\n", regno);
7504 		return -EINVAL;
7505 	}
7506 
7507 	kptr_off = reg->var_off.value;
7508 	kptr_field = btf_record_find(rec, kptr_off, BPF_KPTR);
7509 	if (!kptr_field) {
7510 		verbose(env, "off=%d doesn't point to kptr\n", kptr_off);
7511 		return -EACCES;
7512 	}
7513 	if (kptr_field->type != BPF_KPTR_REF && kptr_field->type != BPF_KPTR_PERCPU) {
7514 		verbose(env, "off=%d kptr isn't referenced kptr\n", kptr_off);
7515 		return -EACCES;
7516 	}
7517 	meta->kptr_field = kptr_field;
7518 	return 0;
7519 }
7520 
7521 static void bpf_diag_call_arg(struct bpf_verifier_env *env, u32 insn_idx, argno_t argno,
7522 			      const char *call_name, const char *reason, const char *suggestion);
7523 __printf(6, 7) static void bpf_diag_call_arg_fmt(struct bpf_verifier_env *env, u32 insn_idx,
7524 						 argno_t argno, const char *call_name,
7525 						 const char *suggestion, const char *fmt, ...);
7526 
7527 /*
7528  * Validate dynptr arguments for helper, kfunc and subprog.
7529  *
7530  * @dynptr is both input and output. It is populated when the argument is
7531  * tagged with MEM_UNINIT (i.e., the dynptr argument that will be constructed)
7532  * and consumed when the argument is expecting to be an initialized dynptr.
7533  * @parent_id is used to track the referenced parent object (e.g., file or skb in
7534  * qdisc program) when constructing a dynptr.
7535  *
7536  * There are two register types representing a bpf_dynptr, one is PTR_TO_STACK
7537  * which points to a stack slot, and the other is CONST_PTR_TO_DYNPTR.
7538  *
7539  * In both cases we deal with the first 8 bytes, but need to mark the next 8
7540  * bytes as STACK_DYNPTR in case of PTR_TO_STACK. In case of
7541  * CONST_PTR_TO_DYNPTR, we are guaranteed to get the beginning of the object.
7542  *
7543  * Mutability of bpf_dynptr is at two levels: the dynptr and the memory the
7544  * dynptr points to. At the first level, the verifier will make sure a
7545  * CONST_PTR_TO_DYNPTR cannot be reinitialized or destroyed. The mutability of
7546  * a dynptr's view (i.e., start and offset) is not tracked as there is not such
7547  * use case. The second level is tracked using the upper bit of bpf_dynptr->size
7548  * and checked dynamically during runtime.
7549  */
process_dynptr_func(struct bpf_verifier_env * env,struct bpf_reg_state * reg,argno_t argno,int insn_idx,const char * call_name,enum bpf_arg_type arg_type,struct ref_obj_desc * ref_obj,struct bpf_dynptr_desc * dynptr)7550 static int process_dynptr_func(struct bpf_verifier_env *env, struct bpf_reg_state *reg,
7551 			       argno_t argno, int insn_idx, const char *call_name,
7552 			       enum bpf_arg_type arg_type,
7553 			       struct ref_obj_desc *ref_obj, struct bpf_dynptr_desc *dynptr)
7554 {
7555 	int spi, err = 0;
7556 
7557 	if (reg->type != PTR_TO_STACK && reg->type != CONST_PTR_TO_DYNPTR) {
7558 		verbose(env,
7559 			"%s expected pointer to stack or const struct bpf_dynptr\n",
7560 			reg_arg_name(env, argno));
7561 		bpf_diag_call_arg_fmt(
7562 			env, insn_idx, argno, call_name,
7563 			"Pass the address of a stack dynptr object, or use a const dynptr pointer returned by the verifier-supported path.",
7564 			"a dynptr argument must be a pointer to a dynptr stack slot or a verifier-provided const struct bpf_dynptr, but %s is %s",
7565 			reg_arg_name(env, argno), bpf_diag_reg_type_plain(env, reg->type));
7566 		return -EINVAL;
7567 	}
7568 
7569 	/*  MEM_UNINIT - Points to memory that is an appropriate candidate for
7570 	 *		 constructing a mutable bpf_dynptr object.
7571 	 *
7572 	 *		 Currently, this is only possible with PTR_TO_STACK
7573 	 *		 pointing to a region of at least 16 bytes which doesn't
7574 	 *		 contain an existing bpf_dynptr.
7575 	 *
7576 	 *  OBJ_RELEASE - Points to a initialized bpf_dynptr that will be
7577 	 *		  destroyed.
7578 	 *
7579 	 *  None       - Points to a initialized dynptr that cannot be
7580 	 *		 reinitialized or destroyed. However, the view of the
7581 	 *		 dynptr and the memory it points to may be mutated.
7582 	 */
7583 	if (arg_type & MEM_UNINIT) {
7584 		int i;
7585 
7586 		if (!is_dynptr_reg_valid_uninit(env, reg)) {
7587 			verbose(env, "Dynptr has to be an uninitialized dynptr\n");
7588 			bpf_diag_res(
7589 				env, insn_idx, "dynptr is already initialized",
7590 				"This kfunc constructs a dynptr and requires an uninitialized dynptr stack slot, but the selected slot already holds dynptr state.",
7591 				"Use a fresh stack dynptr slot, or release/destroy the existing dynptr before reusing the slot.");
7592 			return -EINVAL;
7593 		}
7594 
7595 		/* we write BPF_DW bits (8 bytes) at a time */
7596 		for (i = 0; i < BPF_DYNPTR_SIZE; i += 8) {
7597 			err = check_mem_access(env, insn_idx, reg, argno,
7598 					       i, BPF_DW, BPF_WRITE, -1, false, false);
7599 			if (err)
7600 				return err;
7601 		}
7602 
7603 		err = mark_stack_slots_dynptr(env, reg, arg_type, insn_idx, ref_obj, dynptr);
7604 	} else /* OBJ_RELEASE and None case from above */ {
7605 		/* For the reg->type == PTR_TO_STACK case, bpf_dynptr is never const */
7606 		if (reg->type == CONST_PTR_TO_DYNPTR && (arg_type & OBJ_RELEASE)) {
7607 			verbose(env, "CONST_PTR_TO_DYNPTR cannot be released\n");
7608 			bpf_diag_res(
7609 				env, insn_idx, "const dynptr release",
7610 				"This release operation was given a const dynptr. Const dynptr values are verifier-provided views and cannot be released by the program.",
7611 				"Release only mutable dynptrs that the program initialized or reserved.");
7612 			return -EINVAL;
7613 		}
7614 
7615 		if (!is_dynptr_reg_valid_init(env, reg)) {
7616 			verbose(env, "Expected an initialized dynptr as %s\n",
7617 				reg_arg_name(env, argno));
7618 			bpf_diag_res(
7619 				env, insn_idx, "uninitialized dynptr use",
7620 				"This operation requires an initialized dynptr, but the stack slot does not currently hold a valid dynptr on this path.",
7621 				"Initialize the dynptr on every path before this call, and avoid overwriting or releasing it before this use.");
7622 			return -EINVAL;
7623 		}
7624 
7625 		/* Fold modifiers (in this case, OBJ_RELEASE) when checking expected type */
7626 		if (!is_dynptr_type_expected(env, reg, arg_type & ~OBJ_RELEASE)) {
7627 			enum bpf_dynptr_type expected_type = arg_to_dynptr_type(arg_type);
7628 			enum bpf_dynptr_type actual_type = dynptr_reg_type(env, reg);
7629 
7630 			verbose(env, "Expected a dynptr of type %s as %s\n",
7631 				dynptr_type_str(expected_type), reg_arg_name(env, argno));
7632 			bpf_diag_call_arg_fmt(
7633 				env, insn_idx, argno, call_name,
7634 				"Use a dynptr constructor that matches this operation, or call an operation that accepts the dynptr's current type.",
7635 				"the dynptr is initialized with backing object type %s, but this operation expects dynptr type %s",
7636 				dynptr_type_str(actual_type), dynptr_type_str(expected_type));
7637 			return -EINVAL;
7638 		}
7639 
7640 		if (reg->type != CONST_PTR_TO_DYNPTR) {
7641 			struct bpf_func_state *state = bpf_func(env, reg);
7642 
7643 			spi = dynptr_get_spi(env, reg);
7644 			if (spi < 0)
7645 				return spi;
7646 
7647 			mark_stack_slots_scratched(env, spi, BPF_DYNPTR_NR_SLOTS);
7648 
7649 			reg = &state->stack[spi].spilled_ptr;
7650 		}
7651 
7652 		if (dynptr) {
7653 			dynptr->type = reg->dynptr.type;
7654 			dynptr->id = reg->id;
7655 			dynptr->parent_id = reg->parent_id;
7656 		}
7657 	}
7658 	return err;
7659 }
7660 
is_iter_kfunc(struct bpf_call_arg_meta * meta)7661 static bool is_iter_kfunc(struct bpf_call_arg_meta *meta)
7662 {
7663 	return meta->kfunc_flags & (KF_ITER_NEW | KF_ITER_NEXT | KF_ITER_DESTROY);
7664 }
7665 
is_iter_new_kfunc(struct bpf_call_arg_meta * meta)7666 static bool is_iter_new_kfunc(struct bpf_call_arg_meta *meta)
7667 {
7668 	return meta->kfunc_flags & KF_ITER_NEW;
7669 }
7670 
is_iter_destroy_kfunc(struct bpf_call_arg_meta * meta)7671 static bool is_iter_destroy_kfunc(struct bpf_call_arg_meta *meta)
7672 {
7673 	return meta->kfunc_flags & KF_ITER_DESTROY;
7674 }
7675 
is_kfunc_arg_iter(struct bpf_call_arg_meta * meta,int arg_idx,const struct btf_param * arg)7676 static bool is_kfunc_arg_iter(struct bpf_call_arg_meta *meta, int arg_idx,
7677 			      const struct btf_param *arg)
7678 {
7679 	/* btf_check_iter_kfuncs() guarantees that first argument of any iter
7680 	 * kfunc is iter state pointer
7681 	 */
7682 	if (is_iter_kfunc(meta))
7683 		return arg_idx == 0;
7684 
7685 	/* iter passed as an argument to a generic kfunc */
7686 	return btf_param_match_suffix(meta->btf, arg, "__iter");
7687 }
7688 
process_iter_arg(struct bpf_verifier_env * env,struct bpf_reg_state * reg,argno_t argno,int insn_idx,struct bpf_call_arg_meta * meta)7689 static int process_iter_arg(struct bpf_verifier_env *env, struct bpf_reg_state *reg, argno_t argno, int insn_idx,
7690 			    struct bpf_call_arg_meta *meta)
7691 {
7692 	struct bpf_func_state *state = bpf_func(env, reg);
7693 	const struct btf_type *t;
7694 	u32 arg_idx = arg_idx_from_argno(argno);
7695 	int spi, err, i, nr_slots, btf_id;
7696 
7697 	if (reg->type != PTR_TO_STACK) {
7698 		verbose(env, "%s expected pointer to an iterator on stack\n",
7699 			reg_arg_name(env, argno));
7700 		bpf_diag_call_arg_fmt(
7701 			env, insn_idx, argno, meta->func_name,
7702 			"Pass the address of a stack iterator object for iterator new, next, and destroy calls.",
7703 			"iterator state must live in verifier-tracked stack memory, but %s is %s",
7704 			reg_arg_name(env, argno), bpf_diag_reg_type_plain(env, reg->type));
7705 		return -EINVAL;
7706 	}
7707 
7708 	/* For iter_{new,next,destroy} functions, btf_check_iter_kfuncs()
7709 	 * ensures struct convention, so we wouldn't need to do any BTF
7710 	 * validation here. But given iter state can be passed as a parameter
7711 	 * to any kfunc, if arg has "__iter" suffix, we need to be a bit more
7712 	 * conservative here.
7713 	 */
7714 	btf_id = btf_check_iter_arg(meta->btf, meta->func_proto, arg_idx);
7715 	if (btf_id < 0) {
7716 		verbose(env, "expected valid iter pointer as %s\n",
7717 			reg_arg_name(env, argno));
7718 		bpf_diag_call_arg(
7719 			env, insn_idx, argno, meta->func_name,
7720 			"the kfunc expects a recognized iterator state pointer, but this argument does not match a valid iterator type",
7721 			"Pass the exact iterator state type expected by this kfunc.");
7722 		return -EINVAL;
7723 	}
7724 	t = btf_type_by_id(meta->btf, btf_id);
7725 	nr_slots = t->size / BPF_REG_SIZE;
7726 
7727 	if (is_iter_new_kfunc(meta)) {
7728 		/* bpf_iter_<type>_new() expects pointer to uninit iter state */
7729 		if (!is_iter_reg_valid_uninit(env, reg, nr_slots)) {
7730 			verbose(env, "expected uninitialized iter_%s as %s\n",
7731 				iter_type_str(meta->btf, btf_id), reg_arg_name(env, argno));
7732 			bpf_diag_res(
7733 				env, insn_idx, "iterator is already initialized",
7734 				"Iterator creation requires an uninitialized iterator stack object, but this stack range already contains iterator state.",
7735 				"Use a fresh iterator stack slot, or destroy the existing iterator before reusing the slot.");
7736 			return -EINVAL;
7737 		}
7738 
7739 		for (i = 0; i < nr_slots * 8; i += BPF_REG_SIZE) {
7740 			err = check_mem_access(env, insn_idx, reg, argno,
7741 					       i, BPF_DW, BPF_WRITE, -1, false, false);
7742 			if (err)
7743 				return err;
7744 		}
7745 
7746 		err = mark_stack_slots_iter(env, meta, reg, insn_idx, meta->btf, btf_id, nr_slots);
7747 		if (err)
7748 			return err;
7749 	} else {
7750 		/* iter_next() or iter_destroy(), as well as any kfunc
7751 		 * accepting iter argument, expect initialized iter state
7752 		 */
7753 		err = is_iter_reg_valid_init(env, reg, meta->btf, btf_id, nr_slots);
7754 		switch (err) {
7755 		case 0:
7756 			break;
7757 		case -EINVAL:
7758 			verbose(env, "expected an initialized iter_%s as %s\n",
7759 				iter_type_str(meta->btf, btf_id), reg_arg_name(env, argno));
7760 			bpf_diag_res(
7761 				env, insn_idx, "uninitialized iterator use",
7762 				"This iterator operation requires an initialized iterator state object, but the stack range does not contain a live iterator on this path.",
7763 				"Call the matching iterator new kfunc on every path before calling next or destroy, and do not destroy the iterator before this use.");
7764 			return err;
7765 		case -EPROTO:
7766 			verbose(env, "expected an RCU CS when using %s\n", meta->func_name);
7767 			bpf_diag_ctx_required(
7768 				env, insn_idx, meta->func_name, BPF_DIAG_CONTEXT_RCU,
7769 				"Wrap iterator use in bpf_rcu_read_lock() and bpf_rcu_read_unlock(), keeping all exit paths balanced.");
7770 			return err;
7771 		default:
7772 			return err;
7773 		}
7774 
7775 		spi = iter_get_spi(env, reg, nr_slots);
7776 		if (spi < 0)
7777 			return spi;
7778 
7779 		mark_stack_slots_scratched(env, spi, nr_slots);
7780 
7781 		/* remember meta->iter info for process_iter_next_call() */
7782 		meta->iter.spi = spi;
7783 		meta->iter.frameno = reg->frameno;
7784 		update_ref_obj(&meta->ref_obj, &state->stack[spi].spilled_ptr);
7785 
7786 		if (is_iter_destroy_kfunc(meta)) {
7787 			err = unmark_stack_slots_iter(env, reg, nr_slots);
7788 			if (err)
7789 				return err;
7790 		}
7791 	}
7792 
7793 	return 0;
7794 }
7795 
7796 /* Look for a previous loop entry at insn_idx: nearest parent state
7797  * stopped at insn_idx with callsites matching those in cur->frame.
7798  */
find_prev_entry(struct bpf_verifier_env * env,struct bpf_verifier_state * cur,int insn_idx)7799 static struct bpf_verifier_state *find_prev_entry(struct bpf_verifier_env *env,
7800 						  struct bpf_verifier_state *cur,
7801 						  int insn_idx)
7802 {
7803 	struct bpf_verifier_state_list *sl;
7804 	struct bpf_verifier_state *st;
7805 	struct list_head *pos, *head;
7806 
7807 	/* Explored states are pushed in stack order, most recent states come first */
7808 	head = bpf_explored_state(env, insn_idx);
7809 	list_for_each(pos, head) {
7810 		sl = container_of(pos, struct bpf_verifier_state_list, node);
7811 		/* If st->branches != 0 state is a part of current DFS verification path,
7812 		 * hence cur & st for a loop.
7813 		 */
7814 		st = &sl->state;
7815 		if (st->insn_idx == insn_idx && st->branches && same_callsites(st, cur) &&
7816 		    st->dfs_depth < cur->dfs_depth)
7817 			return st;
7818 	}
7819 
7820 	return NULL;
7821 }
7822 
7823 /*
7824  * Check if scalar registers are exact for the purpose of not widening.
7825  * More lenient than regs_exact()
7826  */
scalars_exact_for_widen(const struct bpf_reg_state * rold,const struct bpf_reg_state * rcur)7827 static bool scalars_exact_for_widen(const struct bpf_reg_state *rold,
7828 				    const struct bpf_reg_state *rcur)
7829 {
7830 	return !memcmp(rold, rcur, offsetof(struct bpf_reg_state, id));
7831 }
7832 
maybe_widen_reg(struct bpf_verifier_env * env,struct bpf_reg_state * rold,struct bpf_reg_state * rcur)7833 static void maybe_widen_reg(struct bpf_verifier_env *env,
7834 			    struct bpf_reg_state *rold, struct bpf_reg_state *rcur)
7835 {
7836 	if (rold->type != SCALAR_VALUE)
7837 		return;
7838 	if (rold->type != rcur->type)
7839 		return;
7840 	if (rold->precise || rcur->precise || scalars_exact_for_widen(rold, rcur))
7841 		return;
7842 	__mark_reg_unknown(env, rcur);
7843 }
7844 
widen_imprecise_scalars(struct bpf_verifier_env * env,struct bpf_verifier_state * old,struct bpf_verifier_state * cur)7845 static int widen_imprecise_scalars(struct bpf_verifier_env *env,
7846 				   struct bpf_verifier_state *old,
7847 				   struct bpf_verifier_state *cur)
7848 {
7849 	struct bpf_func_state *fold, *fcur;
7850 	int i, fr, num_slots;
7851 
7852 	for (fr = old->curframe; fr >= 0; fr--) {
7853 		fold = old->frame[fr];
7854 		fcur = cur->frame[fr];
7855 
7856 		for (i = 0; i < MAX_BPF_REG; i++)
7857 			maybe_widen_reg(env,
7858 					&fold->regs[i],
7859 					&fcur->regs[i]);
7860 
7861 		num_slots = min(fold->allocated_stack / BPF_REG_SIZE,
7862 				fcur->allocated_stack / BPF_REG_SIZE);
7863 		for (i = 0; i < num_slots; i++) {
7864 			if (!bpf_is_spilled_reg(&fold->stack[i]) ||
7865 			    !bpf_is_spilled_reg(&fcur->stack[i]))
7866 				continue;
7867 
7868 			maybe_widen_reg(env,
7869 					&fold->stack[i].spilled_ptr,
7870 					&fcur->stack[i].spilled_ptr);
7871 		}
7872 	}
7873 	return 0;
7874 }
7875 
get_iter_from_state(struct bpf_verifier_state * cur_st,struct bpf_call_arg_meta * meta)7876 static struct bpf_reg_state *get_iter_from_state(struct bpf_verifier_state *cur_st,
7877 						 struct bpf_call_arg_meta *meta)
7878 {
7879 	int iter_frameno = meta->iter.frameno;
7880 	int iter_spi = meta->iter.spi;
7881 
7882 	return &cur_st->frame[iter_frameno]->stack[iter_spi].spilled_ptr;
7883 }
7884 
7885 /* process_iter_next_call() is called when verifier gets to iterator's next
7886  * "method" (e.g., bpf_iter_num_next() for numbers iterator) call. We'll refer
7887  * to it as just "iter_next()" in comments below.
7888  *
7889  * BPF verifier relies on a crucial contract for any iter_next()
7890  * implementation: it should *eventually* return NULL, and once that happens
7891  * it should keep returning NULL. That is, once iterator exhausts elements to
7892  * iterate, it should never reset or spuriously return new elements.
7893  *
7894  * With the assumption of such contract, process_iter_next_call() simulates
7895  * a fork in the verifier state to validate loop logic correctness and safety
7896  * without having to simulate infinite amount of iterations.
7897  *
7898  * In current state, we first assume that iter_next() returned NULL and
7899  * iterator state is set to DRAINED (BPF_ITER_STATE_DRAINED). In such
7900  * conditions we should not form an infinite loop and should eventually reach
7901  * exit.
7902  *
7903  * Besides that, we also fork current state and enqueue it for later
7904  * verification. In a forked state we keep iterator state as ACTIVE
7905  * (BPF_ITER_STATE_ACTIVE) and assume non-NULL return from iter_next(). We
7906  * also bump iteration depth to prevent erroneous infinite loop detection
7907  * later on (see iter_active_depths_differ() comment for details). In this
7908  * state we assume that we'll eventually loop back to another iter_next()
7909  * calls (it could be in exactly same location or in some other instruction,
7910  * it doesn't matter, we don't make any unnecessary assumptions about this,
7911  * everything revolves around iterator state in a stack slot, not which
7912  * instruction is calling iter_next()). When that happens, we either will come
7913  * to iter_next() with equivalent state and can conclude that next iteration
7914  * will proceed in exactly the same way as we just verified, so it's safe to
7915  * assume that loop converges. If not, we'll go on another iteration
7916  * simulation with a different input state, until all possible starting states
7917  * are validated or we reach maximum number of instructions limit.
7918  *
7919  * This way, we will either exhaustively discover all possible input states
7920  * that iterator loop can start with and eventually will converge, or we'll
7921  * effectively regress into bounded loop simulation logic and either reach
7922  * maximum number of instructions if loop is not provably convergent, or there
7923  * is some statically known limit on number of iterations (e.g., if there is
7924  * an explicit `if n > 100 then break;` statement somewhere in the loop).
7925  *
7926  * Iteration convergence logic in is_state_visited() relies on exact
7927  * states comparison, which ignores read and precision marks.
7928  * This is necessary because read and precision marks are not finalized
7929  * while in the loop. Exact comparison might preclude convergence for
7930  * simple programs like below:
7931  *
7932  *     i = 0;
7933  *     while(iter_next(&it))
7934  *       i++;
7935  *
7936  * At each iteration step i++ would produce a new distinct state and
7937  * eventually instruction processing limit would be reached.
7938  *
7939  * To avoid such behavior speculatively forget (widen) range for
7940  * imprecise scalar registers, if those registers were not precise at the
7941  * end of the previous iteration and do not match exactly.
7942  *
7943  * This is a conservative heuristic that allows to verify wide range of programs,
7944  * however it precludes verification of programs that conjure an
7945  * imprecise value on the first loop iteration and use it as precise on a second.
7946  * For example, the following safe program would fail to verify:
7947  *
7948  *     struct bpf_num_iter it;
7949  *     int arr[10];
7950  *     int i = 0, a = 0;
7951  *     bpf_iter_num_new(&it, 0, 10);
7952  *     while (bpf_iter_num_next(&it)) {
7953  *       if (a == 0) {
7954  *         a = 1;
7955  *         i = 7; // Because i changed verifier would forget
7956  *                // it's range on second loop entry.
7957  *       } else {
7958  *         arr[i] = 42; // This would fail to verify.
7959  *       }
7960  *     }
7961  *     bpf_iter_num_destroy(&it);
7962  */
process_iter_next_call(struct bpf_verifier_env * env,int insn_idx,struct bpf_call_arg_meta * meta)7963 static int process_iter_next_call(struct bpf_verifier_env *env, int insn_idx,
7964 				  struct bpf_call_arg_meta *meta)
7965 {
7966 	struct bpf_verifier_state *cur_st = env->cur_state, *queued_st, *prev_st;
7967 	struct bpf_func_state *cur_fr = cur_st->frame[cur_st->curframe], *queued_fr;
7968 	struct bpf_reg_state *cur_iter, *queued_iter;
7969 
7970 	BTF_TYPE_EMIT(struct bpf_iter);
7971 
7972 	cur_iter = get_iter_from_state(cur_st, meta);
7973 
7974 	if (cur_iter->iter.state != BPF_ITER_STATE_ACTIVE &&
7975 	    cur_iter->iter.state != BPF_ITER_STATE_DRAINED) {
7976 		verifier_bug(env, "unexpected iterator state %d (%s)",
7977 			     cur_iter->iter.state, iter_state_str(cur_iter->iter.state));
7978 		return -EFAULT;
7979 	}
7980 
7981 	if (cur_iter->iter.state == BPF_ITER_STATE_ACTIVE) {
7982 		/* Because iter_next() call is a checkpoint is_state_visitied()
7983 		 * should guarantee parent state with same call sites and insn_idx.
7984 		 */
7985 		if (!cur_st->parent || cur_st->parent->insn_idx != insn_idx ||
7986 		    !same_callsites(cur_st->parent, cur_st)) {
7987 			verifier_bug(env, "bad parent state for iter next call");
7988 			return -EFAULT;
7989 		}
7990 		/* Note cur_st->parent in the call below, it is necessary to skip
7991 		 * checkpoint created for cur_st by is_state_visited()
7992 		 * right at this instruction.
7993 		 */
7994 		prev_st = find_prev_entry(env, cur_st->parent, insn_idx);
7995 		/* branch out active iter state */
7996 		queued_st = push_stack(env, insn_idx + 1, insn_idx, false);
7997 		if (IS_ERR(queued_st))
7998 			return PTR_ERR(queued_st);
7999 
8000 		queued_iter = get_iter_from_state(queued_st, meta);
8001 		queued_iter->iter.state = BPF_ITER_STATE_ACTIVE;
8002 		queued_iter->iter.depth++;
8003 		if (prev_st)
8004 			widen_imprecise_scalars(env, prev_st, queued_st);
8005 
8006 		queued_fr = queued_st->frame[queued_st->curframe];
8007 		mark_ptr_not_null_reg(&queued_fr->regs[BPF_REG_0]);
8008 	}
8009 
8010 	/* switch to DRAINED state, but keep the depth unchanged */
8011 	/* mark current iter state as drained and assume returned NULL */
8012 	cur_iter->iter.state = BPF_ITER_STATE_DRAINED;
8013 	__mark_reg_const_zero(env, &cur_fr->regs[BPF_REG_0]);
8014 
8015 	return 0;
8016 }
8017 
arg_type_is_mem_size(enum bpf_arg_type type)8018 static bool arg_type_is_mem_size(enum bpf_arg_type type)
8019 {
8020 	return type == ARG_MEM_SIZE || type == ARG_MEM_SIZE_OR_ZERO;
8021 }
8022 
arg_type_is_raw_mem(enum bpf_arg_type type)8023 static bool arg_type_is_raw_mem(enum bpf_arg_type type)
8024 {
8025 	/*
8026 	 * A map value output buffer (e.g. bpf_map_pop_elem) is also a raw
8027 	 * (uninitialized) memory argument, and like ARG_PTR_TO_MEM it may be
8028 	 * passed as a PTR_TO_STACK that reaches check_stack_range_initialized().
8029 	 */
8030 	return (base_type(type) == ARG_PTR_TO_MEM ||
8031 		base_type(type) == ARG_PTR_TO_MAP_VALUE) &&
8032 	       type & MEM_UNINIT;
8033 }
8034 
arg_type_is_release(enum bpf_arg_type type)8035 static bool arg_type_is_release(enum bpf_arg_type type)
8036 {
8037 	return type & OBJ_RELEASE;
8038 }
8039 
arg_type_is_dynptr(enum bpf_arg_type type)8040 static bool arg_type_is_dynptr(enum bpf_arg_type type)
8041 {
8042 	return base_type(type) == ARG_PTR_TO_DYNPTR;
8043 }
8044 
resolve_map_arg_type(struct bpf_verifier_env * env,const struct bpf_call_arg_meta * meta,enum bpf_arg_type * arg_type)8045 static int resolve_map_arg_type(struct bpf_verifier_env *env,
8046 				 const struct bpf_call_arg_meta *meta,
8047 				 enum bpf_arg_type *arg_type)
8048 {
8049 	if (!meta->map.ptr) {
8050 		/* kernel subsystem misconfigured verifier */
8051 		verifier_bug(env, "invalid map_ptr to access map->type");
8052 		return -EFAULT;
8053 	}
8054 
8055 	switch (meta->map.ptr->map_type) {
8056 	case BPF_MAP_TYPE_SOCKMAP:
8057 	case BPF_MAP_TYPE_SOCKHASH:
8058 		if (*arg_type == ARG_PTR_TO_MAP_VALUE) {
8059 			*arg_type = ARG_PTR_TO_BTF_ID_SOCK_COMMON;
8060 		} else {
8061 			verbose(env, "invalid arg_type for sockmap/sockhash\n");
8062 			return -EINVAL;
8063 		}
8064 		break;
8065 	case BPF_MAP_TYPE_BLOOM_FILTER:
8066 		if (meta->func_id == BPF_FUNC_map_peek_elem)
8067 			*arg_type = ARG_PTR_TO_MAP_VALUE;
8068 		break;
8069 	default:
8070 		break;
8071 	}
8072 	return 0;
8073 }
8074 
8075 struct bpf_reg_types {
8076 	const enum bpf_reg_type types[10];
8077 	u32 *btf_id;
8078 };
8079 
8080 static const struct bpf_reg_types sock_types = {
8081 	.types = {
8082 		PTR_TO_SOCK_COMMON,
8083 		PTR_TO_SOCKET,
8084 		PTR_TO_TCP_SOCK,
8085 		PTR_TO_XDP_SOCK,
8086 	},
8087 };
8088 
8089 #ifdef CONFIG_NET
8090 static const struct bpf_reg_types btf_id_sock_common_types = {
8091 	.types = {
8092 		PTR_TO_SOCK_COMMON,
8093 		PTR_TO_SOCKET,
8094 		PTR_TO_TCP_SOCK,
8095 		PTR_TO_XDP_SOCK,
8096 		PTR_TO_BTF_ID,
8097 		PTR_TO_BTF_ID | PTR_TRUSTED,
8098 	},
8099 	.btf_id = &btf_sock_ids[BTF_SOCK_TYPE_SOCK_COMMON],
8100 };
8101 #endif
8102 
8103 static const struct bpf_reg_types mem_types = {
8104 	.types = {
8105 		PTR_TO_STACK,
8106 		PTR_TO_PACKET,
8107 		PTR_TO_PACKET_META,
8108 		PTR_TO_MAP_KEY,
8109 		PTR_TO_MAP_VALUE,
8110 		PTR_TO_MEM,
8111 		PTR_TO_MEM | MEM_RINGBUF,
8112 		PTR_TO_BUF,
8113 		PTR_TO_BTF_ID | PTR_TRUSTED,
8114 		PTR_TO_CTX,
8115 	},
8116 };
8117 
8118 static const struct bpf_reg_types spin_lock_types = {
8119 	.types = {
8120 		PTR_TO_MAP_VALUE,
8121 		PTR_TO_BTF_ID | MEM_ALLOC,
8122 	}
8123 };
8124 
8125 static const struct bpf_reg_types fullsock_types = { .types = { PTR_TO_SOCKET } };
8126 static const struct bpf_reg_types scalar_types = { .types = { SCALAR_VALUE } };
8127 static const struct bpf_reg_types context_types = { .types = { PTR_TO_CTX } };
8128 static const struct bpf_reg_types ringbuf_mem_types = { .types = { PTR_TO_MEM | MEM_RINGBUF } };
8129 static const struct bpf_reg_types const_map_ptr_types = { .types = { CONST_PTR_TO_MAP } };
8130 static const struct bpf_reg_types btf_ptr_types = {
8131 	.types = {
8132 		PTR_TO_BTF_ID,
8133 		PTR_TO_BTF_ID | PTR_TRUSTED,
8134 		PTR_TO_BTF_ID | MEM_RCU,
8135 	},
8136 };
8137 static const struct bpf_reg_types percpu_btf_ptr_types = {
8138 	.types = {
8139 		PTR_TO_BTF_ID | MEM_PERCPU,
8140 		PTR_TO_BTF_ID | MEM_PERCPU | MEM_RCU,
8141 		PTR_TO_BTF_ID | MEM_PERCPU | PTR_TRUSTED,
8142 	}
8143 };
8144 static const struct bpf_reg_types func_ptr_types = { .types = { PTR_TO_FUNC } };
8145 static const struct bpf_reg_types stack_ptr_types = { .types = { PTR_TO_STACK } };
8146 static const struct bpf_reg_types const_str_ptr_types = { .types = { PTR_TO_MAP_VALUE } };
8147 static const struct bpf_reg_types timer_types = { .types = { PTR_TO_MAP_VALUE } };
8148 static const struct bpf_reg_types kptr_xchg_dest_types = {
8149 	.types = {
8150 		PTR_TO_MAP_VALUE,
8151 		PTR_TO_BTF_ID | MEM_ALLOC,
8152 		PTR_TO_BTF_ID | MEM_ALLOC | NON_OWN_REF,
8153 		PTR_TO_BTF_ID | MEM_ALLOC | NON_OWN_REF | MEM_RCU,
8154 	}
8155 };
8156 static const struct bpf_reg_types dynptr_types = {
8157 	.types = {
8158 		PTR_TO_STACK,
8159 		CONST_PTR_TO_DYNPTR,
8160 	}
8161 };
8162 
8163 static const struct bpf_reg_types *compatible_reg_types[__BPF_ARG_TYPE_MAX] = {
8164 	[ARG_PTR_TO_MAP_KEY]		= &mem_types,
8165 	[ARG_PTR_TO_MAP_VALUE]		= &mem_types,
8166 	[ARG_MEM_SIZE]			= &scalar_types,
8167 	[ARG_MEM_SIZE_OR_ZERO]		= &scalar_types,
8168 	[ARG_CONST_ALLOC_SIZE_OR_ZERO]	= &scalar_types,
8169 	[ARG_CONST_MAP_PTR]		= &const_map_ptr_types,
8170 	[ARG_PTR_TO_CTX]		= &context_types,
8171 	[ARG_PTR_TO_SOCK_COMMON]	= &sock_types,
8172 #ifdef CONFIG_NET
8173 	[ARG_PTR_TO_BTF_ID_SOCK_COMMON]	= &btf_id_sock_common_types,
8174 #endif
8175 	[ARG_PTR_TO_SOCKET]		= &fullsock_types,
8176 	[ARG_PTR_TO_BTF_ID]		= &btf_ptr_types,
8177 	[ARG_PTR_TO_SPIN_LOCK]		= &spin_lock_types,
8178 	[ARG_PTR_TO_MEM]		= &mem_types,
8179 	[ARG_PTR_TO_RINGBUF_MEM]	= &ringbuf_mem_types,
8180 	[ARG_PTR_TO_PERCPU_BTF_ID]	= &percpu_btf_ptr_types,
8181 	[ARG_PTR_TO_FUNC]		= &func_ptr_types,
8182 	[ARG_PTR_TO_STACK]		= &stack_ptr_types,
8183 	[ARG_PTR_TO_CONST_STR]		= &const_str_ptr_types,
8184 	[ARG_PTR_TO_TIMER]		= &timer_types,
8185 	[ARG_KPTR_XCHG_DEST]		= &kptr_xchg_dest_types,
8186 	[ARG_PTR_TO_DYNPTR]		= &dynptr_types,
8187 };
8188 
bpf_diag_call_arg(struct bpf_verifier_env * env,u32 insn_idx,argno_t argno,const char * call_name,const char * reason,const char * suggestion)8189 static void bpf_diag_call_arg(struct bpf_verifier_env *env, u32 insn_idx, argno_t argno,
8190 			      const char *call_name, const char *reason,
8191 			      const char *suggestion)
8192 {
8193 	int arg = arg_from_argno(argno);
8194 	int regno = reg_from_argno(argno);
8195 	int stack_slot = -1;
8196 
8197 	if (arg < 0 && regno >= BPF_REG_1 && regno <= BPF_REG_5)
8198 		arg = regno;
8199 	if (arg > MAX_BPF_FUNC_REG_ARGS)
8200 		stack_slot = arg - MAX_BPF_FUNC_REG_ARGS - 1;
8201 
8202 	bpf_diag_call_type(env, insn_idx, arg, regno, stack_slot,
8203 			   call_name && *call_name ? call_name : "call",
8204 			   reg_arg_name(env, argno), reason, suggestion);
8205 }
8206 
bpf_diag_arg_name(struct bpf_verifier_env * env,argno_t argno)8207 static const char *bpf_diag_arg_name(struct bpf_verifier_env *env, argno_t argno)
8208 {
8209 	return bpf_diag_fmt(env, "%s", reg_arg_name(env, argno));
8210 }
8211 
bpf_diag_call_arg_fmt(struct bpf_verifier_env * env,u32 insn_idx,argno_t argno,const char * call_name,const char * suggestion,const char * fmt,...)8212 __printf(6, 7) static void bpf_diag_call_arg_fmt(struct bpf_verifier_env *env, u32 insn_idx,
8213 						 argno_t argno, const char *call_name,
8214 						 const char *suggestion, const char *fmt, ...)
8215 {
8216 	const char *reason;
8217 	va_list args;
8218 
8219 	va_start(args, fmt);
8220 	reason = bpf_diag_vfmt(env, fmt, args);
8221 	va_end(args);
8222 
8223 	bpf_diag_call_arg(env, insn_idx, argno, call_name, reason, suggestion);
8224 }
8225 
bpf_diag_expected_reg_types(struct bpf_verifier_env * env,const enum bpf_reg_type * types,int count)8226 static const char *bpf_diag_expected_reg_types(struct bpf_verifier_env *env,
8227 					       const enum bpf_reg_type *types, int count)
8228 {
8229 	size_t len = 0, size = 1;
8230 	char *buf;
8231 	int i;
8232 
8233 	for (i = 0; i < count; i++)
8234 		size += strlen(reg_type_str(env, types[i])) + (i ? 2 : 0);
8235 
8236 	buf = bpf_diag_fmt_buf(env, size);
8237 	if (!buf)
8238 		return "";
8239 
8240 	for (i = 0; i < count; i++)
8241 		len += scnprintf(buf + len, size - len, "%s%s", i ? ", " : "",
8242 				 reg_type_str(env, types[i]));
8243 	return buf;
8244 }
8245 
check_reg_type(struct bpf_verifier_env * env,struct bpf_reg_state * reg,argno_t argno,enum bpf_arg_type arg_type,const u32 * arg_btf_id,struct bpf_call_arg_meta * meta,const char * call_name)8246 static int check_reg_type(struct bpf_verifier_env *env, struct bpf_reg_state *reg, argno_t argno,
8247 			  enum bpf_arg_type arg_type, const u32 *arg_btf_id,
8248 			  struct bpf_call_arg_meta *meta, const char *call_name)
8249 {
8250 	enum bpf_reg_type expected, type = reg->type;
8251 	const struct bpf_reg_types *compatible;
8252 	const char *actual, *accepted;
8253 	int i, j, err;
8254 
8255 	compatible = compatible_reg_types[base_type(arg_type)];
8256 	if (!compatible) {
8257 		verifier_bug(env, "unsupported arg type %d", arg_type);
8258 		return -EFAULT;
8259 	}
8260 
8261 	/* ARG_PTR_TO_MEM + RDONLY is compatible with PTR_TO_MEM and PTR_TO_MEM + RDONLY,
8262 	 * but ARG_PTR_TO_MEM is compatible only with PTR_TO_MEM and NOT with PTR_TO_MEM + RDONLY
8263 	 *
8264 	 * Same for MAYBE_NULL:
8265 	 *
8266 	 * ARG_PTR_TO_MEM + MAYBE_NULL is compatible with PTR_TO_MEM and PTR_TO_MEM + MAYBE_NULL,
8267 	 * but ARG_PTR_TO_MEM is compatible only with PTR_TO_MEM but NOT with PTR_TO_MEM + MAYBE_NULL
8268 	 *
8269 	 * ARG_PTR_TO_MEM is compatible with PTR_TO_MEM that is tagged with a dynptr type.
8270 	 *
8271 	 * Therefore we fold these flags depending on the arg_type before comparison.
8272 	 */
8273 	if (arg_type & MEM_RDONLY)
8274 		type &= ~MEM_RDONLY;
8275 	if (arg_type & PTR_MAYBE_NULL)
8276 		type &= ~PTR_MAYBE_NULL;
8277 	if (base_type(arg_type) == ARG_PTR_TO_MEM)
8278 		type &= ~DYNPTR_TYPE_FLAG_MASK;
8279 
8280 	/* Local kptr types are allowed as the source argument of bpf_kptr_xchg */
8281 	if (meta->func_id == BPF_FUNC_kptr_xchg && type_is_alloc(type) && reg_from_argno(argno) == BPF_REG_2) {
8282 		type &= ~MEM_ALLOC;
8283 		type &= ~MEM_PERCPU;
8284 	}
8285 
8286 	for (i = 0; i < ARRAY_SIZE(compatible->types); i++) {
8287 		expected = compatible->types[i];
8288 		if (expected == NOT_INIT)
8289 			break;
8290 
8291 		if (type == expected)
8292 			goto found;
8293 	}
8294 
8295 	verbose(env, "%s type=%s expected=", reg_arg_name(env, argno), reg_type_str(env, reg->type));
8296 	for (j = 0; j + 1 < i; j++)
8297 		verbose(env, "%s, ", reg_type_str(env, compatible->types[j]));
8298 	verbose(env, "%s\n", reg_type_str(env, compatible->types[j]));
8299 	actual = bpf_diag_fmt(env, "%s", reg_type_str(env, reg->type));
8300 	accepted = bpf_diag_expected_reg_types(env, compatible->types, i);
8301 	bpf_diag_call_arg_fmt(env, env->insn_idx, argno, call_name,
8302 			      "Pass a value with one of the accepted pointer or scalar types for this call.",
8303 			      "it has type %s, but this argument accepts %s",
8304 			      actual, accepted);
8305 	return -EACCES;
8306 
8307 found:
8308 	if (base_type(reg->type) != PTR_TO_BTF_ID)
8309 		return 0;
8310 
8311 	if (compatible == &mem_types) {
8312 		if (!(arg_type & MEM_RDONLY)) {
8313 			verbose(env,
8314 				"%s() may write into memory pointed by %s type=%s\n",
8315 				func_id_name(meta->func_id),
8316 				reg_arg_name(env, argno), reg_type_str(env, reg->type));
8317 			return -EACCES;
8318 		}
8319 		return 0;
8320 	}
8321 
8322 	switch ((int)reg->type) {
8323 	case PTR_TO_BTF_ID:
8324 	case PTR_TO_BTF_ID | PTR_TRUSTED:
8325 	case PTR_TO_BTF_ID | PTR_TRUSTED | PTR_MAYBE_NULL:
8326 	case PTR_TO_BTF_ID | MEM_RCU:
8327 	case PTR_TO_BTF_ID | PTR_MAYBE_NULL:
8328 	case PTR_TO_BTF_ID | PTR_MAYBE_NULL | MEM_RCU:
8329 	{
8330 		/* For bpf_sk_release, it needs to match against first member
8331 		 * 'struct sock_common', hence make an exception for it. This
8332 		 * allows bpf_sk_release to work for multiple socket types.
8333 		 */
8334 		bool strict_type_match = arg_type_is_release(arg_type) &&
8335 					 meta->func_id != BPF_FUNC_sk_release;
8336 
8337 		if (type_may_be_null(reg->type) &&
8338 		    (!type_may_be_null(arg_type) || arg_type_is_release(arg_type))) {
8339 			verbose(env, "Possibly NULL pointer passed to helper %s\n",
8340 				reg_arg_name(env, argno));
8341 			bpf_diag_call_arg(
8342 				env, env->insn_idx, argno, call_name,
8343 				"the pointer may be NULL, but this call requires a non-NULL pointer",
8344 				"Add a NULL check and make the call only on the non-NULL path.");
8345 			return -EACCES;
8346 		}
8347 
8348 		if (!arg_btf_id) {
8349 			if (!compatible->btf_id) {
8350 				verifier_bug(env, "missing arg compatible BTF ID");
8351 				return -EFAULT;
8352 			}
8353 			arg_btf_id = compatible->btf_id;
8354 		}
8355 
8356 		if (meta->func_id == BPF_FUNC_kptr_xchg) {
8357 			if (map_kptr_match_type(env, meta->kptr_field, reg, reg_from_argno(argno)))
8358 				return -EACCES;
8359 		} else {
8360 			if (arg_btf_id == BPF_PTR_POISON) {
8361 				verbose(env, "verifier internal error:");
8362 				verbose(env, "%s has non-overwritten BPF_PTR_POISON type\n",
8363 					reg_arg_name(env, argno));
8364 				return -EACCES;
8365 			}
8366 
8367 			err = __check_ptr_off_reg(env, reg, argno, true);
8368 			if (err)
8369 				return err;
8370 
8371 			if (!btf_struct_ids_match(&env->log, reg->btf, reg->btf_id,
8372 						  reg->var_off.value, btf_vmlinux, *arg_btf_id,
8373 						  strict_type_match, !type_is_alloc(reg->type))) {
8374 				verbose(env, "%s is of type %s but %s is expected\n",
8375 					reg_arg_name(env, argno),
8376 					btf_type_name(reg->btf, reg->btf_id),
8377 					btf_type_name(btf_vmlinux, *arg_btf_id));
8378 				return -EACCES;
8379 			}
8380 		}
8381 		break;
8382 	}
8383 	case PTR_TO_BTF_ID | MEM_ALLOC:
8384 	case PTR_TO_BTF_ID | MEM_PERCPU | MEM_ALLOC:
8385 	case PTR_TO_BTF_ID | MEM_ALLOC | NON_OWN_REF:
8386 	case PTR_TO_BTF_ID | MEM_ALLOC | NON_OWN_REF | MEM_RCU:
8387 		if (meta->func_id != BPF_FUNC_spin_lock && meta->func_id != BPF_FUNC_spin_unlock &&
8388 		    meta->func_id != BPF_FUNC_kptr_xchg) {
8389 			verifier_bug(env, "unimplemented handling of MEM_ALLOC");
8390 			return -EFAULT;
8391 		}
8392 		/* Check if local kptr in src arg matches kptr in dst arg */
8393 		if (meta->func_id == BPF_FUNC_kptr_xchg) {
8394 			int regno = reg_from_argno(argno);
8395 
8396 			if (regno == BPF_REG_2 &&
8397 			    map_kptr_match_type(env, meta->kptr_field, reg, regno))
8398 				return -EACCES;
8399 		}
8400 		break;
8401 	case PTR_TO_BTF_ID | MEM_PERCPU:
8402 	case PTR_TO_BTF_ID | MEM_PERCPU | MEM_RCU:
8403 	case PTR_TO_BTF_ID | MEM_PERCPU | PTR_TRUSTED:
8404 		/* Handled by helper specific checks */
8405 		break;
8406 	default:
8407 		verifier_bug(env, "invalid PTR_TO_BTF_ID register for type match");
8408 		return -EFAULT;
8409 	}
8410 	return 0;
8411 }
8412 
8413 static struct btf_field *
reg_find_field_offset(const struct bpf_reg_state * reg,s32 off,u32 fields)8414 reg_find_field_offset(const struct bpf_reg_state *reg, s32 off, u32 fields)
8415 {
8416 	struct btf_field *field;
8417 	struct btf_record *rec;
8418 
8419 	rec = reg_btf_record(reg);
8420 	if (!rec)
8421 		return NULL;
8422 
8423 	field = btf_record_find(rec, off, fields);
8424 	if (!field)
8425 		return NULL;
8426 
8427 	return field;
8428 }
8429 
__check_func_arg_reg_off(struct bpf_verifier_env * env,const struct bpf_reg_state * reg,argno_t argno,enum bpf_arg_type arg_type,bool btf_id_fixed_off_ok)8430 static int __check_func_arg_reg_off(struct bpf_verifier_env *env,
8431 				    const struct bpf_reg_state *reg, argno_t argno,
8432 				    enum bpf_arg_type arg_type,
8433 				    bool btf_id_fixed_off_ok)
8434 {
8435 	u32 type = reg->type;
8436 
8437 	/* When referenced register is passed to release function, its fixed
8438 	 * offset must be 0.
8439 	 *
8440 	 * We will check arg_type_is_release reg has id when storing
8441 	 * meta->release_regno.
8442 	 */
8443 	if (arg_type_is_release(arg_type)) {
8444 		/* ARG_PTR_TO_DYNPTR with OBJ_RELEASE is a bit special, as it
8445 		 * may not directly point to the object being released, but to
8446 		 * dynptr pointing to such object, which might be at some offset
8447 		 * on the stack. In that case, we simply to fallback to the
8448 		 * default handling.
8449 		 */
8450 		if (arg_type_is_dynptr(arg_type) && type == PTR_TO_STACK)
8451 			return 0;
8452 
8453 		/* Doing check_ptr_off_reg check for the offset will catch this
8454 		 * because fixed_off_ok is false, but checking here allows us
8455 		 * to give the user a better error message.
8456 		 */
8457 		if (!tnum_is_const(reg->var_off) || reg->var_off.value != 0) {
8458 			verbose(env, "%s must have zero offset when passed to release func or trusted arg to kfunc\n",
8459 				reg_arg_name(env, argno));
8460 			return -EINVAL;
8461 		}
8462 	}
8463 
8464 	switch (type) {
8465 	/* Pointer types where both fixed and variable offset is explicitly allowed: */
8466 	case PTR_TO_STACK:
8467 	case PTR_TO_PACKET:
8468 	case PTR_TO_PACKET_META:
8469 	case PTR_TO_MAP_KEY:
8470 	case PTR_TO_MAP_VALUE:
8471 	case PTR_TO_MEM:
8472 	case PTR_TO_MEM | MEM_RDONLY:
8473 	case PTR_TO_MEM | MEM_RINGBUF:
8474 	case PTR_TO_BUF:
8475 	case PTR_TO_BUF | MEM_RDONLY:
8476 	case PTR_TO_ARENA:
8477 	case SCALAR_VALUE:
8478 		return 0;
8479 	/* All the rest must be rejected, except PTR_TO_BTF_ID which allows
8480 	 * fixed offset.
8481 	 */
8482 	case PTR_TO_BTF_ID:
8483 	case PTR_TO_BTF_ID | MEM_ALLOC:
8484 	case PTR_TO_BTF_ID | PTR_TRUSTED:
8485 	case PTR_TO_BTF_ID | MEM_RCU:
8486 	case PTR_TO_BTF_ID | MEM_ALLOC | NON_OWN_REF:
8487 	case PTR_TO_BTF_ID | MEM_ALLOC | NON_OWN_REF | MEM_RCU:
8488 		/* When referenced PTR_TO_BTF_ID is passed to release function,
8489 		 * its fixed offset must be 0. In the other cases, fixed offset
8490 		 * can be non-zero unless the caller requires otherwise.
8491 		 * var_off always must be 0 for PTR_TO_BTF_ID, hence we still
8492 		 * need to do checks instead of returning.
8493 		 */
8494 		return __check_ptr_off_reg(env, reg, argno, btf_id_fixed_off_ok);
8495 	case PTR_TO_CTX:
8496 		/*
8497 		 * Allow fixed and variable offsets for syscall context, but
8498 		 * only when the argument is passed as memory, not ctx,
8499 		 * otherwise we may get modified ctx in tail called programs and
8500 		 * global subprogs (that may act as extension prog hooks).
8501 		 */
8502 		if (arg_type != ARG_PTR_TO_CTX && is_var_ctx_off_allowed(env->prog))
8503 			return 0;
8504 		fallthrough;
8505 	default:
8506 		return __check_ptr_off_reg(env, reg, argno, false);
8507 	}
8508 }
8509 
check_func_arg_reg_off(struct bpf_verifier_env * env,const struct bpf_reg_state * reg,argno_t argno,enum bpf_arg_type arg_type)8510 static int check_func_arg_reg_off(struct bpf_verifier_env *env,
8511 				  const struct bpf_reg_state *reg, argno_t argno,
8512 				  enum bpf_arg_type arg_type)
8513 {
8514 	return __check_func_arg_reg_off(env, reg, argno, arg_type, true);
8515 }
8516 
check_arg_const_str(struct bpf_verifier_env * env,struct bpf_reg_state * reg,argno_t argno)8517 static int check_arg_const_str(struct bpf_verifier_env *env,
8518 			       struct bpf_reg_state *reg, argno_t argno)
8519 {
8520 	struct bpf_map *map = reg->map_ptr;
8521 	int err;
8522 	int map_off;
8523 	u64 map_addr;
8524 	char *str_ptr;
8525 
8526 	if (reg->type != PTR_TO_MAP_VALUE)
8527 		return -EINVAL;
8528 
8529 	if (map->map_type == BPF_MAP_TYPE_INSN_ARRAY) {
8530 		verbose(env, "%s points to insn_array map which cannot be used as const string\n",
8531 			reg_arg_name(env, argno));
8532 		return -EACCES;
8533 	}
8534 
8535 	if (map->map_type == BPF_MAP_TYPE_PERCPU_ARRAY) {
8536 		verbose(env, "%s points to percpu_array map which cannot be used as const string\n",
8537 			reg_arg_name(env, argno));
8538 		return -EACCES;
8539 	}
8540 
8541 	if (!bpf_map_is_rdonly(map)) {
8542 		verbose(env, "%s does not point to a readonly map'\n", reg_arg_name(env, argno));
8543 		return -EACCES;
8544 	}
8545 
8546 	if (!tnum_is_const(reg->var_off)) {
8547 		verbose(env, "%s is not a constant address'\n", reg_arg_name(env, argno));
8548 		return -EACCES;
8549 	}
8550 
8551 	if (!map->ops->map_direct_value_addr) {
8552 		verbose(env, "no direct value access support for this map type\n");
8553 		return -EACCES;
8554 	}
8555 
8556 	err = check_map_access(env, reg, argno, 0,
8557 			       map->value_size - reg->var_off.value, false,
8558 			       ACCESS_HELPER);
8559 	if (err)
8560 		return err;
8561 
8562 	map_off = reg->var_off.value;
8563 	err = map->ops->map_direct_value_addr(map, &map_addr, map_off);
8564 	if (err) {
8565 		verbose(env, "direct value access on string failed\n");
8566 		return err;
8567 	}
8568 
8569 	str_ptr = (char *)(long)(map_addr);
8570 	if (!strnchr(str_ptr + map_off, map->value_size - map_off, 0)) {
8571 		verbose(env, "string is not zero-terminated\n");
8572 		return -EINVAL;
8573 	}
8574 	return 0;
8575 }
8576 
8577 /* Returns constant key value in `value` if possible, else negative error */
get_constant_map_key(struct bpf_verifier_env * env,struct bpf_reg_state * key,u32 key_size,s64 * value)8578 static int get_constant_map_key(struct bpf_verifier_env *env,
8579 				struct bpf_reg_state *key,
8580 				u32 key_size,
8581 				s64 *value)
8582 {
8583 	struct bpf_func_state *state = bpf_func(env, key);
8584 	struct bpf_reg_state *reg;
8585 	int slot, spi, off;
8586 	int spill_size = 0;
8587 	int zero_size = 0;
8588 	int stack_off;
8589 	int i, err;
8590 	u8 *stype;
8591 
8592 	if (!env->bpf_capable)
8593 		return -EOPNOTSUPP;
8594 	if (key->type != PTR_TO_STACK)
8595 		return -EOPNOTSUPP;
8596 	if (!tnum_is_const(key->var_off))
8597 		return -EOPNOTSUPP;
8598 
8599 	stack_off = key->var_off.value;
8600 	slot = -stack_off - 1;
8601 	spi = slot / BPF_REG_SIZE;
8602 	off = slot % BPF_REG_SIZE;
8603 	stype = state->stack[spi].slot_type;
8604 
8605 	/* First handle precisely tracked STACK_ZERO */
8606 	for (i = off; i >= 0 && stype[i] == STACK_ZERO; i--)
8607 		zero_size++;
8608 	if (zero_size >= key_size) {
8609 		*value = 0;
8610 		return 0;
8611 	}
8612 
8613 	/* Check that stack contains a scalar spill of expected size */
8614 	if (!bpf_is_spilled_scalar_reg(&state->stack[spi]))
8615 		return -EOPNOTSUPP;
8616 	for (i = off; i >= 0 && stype[i] == STACK_SPILL; i--)
8617 		spill_size++;
8618 	if (spill_size != key_size)
8619 		return -EOPNOTSUPP;
8620 
8621 	reg = &state->stack[spi].spilled_ptr;
8622 	if (!tnum_is_const(reg->var_off))
8623 		/* Stack value not statically known */
8624 		return -EOPNOTSUPP;
8625 
8626 	/* We are relying on a constant value. So mark as precise
8627 	 * to prevent pruning on it.
8628 	 */
8629 	bpf_bt_set_frame_slot(&env->bt, key->frameno, spi);
8630 	err = mark_chain_precision_batch(env, env->cur_state);
8631 	if (err < 0)
8632 		return err;
8633 
8634 	*value = reg->var_off.value;
8635 	return 0;
8636 }
8637 
8638 static bool can_elide_value_nullness(const struct bpf_map *map);
8639 
process_map_ptr_arg(struct bpf_verifier_env * env,struct bpf_reg_state * reg,argno_t argno,struct bpf_call_arg_meta * meta)8640 static int process_map_ptr_arg(struct bpf_verifier_env *env, struct bpf_reg_state *reg,
8641 			       argno_t argno, struct bpf_call_arg_meta *meta)
8642 {
8643 	/* Use map_uid (which is unique id of inner map) to reject:
8644 	 * inner_map1 = bpf_map_lookup_elem(outer_map, key1)
8645 	 * inner_map2 = bpf_map_lookup_elem(outer_map, key2)
8646 	 * if (inner_map1 && inner_map2) {
8647 	 *     timer = bpf_map_lookup_elem(inner_map1);
8648 	 *     if (timer)
8649 	 *         // mismatch would have been allowed
8650 	 *         bpf_timer_init(timer, inner_map2);
8651 	 * }
8652 	 *
8653 	 * Comparing map_ptr is enough to distinguish normal and outer maps.
8654 	 */
8655 	if (meta->map.ptr &&
8656 	    (meta->map.ptr != reg->map_ptr || meta->map.uid != reg->map_uid)) {
8657 		argno_t obj_argno = argno_from_reg(reg_from_argno(argno) - 1);
8658 		struct btf_record *rec = meta->map.ptr->record;
8659 		const char *obj_name = "workqueue";
8660 
8661 		if (rec->timer_off >= 0)
8662 			obj_name = "timer";
8663 		else if (rec->task_work_off >= 0)
8664 			obj_name = "bpf_task_work";
8665 
8666 		verbose(env, "%s pointer in %s map_uid=%d ",
8667 			obj_name, reg_arg_name(env, obj_argno), meta->map.uid);
8668 		verbose(env, "doesn't match map pointer in %s map_uid=%d\n",
8669 			reg_arg_name(env, argno), reg->map_uid);
8670 		return -EINVAL;
8671 	}
8672 
8673 	meta->map.ptr = reg->map_ptr;
8674 	meta->map.uid = reg->map_uid;
8675 	return 0;
8676 }
8677 
check_func_arg(struct bpf_verifier_env * env,u32 arg,struct bpf_call_arg_meta * meta,int insn_idx)8678 static int check_func_arg(struct bpf_verifier_env *env, u32 arg,
8679 			  struct bpf_call_arg_meta *meta,
8680 			  int insn_idx)
8681 {
8682 	const struct bpf_func_proto *fn = meta->fn;
8683 	u32 regno = BPF_REG_1 + arg;
8684 	struct bpf_reg_state *reg = reg_state(env, regno);
8685 	enum bpf_arg_type arg_type = fn->arg_type[arg];
8686 	argno_t argno = argno_from_reg(regno);
8687 	enum bpf_reg_type type = reg->type;
8688 	u32 *arg_btf_id = NULL;
8689 	u32 key_size;
8690 	int err = 0;
8691 
8692 	if (arg_type == ARG_DONTCARE)
8693 		return 0;
8694 
8695 	err = check_reg_arg(env, regno, SRC_OP);
8696 	if (err)
8697 		return err;
8698 
8699 	if (arg_type == ARG_ANYTHING) {
8700 		if (is_pointer_value(env, regno)) {
8701 			verbose(env, "R%d leaks addr into helper function\n",
8702 				regno);
8703 			return -EACCES;
8704 		}
8705 		return 0;
8706 	}
8707 
8708 	if (type_is_pkt_pointer(type) &&
8709 	    !may_access_direct_pkt_data(env, fn, BPF_READ)) {
8710 		verbose(env, "helper access to the packet is not allowed\n");
8711 		return -EACCES;
8712 	}
8713 
8714 	if (base_type(arg_type) == ARG_PTR_TO_MAP_VALUE) {
8715 		err = resolve_map_arg_type(env, meta, &arg_type);
8716 		if (err)
8717 			return err;
8718 	}
8719 
8720 	if (bpf_register_is_null(reg) && type_may_be_null(arg_type))
8721 		/* A NULL register has a SCALAR_VALUE type, so skip
8722 		 * type checking.
8723 		 */
8724 		goto skip_type_check;
8725 
8726 	/* arg_btf_id and arg_size are in a union. */
8727 	if (base_type(arg_type) == ARG_PTR_TO_BTF_ID ||
8728 	    base_type(arg_type) == ARG_PTR_TO_SPIN_LOCK)
8729 		arg_btf_id = fn->arg_btf_id[arg];
8730 
8731 	err = check_reg_type(env, reg, argno, arg_type, arg_btf_id, meta,
8732 			     func_id_name(meta->func_id));
8733 	if (err)
8734 		return err;
8735 
8736 	err = check_func_arg_reg_off(env, reg, argno, arg_type);
8737 	if (err)
8738 		return err;
8739 
8740 skip_type_check:
8741 	if (arg_type_is_release(arg_type) && !arg_type_is_dynptr(arg_type) &&
8742 	    !reg_is_referenced(env, reg) && !bpf_register_is_null(reg)) {
8743 		verbose(env, "release helper %s expects referenced PTR_TO_BTF_ID passed to %s\n",
8744 			func_id_name(meta->func_id), reg_arg_name(env, argno));
8745 		bpf_diag_call_arg(
8746 			env, insn_idx, argno, func_id_name(meta->func_id),
8747 			"release helpers require a value that owns a live resource returned by a matching acquire helper",
8748 			"Pass the resource-owning pointer returned by the matching acquire helper, and avoid calling the release helper after ownership has already been transferred or released.");
8749 		return -EINVAL;
8750 	}
8751 
8752 	if (reg_is_referenced(env, reg))
8753 		update_ref_obj(&meta->ref_obj, reg);
8754 
8755 	switch (base_type(arg_type)) {
8756 	case ARG_CONST_MAP_PTR:
8757 		/* bpf_map_xxx(map_ptr) call: remember that map_ptr */
8758 		err = process_map_ptr_arg(env, reg, argno, meta);
8759 		if (err)
8760 			return err;
8761 		break;
8762 	case ARG_PTR_TO_MAP_KEY:
8763 		/* bpf_map_xxx(..., map_ptr, ..., key) call:
8764 		 * check that [key, key + map->key_size) are within
8765 		 * stack limits and initialized
8766 		 */
8767 		if (!meta->map.ptr) {
8768 			/* in function declaration map_ptr must come before
8769 			 * map_key, so that it's verified and known before
8770 			 * we have to check map_key here. Otherwise it means
8771 			 * that kernel subsystem misconfigured verifier
8772 			 */
8773 			verifier_bug(env, "invalid map_ptr to access map->key");
8774 			return -EFAULT;
8775 		}
8776 		key_size = meta->map.ptr->key_size;
8777 		err = check_helper_mem_access(env, reg, argno, key_size, BPF_READ, false, NULL,
8778 					      NULL);
8779 		if (err)
8780 			return err;
8781 		if (can_elide_value_nullness(meta->map.ptr)) {
8782 			err = get_constant_map_key(env, reg, key_size, &meta->const_map_key);
8783 			if (err < 0) {
8784 				meta->const_map_key = -1;
8785 				if (err == -EOPNOTSUPP)
8786 					err = 0;
8787 				else
8788 					return err;
8789 			}
8790 		}
8791 		break;
8792 	case ARG_PTR_TO_MAP_VALUE:
8793 		if (type_may_be_null(arg_type) && bpf_register_is_null(reg))
8794 			return 0;
8795 
8796 		/* bpf_map_xxx(..., map_ptr, ..., value) call:
8797 		 * check [value, value + map->value_size) validity
8798 		 */
8799 		if (!meta->map.ptr) {
8800 			/* kernel subsystem misconfigured verifier */
8801 			verifier_bug(env, "invalid map_ptr to access map->value");
8802 			return -EFAULT;
8803 		}
8804 
8805 		/*
8806 		 * Disable raw mode for bpf_map_peek_elem() on a bloom filter. The helper reads
8807 		 * the value buffer as an input rather than filling it.
8808 		 */
8809 		if (meta->func_id == BPF_FUNC_map_peek_elem &&
8810 		    meta->map.ptr->map_type == BPF_MAP_TYPE_BLOOM_FILTER)
8811 			meta->arg_raw_mem.regno = 0;
8812 
8813 		err = check_helper_mem_access(env, reg, argno, meta->map.ptr->value_size,
8814 					      arg_type & MEM_WRITE ? BPF_WRITE : BPF_READ,
8815 					      false, meta, NULL);
8816 		break;
8817 	case ARG_PTR_TO_PERCPU_BTF_ID:
8818 		if (!reg->btf_id) {
8819 			verbose(env, "Helper has invalid btf_id in R%d\n", regno);
8820 			return -EACCES;
8821 		}
8822 		meta->ret_btf = reg->btf;
8823 		meta->ret_btf_id = reg->btf_id;
8824 		break;
8825 	case ARG_PTR_TO_SPIN_LOCK:
8826 		if (in_rbtree_lock_required_cb(env)) {
8827 			verbose(env, "can't spin_{lock,unlock} in rbtree cb\n");
8828 			return -EACCES;
8829 		}
8830 		if (meta->func_id == BPF_FUNC_spin_lock) {
8831 			err = process_spin_lock(env, reg, argno, PROCESS_SPIN_LOCK);
8832 			if (err)
8833 				return err;
8834 		} else if (meta->func_id == BPF_FUNC_spin_unlock) {
8835 			err = process_spin_lock(env, reg, argno, 0);
8836 			if (err)
8837 				return err;
8838 		} else {
8839 			verifier_bug(env, "spin lock arg on unexpected helper");
8840 			return -EFAULT;
8841 		}
8842 		break;
8843 	case ARG_PTR_TO_TIMER:
8844 		err = process_timer_func(env, reg, argno, &meta->map);
8845 		if (err)
8846 			return err;
8847 		break;
8848 	case ARG_PTR_TO_FUNC:
8849 		meta->subprogno = reg->subprogno;
8850 		break;
8851 	case ARG_PTR_TO_MEM:
8852 		/* The access to this pointer is only checked when we hit the
8853 		 * next is_mem_size argument below.
8854 		 */
8855 		if (arg_type & MEM_FIXED_SIZE) {
8856 			err = check_mem_reg(env, reg, argno_from_reg(regno), fn->arg_size[arg],
8857 					    arg_type & MEM_WRITE ? BPF_WRITE : BPF_READ, meta, NULL);
8858 			if (err)
8859 				return err;
8860 			if (arg_type & MEM_ALIGNED)
8861 				err = check_ptr_alignment(env, reg, 0, fn->arg_size[arg], true);
8862 		}
8863 		break;
8864 	case ARG_MEM_SIZE:
8865 		err = check_mem_size_reg(env, reg_state(env, regno - 1), reg,
8866 					 argno_from_reg(regno - 1), argno,
8867 					 fn->arg_type[arg - 1] & MEM_WRITE ? BPF_WRITE : BPF_READ,
8868 					 false, meta, NULL);
8869 		break;
8870 	case ARG_MEM_SIZE_OR_ZERO:
8871 		err = check_mem_size_reg(env, reg_state(env, regno - 1), reg,
8872 					 argno_from_reg(regno - 1), argno,
8873 					 fn->arg_type[arg - 1] & MEM_WRITE ? BPF_WRITE : BPF_READ,
8874 					 true, meta, NULL);
8875 		break;
8876 	case ARG_PTR_TO_DYNPTR:
8877 		err = process_dynptr_func(env, reg, argno, insn_idx, func_id_name(meta->func_id),
8878 					  arg_type, &meta->ref_obj, &meta->dynptr);
8879 		if (err)
8880 			return err;
8881 		break;
8882 	case ARG_CONST_ALLOC_SIZE_OR_ZERO:
8883 		err = process_const_alloc_mem_size(env, reg, argno, &meta->ret_mem);
8884 		if (err)
8885 			return err;
8886 		break;
8887 	case ARG_PTR_TO_CONST_STR:
8888 	{
8889 		err = check_arg_const_str(env, reg, argno);
8890 		if (err)
8891 			return err;
8892 		break;
8893 	}
8894 	case ARG_KPTR_XCHG_DEST:
8895 		err = process_kptr_func(env, regno, meta);
8896 		if (err)
8897 			return err;
8898 		break;
8899 	}
8900 
8901 	return err;
8902 }
8903 
may_update_sockmap(struct bpf_verifier_env * env,int func_id)8904 static bool may_update_sockmap(struct bpf_verifier_env *env, int func_id)
8905 {
8906 	enum bpf_attach_type eatype = env->prog->expected_attach_type;
8907 	enum bpf_prog_type type = resolve_prog_type(env->prog);
8908 
8909 	if (func_id != BPF_FUNC_map_update_elem &&
8910 	    func_id != BPF_FUNC_map_delete_elem)
8911 		return false;
8912 
8913 	/* It's not possible to get access to a locked struct sock in these
8914 	 * contexts, so updating is safe.
8915 	 */
8916 	switch (type) {
8917 	case BPF_PROG_TYPE_TRACING:
8918 		if (eatype == BPF_TRACE_ITER)
8919 			return true;
8920 		break;
8921 	case BPF_PROG_TYPE_SOCK_OPS:
8922 		/* map_update allowed only via dedicated helpers with event type checks */
8923 		if (func_id == BPF_FUNC_map_delete_elem)
8924 			return true;
8925 		break;
8926 	case BPF_PROG_TYPE_SK_REUSEPORT:
8927 	case BPF_PROG_TYPE_SK_LOOKUP:
8928 		return true;
8929 	default:
8930 		break;
8931 	}
8932 
8933 	verbose(env, "cannot update sockmap in this context\n");
8934 	return false;
8935 }
8936 
bpf_allow_tail_call_in_subprogs(struct bpf_verifier_env * env)8937 bool bpf_allow_tail_call_in_subprogs(struct bpf_verifier_env *env)
8938 {
8939 	return env->prog->jit_requested &&
8940 	       bpf_jit_supports_subprog_tailcalls();
8941 }
8942 
check_map_func_compatibility(struct bpf_verifier_env * env,struct bpf_map * map,int func_id)8943 static int check_map_func_compatibility(struct bpf_verifier_env *env,
8944 					struct bpf_map *map, int func_id)
8945 {
8946 	if (!map)
8947 		return 0;
8948 
8949 	/* We need a two way check, first is from map perspective ... */
8950 	switch (map->map_type) {
8951 	case BPF_MAP_TYPE_PROG_ARRAY:
8952 		if (func_id != BPF_FUNC_tail_call)
8953 			goto error;
8954 		break;
8955 	case BPF_MAP_TYPE_PERF_EVENT_ARRAY:
8956 		if (func_id != BPF_FUNC_perf_event_read &&
8957 		    func_id != BPF_FUNC_perf_event_output &&
8958 		    func_id != BPF_FUNC_skb_output &&
8959 		    func_id != BPF_FUNC_perf_event_read_value &&
8960 		    func_id != BPF_FUNC_xdp_output)
8961 			goto error;
8962 		break;
8963 	case BPF_MAP_TYPE_RINGBUF:
8964 		if (func_id != BPF_FUNC_ringbuf_output &&
8965 		    func_id != BPF_FUNC_ringbuf_reserve &&
8966 		    func_id != BPF_FUNC_ringbuf_query &&
8967 		    func_id != BPF_FUNC_ringbuf_reserve_dynptr &&
8968 		    func_id != BPF_FUNC_ringbuf_submit_dynptr &&
8969 		    func_id != BPF_FUNC_ringbuf_discard_dynptr)
8970 			goto error;
8971 		break;
8972 	case BPF_MAP_TYPE_USER_RINGBUF:
8973 		if (func_id != BPF_FUNC_user_ringbuf_drain)
8974 			goto error;
8975 		break;
8976 	case BPF_MAP_TYPE_STACK_TRACE:
8977 		if (func_id != BPF_FUNC_get_stackid)
8978 			goto error;
8979 		break;
8980 	case BPF_MAP_TYPE_CGROUP_ARRAY:
8981 		if (func_id != BPF_FUNC_skb_under_cgroup &&
8982 		    func_id != BPF_FUNC_current_task_under_cgroup)
8983 			goto error;
8984 		break;
8985 	case BPF_MAP_TYPE_CGROUP_STORAGE:
8986 	case BPF_MAP_TYPE_PERCPU_CGROUP_STORAGE:
8987 		if (func_id != BPF_FUNC_get_local_storage)
8988 			goto error;
8989 		break;
8990 	case BPF_MAP_TYPE_DEVMAP:
8991 	case BPF_MAP_TYPE_DEVMAP_HASH:
8992 		if (func_id != BPF_FUNC_redirect_map &&
8993 		    func_id != BPF_FUNC_map_lookup_elem)
8994 			goto error;
8995 		break;
8996 	/* Restrict bpf side of cpumap and xskmap, open when use-cases
8997 	 * appear.
8998 	 */
8999 	case BPF_MAP_TYPE_CPUMAP:
9000 		if (func_id != BPF_FUNC_redirect_map)
9001 			goto error;
9002 		break;
9003 	case BPF_MAP_TYPE_XSKMAP:
9004 		if (func_id != BPF_FUNC_redirect_map &&
9005 		    func_id != BPF_FUNC_map_lookup_elem)
9006 			goto error;
9007 		break;
9008 	case BPF_MAP_TYPE_ARRAY_OF_MAPS:
9009 	case BPF_MAP_TYPE_HASH_OF_MAPS:
9010 		if (func_id != BPF_FUNC_map_lookup_elem)
9011 			goto error;
9012 		break;
9013 	case BPF_MAP_TYPE_SOCKMAP:
9014 		if (func_id != BPF_FUNC_sk_redirect_map &&
9015 		    func_id != BPF_FUNC_sock_map_update &&
9016 		    func_id != BPF_FUNC_msg_redirect_map &&
9017 		    func_id != BPF_FUNC_sk_select_reuseport &&
9018 		    func_id != BPF_FUNC_map_lookup_elem &&
9019 		    !may_update_sockmap(env, func_id))
9020 			goto error;
9021 		break;
9022 	case BPF_MAP_TYPE_SOCKHASH:
9023 		if (func_id != BPF_FUNC_sk_redirect_hash &&
9024 		    func_id != BPF_FUNC_sock_hash_update &&
9025 		    func_id != BPF_FUNC_msg_redirect_hash &&
9026 		    func_id != BPF_FUNC_sk_select_reuseport &&
9027 		    func_id != BPF_FUNC_map_lookup_elem &&
9028 		    !may_update_sockmap(env, func_id))
9029 			goto error;
9030 		break;
9031 	case BPF_MAP_TYPE_REUSEPORT_SOCKARRAY:
9032 		if (func_id != BPF_FUNC_sk_select_reuseport)
9033 			goto error;
9034 		break;
9035 	case BPF_MAP_TYPE_QUEUE:
9036 	case BPF_MAP_TYPE_STACK:
9037 		if (func_id != BPF_FUNC_map_peek_elem &&
9038 		    func_id != BPF_FUNC_map_pop_elem &&
9039 		    func_id != BPF_FUNC_map_push_elem)
9040 			goto error;
9041 		break;
9042 	case BPF_MAP_TYPE_SK_STORAGE:
9043 		if (func_id != BPF_FUNC_sk_storage_get &&
9044 		    func_id != BPF_FUNC_sk_storage_delete &&
9045 		    func_id != BPF_FUNC_kptr_xchg)
9046 			goto error;
9047 		break;
9048 	case BPF_MAP_TYPE_INODE_STORAGE:
9049 		if (func_id != BPF_FUNC_inode_storage_get &&
9050 		    func_id != BPF_FUNC_inode_storage_delete &&
9051 		    func_id != BPF_FUNC_kptr_xchg)
9052 			goto error;
9053 		break;
9054 	case BPF_MAP_TYPE_TASK_STORAGE:
9055 		if (func_id != BPF_FUNC_task_storage_get &&
9056 		    func_id != BPF_FUNC_task_storage_delete &&
9057 		    func_id != BPF_FUNC_kptr_xchg)
9058 			goto error;
9059 		break;
9060 	case BPF_MAP_TYPE_CGRP_STORAGE:
9061 		if (func_id != BPF_FUNC_cgrp_storage_get &&
9062 		    func_id != BPF_FUNC_cgrp_storage_delete &&
9063 		    func_id != BPF_FUNC_kptr_xchg)
9064 			goto error;
9065 		break;
9066 	case BPF_MAP_TYPE_BLOOM_FILTER:
9067 		if (func_id != BPF_FUNC_map_peek_elem &&
9068 		    func_id != BPF_FUNC_map_push_elem)
9069 			goto error;
9070 		break;
9071 	case BPF_MAP_TYPE_INSN_ARRAY:
9072 		goto error;
9073 	default:
9074 		break;
9075 	}
9076 
9077 	/* ... and second from the function itself. */
9078 	switch (func_id) {
9079 	case BPF_FUNC_tail_call:
9080 		if (map->map_type != BPF_MAP_TYPE_PROG_ARRAY)
9081 			goto error;
9082 		if (env->subprog_cnt > 1 && !bpf_allow_tail_call_in_subprogs(env)) {
9083 			verbose(env, "mixing of tail_calls and bpf-to-bpf calls is not supported\n");
9084 			return -EINVAL;
9085 		}
9086 		break;
9087 	case BPF_FUNC_perf_event_read:
9088 	case BPF_FUNC_perf_event_output:
9089 	case BPF_FUNC_perf_event_read_value:
9090 	case BPF_FUNC_skb_output:
9091 	case BPF_FUNC_xdp_output:
9092 		if (map->map_type != BPF_MAP_TYPE_PERF_EVENT_ARRAY)
9093 			goto error;
9094 		break;
9095 	case BPF_FUNC_ringbuf_output:
9096 	case BPF_FUNC_ringbuf_reserve:
9097 	case BPF_FUNC_ringbuf_query:
9098 	case BPF_FUNC_ringbuf_reserve_dynptr:
9099 	case BPF_FUNC_ringbuf_submit_dynptr:
9100 	case BPF_FUNC_ringbuf_discard_dynptr:
9101 		if (map->map_type != BPF_MAP_TYPE_RINGBUF)
9102 			goto error;
9103 		break;
9104 	case BPF_FUNC_user_ringbuf_drain:
9105 		if (map->map_type != BPF_MAP_TYPE_USER_RINGBUF)
9106 			goto error;
9107 		break;
9108 	case BPF_FUNC_get_stackid:
9109 		if (map->map_type != BPF_MAP_TYPE_STACK_TRACE)
9110 			goto error;
9111 		break;
9112 	case BPF_FUNC_current_task_under_cgroup:
9113 	case BPF_FUNC_skb_under_cgroup:
9114 		if (map->map_type != BPF_MAP_TYPE_CGROUP_ARRAY)
9115 			goto error;
9116 		break;
9117 	case BPF_FUNC_redirect_map:
9118 		if (map->map_type != BPF_MAP_TYPE_DEVMAP &&
9119 		    map->map_type != BPF_MAP_TYPE_DEVMAP_HASH &&
9120 		    map->map_type != BPF_MAP_TYPE_CPUMAP &&
9121 		    map->map_type != BPF_MAP_TYPE_XSKMAP)
9122 			goto error;
9123 		break;
9124 	case BPF_FUNC_sk_redirect_map:
9125 	case BPF_FUNC_msg_redirect_map:
9126 	case BPF_FUNC_sock_map_update:
9127 		if (map->map_type != BPF_MAP_TYPE_SOCKMAP)
9128 			goto error;
9129 		break;
9130 	case BPF_FUNC_sk_redirect_hash:
9131 	case BPF_FUNC_msg_redirect_hash:
9132 	case BPF_FUNC_sock_hash_update:
9133 		if (map->map_type != BPF_MAP_TYPE_SOCKHASH)
9134 			goto error;
9135 		break;
9136 	case BPF_FUNC_get_local_storage:
9137 		if (map->map_type != BPF_MAP_TYPE_CGROUP_STORAGE &&
9138 		    map->map_type != BPF_MAP_TYPE_PERCPU_CGROUP_STORAGE)
9139 			goto error;
9140 		break;
9141 	case BPF_FUNC_sk_select_reuseport:
9142 		if (map->map_type != BPF_MAP_TYPE_REUSEPORT_SOCKARRAY &&
9143 		    map->map_type != BPF_MAP_TYPE_SOCKMAP &&
9144 		    map->map_type != BPF_MAP_TYPE_SOCKHASH)
9145 			goto error;
9146 		break;
9147 	case BPF_FUNC_map_pop_elem:
9148 		if (map->map_type != BPF_MAP_TYPE_QUEUE &&
9149 		    map->map_type != BPF_MAP_TYPE_STACK)
9150 			goto error;
9151 		break;
9152 	case BPF_FUNC_map_peek_elem:
9153 	case BPF_FUNC_map_push_elem:
9154 		if (map->map_type != BPF_MAP_TYPE_QUEUE &&
9155 		    map->map_type != BPF_MAP_TYPE_STACK &&
9156 		    map->map_type != BPF_MAP_TYPE_BLOOM_FILTER)
9157 			goto error;
9158 		break;
9159 	case BPF_FUNC_map_lookup_percpu_elem:
9160 		if (map->map_type != BPF_MAP_TYPE_PERCPU_ARRAY &&
9161 		    map->map_type != BPF_MAP_TYPE_PERCPU_HASH &&
9162 		    map->map_type != BPF_MAP_TYPE_LRU_PERCPU_HASH)
9163 			goto error;
9164 		break;
9165 	case BPF_FUNC_sk_storage_get:
9166 	case BPF_FUNC_sk_storage_delete:
9167 		if (map->map_type != BPF_MAP_TYPE_SK_STORAGE)
9168 			goto error;
9169 		break;
9170 	case BPF_FUNC_inode_storage_get:
9171 	case BPF_FUNC_inode_storage_delete:
9172 		if (map->map_type != BPF_MAP_TYPE_INODE_STORAGE)
9173 			goto error;
9174 		break;
9175 	case BPF_FUNC_task_storage_get:
9176 	case BPF_FUNC_task_storage_delete:
9177 		if (map->map_type != BPF_MAP_TYPE_TASK_STORAGE)
9178 			goto error;
9179 		break;
9180 	case BPF_FUNC_cgrp_storage_get:
9181 	case BPF_FUNC_cgrp_storage_delete:
9182 		if (map->map_type != BPF_MAP_TYPE_CGRP_STORAGE)
9183 			goto error;
9184 		break;
9185 	default:
9186 		break;
9187 	}
9188 
9189 	return 0;
9190 error:
9191 	verbose(env, "cannot pass map_type %d into func %s#%d\n",
9192 		map->map_type, func_id_name(func_id), func_id);
9193 	return -EINVAL;
9194 }
9195 
check_raw_mode_ok(const struct bpf_func_proto * fn,struct bpf_call_arg_meta * meta)9196 static bool check_raw_mode_ok(const struct bpf_func_proto *fn, struct bpf_call_arg_meta *meta)
9197 {
9198 	int i;
9199 
9200 	for (i = 0; i < ARRAY_SIZE(fn->arg_type); i++) {
9201 		if (fn->arg_type[i] == ARG_DONTCARE)
9202 			break;
9203 		if (!arg_type_is_raw_mem(fn->arg_type[i]))
9204 			continue;
9205 		if (meta->arg_raw_mem.regno)
9206 			return false;
9207 		meta->arg_raw_mem.regno = i + 1;
9208 	}
9209 
9210 	return true;
9211 }
9212 
check_args_pair_invalid(const struct bpf_func_proto * fn,int arg)9213 static bool check_args_pair_invalid(const struct bpf_func_proto *fn, int arg)
9214 {
9215 	bool is_fixed = fn->arg_type[arg] & MEM_FIXED_SIZE;
9216 	bool has_size = fn->arg_size[arg] != 0;
9217 	bool is_next_size = false;
9218 
9219 	if (arg + 1 < ARRAY_SIZE(fn->arg_type))
9220 		is_next_size = arg_type_is_mem_size(fn->arg_type[arg + 1]);
9221 
9222 	if (base_type(fn->arg_type[arg]) != ARG_PTR_TO_MEM)
9223 		return is_next_size;
9224 
9225 	return has_size == is_next_size || is_next_size == is_fixed;
9226 }
9227 
check_arg_pair_ok(const struct bpf_func_proto * fn)9228 static bool check_arg_pair_ok(const struct bpf_func_proto *fn)
9229 {
9230 	/* bpf_xxx(..., buf, len) call will access 'len'
9231 	 * bytes from memory 'buf'. Both arg types need
9232 	 * to be paired, so make sure there's no buggy
9233 	 * helper function specification.
9234 	 */
9235 	if (arg_type_is_mem_size(fn->arg1_type) ||
9236 	    check_args_pair_invalid(fn, 0) ||
9237 	    check_args_pair_invalid(fn, 1) ||
9238 	    check_args_pair_invalid(fn, 2) ||
9239 	    check_args_pair_invalid(fn, 3) ||
9240 	    check_args_pair_invalid(fn, 4))
9241 		return false;
9242 
9243 	return true;
9244 }
9245 
check_btf_id_ok(const struct bpf_func_proto * fn)9246 static bool check_btf_id_ok(const struct bpf_func_proto *fn)
9247 {
9248 	int i;
9249 
9250 	for (i = 0; i < ARRAY_SIZE(fn->arg_type); i++) {
9251 		if (fn->arg_type[i] == ARG_DONTCARE)
9252 			break;
9253 		if (base_type(fn->arg_type[i]) == ARG_PTR_TO_BTF_ID)
9254 			return !!fn->arg_btf_id[i];
9255 		if (base_type(fn->arg_type[i]) == ARG_PTR_TO_SPIN_LOCK)
9256 			return fn->arg_btf_id[i] == BPF_PTR_POISON;
9257 		if (base_type(fn->arg_type[i]) != ARG_PTR_TO_BTF_ID && fn->arg_btf_id[i] &&
9258 		    /* arg_btf_id and arg_size are in a union. */
9259 		    (base_type(fn->arg_type[i]) != ARG_PTR_TO_MEM ||
9260 		     !(fn->arg_type[i] & MEM_FIXED_SIZE)))
9261 			return false;
9262 	}
9263 
9264 	return true;
9265 }
9266 
check_mem_arg_rw_flag_ok(const struct bpf_func_proto * fn)9267 static bool check_mem_arg_rw_flag_ok(const struct bpf_func_proto *fn)
9268 {
9269 	int i;
9270 
9271 	for (i = 0; i < ARRAY_SIZE(fn->arg_type); i++) {
9272 		enum bpf_arg_type arg_type = fn->arg_type[i];
9273 
9274 		if (arg_type == ARG_DONTCARE)
9275 			break;
9276 		if (base_type(arg_type) != ARG_PTR_TO_MEM)
9277 			continue;
9278 		if (!(arg_type & (MEM_WRITE | MEM_RDONLY)))
9279 			return false;
9280 	}
9281 
9282 	return true;
9283 }
9284 
check_proto_release_reg(const struct bpf_func_proto * fn,struct bpf_call_arg_meta * meta)9285 static bool check_proto_release_reg(const struct bpf_func_proto *fn, struct bpf_call_arg_meta *meta)
9286 {
9287 	int i;
9288 
9289 	for (i = 0; i < ARRAY_SIZE(fn->arg_type); i++) {
9290 		enum bpf_arg_type arg_type = fn->arg_type[i];
9291 
9292 		if (arg_type == ARG_DONTCARE)
9293 			break;
9294 		if (arg_type_is_release(arg_type)) {
9295 			if (meta->release_regno)
9296 				return false;
9297 			meta->release_regno = i + 1;
9298 		}
9299 	}
9300 
9301 	return true;
9302 }
9303 
check_func_proto(const struct bpf_func_proto * fn,struct bpf_call_arg_meta * meta)9304 static int check_func_proto(const struct bpf_func_proto *fn, struct bpf_call_arg_meta *meta)
9305 {
9306 	return check_raw_mode_ok(fn, meta) &&
9307 	       check_arg_pair_ok(fn) &&
9308 	       check_mem_arg_rw_flag_ok(fn) &&
9309 	       check_proto_release_reg(fn, meta) &&
9310 	       check_btf_id_ok(fn) ? 0 : -EINVAL;
9311 }
9312 
9313 /* Packet data might have moved, any old PTR_TO_PACKET[_META,_END]
9314  * are now invalid, so turn them into unknown SCALAR_VALUE.
9315  *
9316  * This also applies to dynptr slices belonging to skb and xdp dynptrs,
9317  * since these slices point to packet data.
9318  */
clear_all_pkt_pointers(struct bpf_verifier_env * env)9319 static void clear_all_pkt_pointers(struct bpf_verifier_env *env)
9320 {
9321 	struct bpf_func_state *state;
9322 	struct bpf_reg_state *reg;
9323 
9324 	bpf_for_each_reg_in_vstate(env->cur_state, state, reg, ({
9325 		if (reg_is_pkt_pointer_any(reg) || reg_is_dynptr_slice_pkt(reg)) {
9326 			bpf_diag_record_scrub(env, reg, BPF_DIAG_MOD_PKT_DATA_CHANGE);
9327 			mark_reg_invalid(env, reg);
9328 		}
9329 	}));
9330 }
9331 
9332 enum {
9333 	AT_PKT_END = -1,
9334 	BEYOND_PKT_END = -2,
9335 };
9336 
mark_pkt_end(struct bpf_verifier_state * vstate,int regn,bool range_open)9337 static void mark_pkt_end(struct bpf_verifier_state *vstate, int regn, bool range_open)
9338 {
9339 	struct bpf_func_state *state = vstate->frame[vstate->curframe];
9340 	struct bpf_reg_state *reg = &state->regs[regn];
9341 
9342 	if (reg->type != PTR_TO_PACKET)
9343 		/* PTR_TO_PACKET_META is not supported yet */
9344 		return;
9345 
9346 	/* The 'reg' is pkt > pkt_end or pkt >= pkt_end.
9347 	 * How far beyond pkt_end it goes is unknown.
9348 	 * if (!range_open) it's the case of pkt >= pkt_end
9349 	 * if (range_open) it's the case of pkt > pkt_end
9350 	 * hence this pointer is at least 1 byte bigger than pkt_end
9351 	 */
9352 	if (range_open)
9353 		reg->range = BEYOND_PKT_END;
9354 	else
9355 		reg->range = AT_PKT_END;
9356 }
9357 
__release_reference_nomark(struct bpf_verifier_state * state,int id)9358 static int __release_reference_nomark(struct bpf_verifier_state *state, int id)
9359 {
9360 	int i;
9361 
9362 	for (i = 0; i < state->acquired_refs; i++) {
9363 		if (state->refs[i].type != REF_TYPE_PTR)
9364 			continue;
9365 		if (state->refs[i].id == id) {
9366 			release_reference_state(state, i);
9367 			return 0;
9368 		}
9369 	}
9370 	return -EINVAL;
9371 }
9372 
release_reference_nomark(struct bpf_verifier_env * env,int id)9373 static int release_reference_nomark(struct bpf_verifier_env *env, int id)
9374 {
9375 	int err;
9376 
9377 	err = __release_reference_nomark(env->cur_state, id);
9378 	if (!err)
9379 		bpf_diag_record_ref_release(env, env->insn_idx, id);
9380 	return err;
9381 }
9382 
idstack_push(struct bpf_idmap * idmap,u32 id)9383 static int idstack_push(struct bpf_idmap *idmap, u32 id)
9384 {
9385 	int i;
9386 
9387 	if (!id)
9388 		return 0;
9389 
9390 	for (i = 0; i < idmap->cnt; i++)
9391 		if (idmap->map[i].old == id)
9392 			return 0;
9393 
9394 	if (WARN_ON_ONCE(idmap->cnt >= BPF_ID_MAP_SIZE))
9395 		return -EFAULT;
9396 
9397 	idmap->map[idmap->cnt++].old = id;
9398 	return 0;
9399 }
9400 
idstack_pop(struct bpf_idmap * idmap)9401 static int idstack_pop(struct bpf_idmap *idmap)
9402 {
9403 	if (!idmap->cnt)
9404 		return 0;
9405 
9406 	return idmap->map[--idmap->cnt].old;
9407 }
9408 
9409 /* Release id and objects derived from it iteratively in a DFS manner */
release_reference(struct bpf_verifier_env * env,int id)9410 static int release_reference(struct bpf_verifier_env *env, int id)
9411 {
9412 	u32 mask = (1 << STACK_SPILL) | (1 << STACK_DYNPTR);
9413 	struct bpf_verifier_state *vstate = env->cur_state;
9414 	struct bpf_idmap *idstack = &env->idmap_scratch;
9415 	struct bpf_stack_state *stack;
9416 	struct bpf_func_state *state;
9417 	struct bpf_reg_state *reg;
9418 	int i, err;
9419 
9420 	idstack->cnt = 0;
9421 	err = idstack_push(idstack, id);
9422 	if (err)
9423 		return err;
9424 
9425 	if (find_reference_state(vstate, id)) {
9426 		err = release_reference_nomark(env, id);
9427 		WARN_ON_ONCE(err);
9428 	}
9429 
9430 	while ((id = idstack_pop(idstack))) {
9431 		/*
9432 		 * Child references are inaccessible after parent is released,
9433 		 * any child references that exist at this point are a leak.
9434 		 */
9435 		for (i = 0; i < vstate->acquired_refs; i++) {
9436 			if (vstate->refs[i].type != REF_TYPE_PTR)
9437 				continue;
9438 			if (vstate->refs[i].parent_id != id)
9439 				continue;
9440 			verbose(env, "Leaking reference id=%d alloc_insn=%d. Release it first.\n",
9441 				vstate->refs[i].id, vstate->refs[i].insn_idx);
9442 			return -EINVAL;
9443 		}
9444 
9445 		bpf_for_each_reg_in_vstate_mask(vstate, state, reg, stack, mask, ({
9446 			if (reg->id != id && reg->parent_id != id)
9447 				continue;
9448 
9449 			/* Free objects derived from the current object */
9450 			if (reg->parent_id == id) {
9451 				err = idstack_push(idstack, reg->id);
9452 				if (err)
9453 					return err;
9454 			}
9455 
9456 			/*
9457 			 * A dynptr occupies two stack slots that invalidate_dynptr()
9458 			 * clears together. Record both scrubs before invalidating it.
9459 			 */
9460 			if (stack && stack->slot_type[BPF_REG_SIZE - 1] == STACK_DYNPTR) {
9461 				struct bpf_stack_state *dyn_stack = stack;
9462 
9463 				if (reg->dynptr.first_slot)
9464 					dyn_stack--;
9465 				bpf_diag_record_scrub(env, &dyn_stack[0].spilled_ptr,
9466 						      BPF_DIAG_MOD_REF_RELEASE);
9467 				bpf_diag_record_scrub(env, &dyn_stack[1].spilled_ptr,
9468 						      BPF_DIAG_MOD_REF_RELEASE);
9469 				invalidate_dynptr(env, dyn_stack);
9470 				continue;
9471 			}
9472 			bpf_diag_record_scrub(env, reg, BPF_DIAG_MOD_REF_RELEASE);
9473 			if (!stack || stack->slot_type[BPF_REG_SIZE - 1] == STACK_SPILL)
9474 				mark_reg_invalid(env, reg);
9475 		}));
9476 	}
9477 
9478 	return 0;
9479 }
9480 
invalidate_non_owning_refs(struct bpf_verifier_env * env)9481 static void invalidate_non_owning_refs(struct bpf_verifier_env *env)
9482 {
9483 	struct bpf_func_state *unused;
9484 	struct bpf_reg_state *reg;
9485 
9486 	bpf_for_each_reg_in_vstate(env->cur_state, unused, reg, ({
9487 		if (type_is_non_owning_ref(reg->type)) {
9488 			bpf_diag_record_scrub(env, reg, BPF_DIAG_MOD_NON_OWN_REF);
9489 			mark_reg_invalid(env, reg);
9490 		}
9491 	}));
9492 }
9493 
invalidate_rcu_protected_refs(struct bpf_verifier_env * env)9494 static void invalidate_rcu_protected_refs(struct bpf_verifier_env *env)
9495 {
9496 	struct bpf_stack_state *stack;
9497 	struct bpf_func_state *state;
9498 	struct bpf_reg_state *reg;
9499 	u32 clear_mask = (1 << STACK_SPILL) | (1 << STACK_ITER);
9500 
9501 	bpf_for_each_reg_in_vstate_mask(env->cur_state, state, reg, stack, clear_mask, ({
9502 		if (reg->type & MEM_RCU) {
9503 			bpf_diag_mod_begin(env, reg, NULL, BPF_DIAG_MOD_WRITE);
9504 			reg->type &= ~(MEM_RCU | PTR_MAYBE_NULL);
9505 			reg->type |= PTR_UNTRUSTED;
9506 			bpf_diag_mod_end(env);
9507 		}
9508 	}));
9509 }
9510 
ref_convert_alloc_rcu_protected(struct bpf_verifier_env * env,u32 id)9511 static int ref_convert_alloc_rcu_protected(struct bpf_verifier_env *env, u32 id)
9512 {
9513 	struct bpf_func_state *state;
9514 	struct bpf_reg_state *reg;
9515 	int err;
9516 
9517 	err = release_reference_nomark(env, id);
9518 	if (err)
9519 		return err;
9520 
9521 	bpf_for_each_reg_in_vstate(env->cur_state, state, reg, ({
9522 		if (reg->id != id)
9523 			continue;
9524 		if ((reg->type & MEM_ALLOC) && (reg->type & MEM_PERCPU)) {
9525 			bpf_diag_mod_begin(env, reg, NULL, BPF_DIAG_MOD_WRITE);
9526 			reg->id = 0;
9527 			reg->type &= ~MEM_ALLOC;
9528 			reg->type |= MEM_RCU;
9529 			bpf_diag_mod_end(env);
9530 		}
9531 	}));
9532 
9533 	return err;
9534 }
9535 
clear_caller_saved_regs(struct bpf_verifier_env * env,struct bpf_reg_state * regs)9536 static void clear_caller_saved_regs(struct bpf_verifier_env *env,
9537 				    struct bpf_reg_state *regs)
9538 {
9539 	int i;
9540 
9541 	bpf_diag_record_caller_saved(env, regs);
9542 
9543 	/* after the call registers r0 - r5 were scratched */
9544 	for (i = 0; i < CALLER_SAVED_REGS; i++) {
9545 		bpf_mark_reg_not_init(env, &regs[caller_saved[i]]);
9546 		__check_reg_arg(env, regs, caller_saved[i], DST_OP_NO_MARK);
9547 	}
9548 }
9549 
invalidate_outgoing_stack_args(struct bpf_verifier_env * env,struct bpf_func_state * state)9550 static void invalidate_outgoing_stack_args(struct bpf_verifier_env *env,
9551 					   struct bpf_func_state *state)
9552 {
9553 	int i, nslots = state->out_stack_arg_cnt;
9554 
9555 	for (i = 0; i < nslots; i++) {
9556 		bpf_diag_record_scrub(env, &state->stack_arg_regs[i], BPF_DIAG_MOD_CALLER_SAVED);
9557 		bpf_mark_reg_not_init(env, &state->stack_arg_regs[i]);
9558 	}
9559 }
9560 
9561 typedef int (*set_callee_state_fn)(struct bpf_verifier_env *env,
9562 				   struct bpf_func_state *caller,
9563 				   struct bpf_func_state *callee,
9564 				   int insn_idx);
9565 
9566 static int set_callee_state(struct bpf_verifier_env *env,
9567 			    struct bpf_func_state *caller,
9568 			    struct bpf_func_state *callee, int insn_idx);
9569 
setup_func_entry(struct bpf_verifier_env * env,int subprog,int callsite,set_callee_state_fn set_callee_state_cb,struct bpf_verifier_state * state)9570 static int setup_func_entry(struct bpf_verifier_env *env, int subprog, int callsite,
9571 			    set_callee_state_fn set_callee_state_cb,
9572 			    struct bpf_verifier_state *state)
9573 {
9574 	struct bpf_func_state *caller, *callee;
9575 	int err;
9576 
9577 	if (state->curframe + 1 >= MAX_CALL_FRAMES) {
9578 		verbose(env, "the call stack of %d frames is too deep\n",
9579 			state->curframe + 2);
9580 		return -E2BIG;
9581 	}
9582 
9583 	if (state->frame[state->curframe + 1]) {
9584 		verifier_bug(env, "Frame %d already allocated", state->curframe + 1);
9585 		return -EFAULT;
9586 	}
9587 
9588 	caller = state->frame[state->curframe];
9589 	callee = kzalloc_obj(*callee, GFP_KERNEL_ACCOUNT);
9590 	if (!callee)
9591 		return -ENOMEM;
9592 	state->frame[state->curframe + 1] = callee;
9593 
9594 	/* callee cannot access r0, r6 - r9 for reading and has to write
9595 	 * into its own stack before reading from it.
9596 	 * callee can read/write into caller's stack
9597 	 */
9598 	init_func_state(env, callee,
9599 			/* remember the callsite, it will be used by bpf_exit */
9600 			callsite,
9601 			state->curframe + 1 /* frameno within this callchain */,
9602 			subprog /* subprog number within this prog */);
9603 	err = set_callee_state_cb(env, caller, callee, callsite);
9604 	if (err)
9605 		goto err_out;
9606 
9607 	/* only increment it after check_reg_arg() finished */
9608 	state->curframe++;
9609 
9610 	return 0;
9611 
9612 err_out:
9613 	free_func_state(callee);
9614 	state->frame[state->curframe + 1] = NULL;
9615 	return err;
9616 }
9617 
btf_check_func_arg_match(struct bpf_verifier_env * env,int subprog,const struct btf * btf,struct bpf_reg_state * regs)9618 static int btf_check_func_arg_match(struct bpf_verifier_env *env, int subprog,
9619 				    const struct btf *btf,
9620 				    struct bpf_reg_state *regs)
9621 {
9622 	struct bpf_subprog_info *sub = subprog_info(env, subprog);
9623 	struct bpf_func_state *caller = cur_func(env);
9624 	struct bpf_verifier_log *log = &env->log;
9625 	struct ref_obj_desc ref_obj = {};
9626 	const struct btf_param *args;
9627 	const struct btf_type *func, *func_proto;
9628 	u32 i;
9629 	int ret, err;
9630 
9631 	ret = btf_prepare_func_args(env, subprog);
9632 	if (ret) {
9633 		if (bpf_in_stack_arg_cnt(sub) > 0) {
9634 			err = check_outgoing_stack_args(env, caller, sub->arg_cnt,
9635 							bpf_subprog_name(env, subprog),
9636 							NULL, NULL);
9637 			if (err)
9638 				return err;
9639 		}
9640 		return ret;
9641 	}
9642 
9643 	func = btf_type_by_id(btf, env->prog->aux->func_info[subprog].type_id);
9644 	func_proto = btf_type_by_id(btf, func->type);
9645 	args = btf_params(func_proto);
9646 	ret = check_outgoing_stack_args(env, caller, sub->arg_cnt,
9647 					bpf_subprog_name(env, subprog), btf, args);
9648 	if (ret)
9649 		return ret;
9650 
9651 	/* check that BTF function arguments match actual types that the
9652 	 * verifier sees.
9653 	 */
9654 	for (i = 0; i < sub->arg_cnt; i++) {
9655 		argno_t argno = argno_from_arg(i + 1);
9656 		struct bpf_reg_state *reg = get_func_arg_reg(caller, regs, i);
9657 		struct bpf_subprog_arg_info *arg = &sub->args[i];
9658 
9659 		if (arg->arg_type == ARG_ANYTHING) {
9660 			if (reg->type != SCALAR_VALUE) {
9661 				bpf_log(log, "%s is not a scalar\n", reg_arg_name(env, argno));
9662 				return -EINVAL;
9663 			}
9664 		} else if (arg->arg_type & PTR_UNTRUSTED) {
9665 			/*
9666 			 * Anything is allowed for untrusted arguments, as these are
9667 			 * read-only and probe read instructions would protect against
9668 			 * invalid memory access.
9669 			 */
9670 		} else if (arg->arg_type == ARG_PTR_TO_CTX) {
9671 			ret = check_func_arg_reg_off(env, reg, argno, ARG_PTR_TO_CTX);
9672 			if (ret < 0)
9673 				return ret;
9674 			/* If function expects ctx type in BTF check that caller
9675 			 * is passing PTR_TO_CTX.
9676 			 */
9677 			if (reg->type != PTR_TO_CTX) {
9678 				bpf_log(log, "%s expects pointer to ctx\n",
9679 					reg_arg_name(env, argno));
9680 				return -EINVAL;
9681 			}
9682 		} else if (base_type(arg->arg_type) == ARG_PTR_TO_MEM) {
9683 			ret = check_func_arg_reg_off(env, reg, argno, ARG_DONTCARE);
9684 			if (ret < 0)
9685 				return ret;
9686 			if (check_mem_reg(env, reg, argno, arg->mem_size, BPF_READ | BPF_WRITE, NULL,
9687 					  NULL))
9688 				return -EINVAL;
9689 			if (!(arg->arg_type & PTR_MAYBE_NULL) &&
9690 			    (type_may_be_null(reg->type) || bpf_register_is_null(reg))) {
9691 				bpf_log(log, "%s is expected to be non-NULL\n",
9692 					reg_arg_name(env, argno));
9693 				return -EINVAL;
9694 			}
9695 		} else if (base_type(arg->arg_type) == ARG_PTR_TO_ARENA) {
9696 			/*
9697 			 * Can pass any value and the kernel won't crash, but
9698 			 * only PTR_TO_ARENA or SCALAR make sense. Everything
9699 			 * else is a bug in the bpf program. Point it out to
9700 			 * the user at the verification time instead of
9701 			 * run-time debug nightmare.
9702 			 */
9703 			if (reg->type != PTR_TO_ARENA && reg->type != SCALAR_VALUE) {
9704 				bpf_log(log, "%s is not a pointer to arena or scalar.\n",
9705 					reg_arg_name(env, argno));
9706 				return -EINVAL;
9707 			}
9708 		} else if (arg->arg_type == ARG_PTR_TO_DYNPTR) {
9709 			ret = check_func_arg_reg_off(env, reg, argno, ARG_PTR_TO_DYNPTR);
9710 			if (ret)
9711 				return ret;
9712 
9713 			ret = process_dynptr_func(env, reg, argno, env->insn_idx,
9714 						  bpf_subprog_name(env, subprog), arg->arg_type,
9715 						  &ref_obj, NULL);
9716 			if (ret)
9717 				return ret;
9718 		} else if (base_type(arg->arg_type) == ARG_PTR_TO_BTF_ID) {
9719 			struct bpf_call_arg_meta meta;
9720 			int err;
9721 
9722 			if (bpf_register_is_null(reg) && type_may_be_null(arg->arg_type))
9723 				continue;
9724 
9725 			memset(&meta, 0, sizeof(meta)); /* leave func_id as zero */
9726 			err = check_reg_type(env, reg, argno, arg->arg_type, &arg->btf_id, &meta,
9727 					     bpf_subprog_name(env, subprog));
9728 			err = err ?: check_func_arg_reg_off(env, reg, argno, arg->arg_type);
9729 			if (err)
9730 				return err;
9731 		} else {
9732 			verifier_bug(env, "unrecognized %s type %d",
9733 				     reg_arg_name(env, argno), arg->arg_type);
9734 			return -EFAULT;
9735 		}
9736 	}
9737 
9738 	return 0;
9739 }
9740 
9741 /* Compare BTF of a function call with given bpf_reg_state.
9742  * Returns:
9743  * EFAULT - there is a verifier bug. Abort verification.
9744  * EINVAL - there is a type mismatch or BTF is not available.
9745  * 0 - BTF matches with what bpf_reg_state expects.
9746  * Only PTR_TO_CTX and SCALAR_VALUE states are recognized.
9747  */
btf_check_subprog_call(struct bpf_verifier_env * env,int subprog,struct bpf_reg_state * regs)9748 static int btf_check_subprog_call(struct bpf_verifier_env *env, int subprog,
9749 				  struct bpf_reg_state *regs)
9750 {
9751 	struct bpf_prog *prog = env->prog;
9752 	struct btf *btf = prog->aux->btf;
9753 	u32 btf_id;
9754 	int err;
9755 
9756 	if (!prog->aux->func_info)
9757 		return -EINVAL;
9758 
9759 	btf_id = prog->aux->func_info[subprog].type_id;
9760 	if (!btf_id)
9761 		return -EFAULT;
9762 
9763 	if (prog->aux->func_info_aux[subprog].unreliable)
9764 		return -EINVAL;
9765 
9766 	err = btf_check_func_arg_match(env, subprog, btf, regs);
9767 	/* Compiler optimizations can remove arguments from static functions
9768 	 * or mismatched type can be passed into a global function.
9769 	 * In such cases mark the function as unreliable from BTF point of view.
9770 	 */
9771 	if (err)
9772 		prog->aux->func_info_aux[subprog].unreliable = true;
9773 	return err;
9774 }
9775 
push_callback_call(struct bpf_verifier_env * env,struct bpf_insn * insn,int insn_idx,int subprog,set_callee_state_fn set_callee_state_cb)9776 static int push_callback_call(struct bpf_verifier_env *env, struct bpf_insn *insn,
9777 			      int insn_idx, int subprog,
9778 			      set_callee_state_fn set_callee_state_cb)
9779 {
9780 	struct bpf_verifier_state *state = env->cur_state, *callback_state;
9781 	struct bpf_func_state *caller, *callee;
9782 	int err;
9783 
9784 	caller = state->frame[state->curframe];
9785 	err = btf_check_subprog_call(env, subprog, caller->regs);
9786 	if (err == -EFAULT)
9787 		return err;
9788 
9789 	/* set_callee_state is used for direct subprog calls, but we are
9790 	 * interested in validating only BPF helpers that can call subprogs as
9791 	 * callbacks
9792 	 */
9793 	env->subprog_info[subprog].is_cb = true;
9794 	if (bpf_pseudo_kfunc_call(insn) &&
9795 	    !is_callback_calling_kfunc(insn->imm)) {
9796 		verifier_bug(env, "kfunc %s#%d not marked as callback-calling",
9797 			     func_id_name(insn->imm), insn->imm);
9798 		return -EFAULT;
9799 	} else if (!bpf_pseudo_kfunc_call(insn) &&
9800 		   !is_callback_calling_function(insn->imm)) { /* helper */
9801 		verifier_bug(env, "helper %s#%d not marked as callback-calling",
9802 			     func_id_name(insn->imm), insn->imm);
9803 		return -EFAULT;
9804 	}
9805 
9806 	if (bpf_is_async_callback_calling_insn(insn)) {
9807 		struct bpf_verifier_state *async_cb;
9808 
9809 		/* there is no real recursion here. timer and workqueue callbacks are async */
9810 		env->subprog_info[subprog].is_async_cb = true;
9811 		async_cb = push_async_cb(env, env->subprog_info[subprog].start,
9812 					 insn_idx, subprog,
9813 					 is_async_cb_sleepable(env, insn));
9814 		if (IS_ERR(async_cb))
9815 			return PTR_ERR(async_cb);
9816 		callee = async_cb->frame[0];
9817 		callee->async_entry_cnt = caller->async_entry_cnt + 1;
9818 
9819 		/* Convert bpf_timer_set_callback() args into timer callback args */
9820 		err = set_callee_state_cb(env, caller, callee, insn_idx);
9821 		if (err)
9822 			return err;
9823 
9824 		return 0;
9825 	}
9826 
9827 	/* for callback functions enqueue entry to callback and
9828 	 * proceed with next instruction within current frame.
9829 	 */
9830 	callback_state = push_stack(env, env->subprog_info[subprog].start, insn_idx, false);
9831 	if (IS_ERR(callback_state))
9832 		return PTR_ERR(callback_state);
9833 
9834 	err = setup_func_entry(env, subprog, insn_idx, set_callee_state_cb,
9835 			       callback_state);
9836 	if (err)
9837 		return err;
9838 
9839 	callback_state->callback_unroll_depth++;
9840 	callback_state->frame[callback_state->curframe - 1]->callback_depth++;
9841 	caller->callback_depth = 0;
9842 	return 0;
9843 }
9844 
9845 static int process_bpf_exit_full(struct bpf_verifier_env *env,
9846 				 bool *do_print_state, bool exception_exit);
9847 
check_func_call(struct bpf_verifier_env * env,struct bpf_insn * insn,int * insn_idx)9848 static int check_func_call(struct bpf_verifier_env *env, struct bpf_insn *insn,
9849 			   int *insn_idx)
9850 {
9851 	struct bpf_verifier_state *state = env->cur_state;
9852 	struct bpf_subprog_info *caller_info;
9853 	u16 callee_incoming, stack_arg_cnt;
9854 	struct bpf_func_state *caller;
9855 	int err, subprog, target_insn;
9856 
9857 	target_insn = *insn_idx + insn->imm + 1;
9858 	subprog = bpf_find_subprog(env, target_insn);
9859 	if (verifier_bug_if(subprog < 0, env, "target of func call at insn %d is not a program",
9860 			    target_insn))
9861 		return -EFAULT;
9862 
9863 	caller = state->frame[state->curframe];
9864 	err = btf_check_subprog_call(env, subprog, caller->regs);
9865 	if (err == -EFAULT)
9866 		return err;
9867 	if (bpf_subprog_is_global(env, subprog)) {
9868 		const char *sub_name = bpf_subprog_name(env, subprog);
9869 		const char *operation;
9870 		bool returns_void;
9871 
9872 		if (env->cur_state->active_locks) {
9873 			verbose(env, "global function calls are not allowed while holding a lock,\n"
9874 				     "use static function instead\n");
9875 			operation = bpf_diag_fmt(env, "global function %s()", sub_name);
9876 			bpf_diag_ctx_active(env, *insn_idx, operation, BPF_DIAG_CONTEXT_LOCK,
9877 					    "Release the lock before calling the global function, or use a static function instead.");
9878 			return -EINVAL;
9879 		}
9880 
9881 		if (env->subprog_info[subprog].might_sleep && !in_sleepable_context(env)) {
9882 			verbose(env, "sleepable global function %s() called in %s\n",
9883 				sub_name, non_sleepable_context_description(env));
9884 			operation = bpf_diag_fmt(env, "sleepable global function %s()", sub_name);
9885 			bpf_diag_ctx_forbidden(env, *insn_idx, operation,
9886 				"Move the call outside the critical section, or use a non-sleepable function.");
9887 			return -EINVAL;
9888 		}
9889 
9890 		if (err) {
9891 			verbose(env, "Caller passes invalid args into func#%d ('%s')\n",
9892 				subprog, sub_name);
9893 			return err;
9894 		}
9895 
9896 		if (env->log.level & BPF_LOG_LEVEL)
9897 			verbose(env, "Func#%d ('%s') is global and assumed valid.\n",
9898 				subprog, sub_name);
9899 		returns_void = subprog_returns_void(env, subprog);
9900 		if (env->subprog_info[subprog].changes_pkt_data)
9901 			clear_all_pkt_pointers(env);
9902 		/* mark global subprog for verifying after main prog */
9903 		subprog_aux(env, subprog)->called = true;
9904 		if (returns_void)
9905 			bpf_diag_record_scrub(env, &caller->regs[BPF_REG_0], BPF_DIAG_MOD_CALLER_SAVED);
9906 		else
9907 			bpf_diag_mod_begin(env, &caller->regs[BPF_REG_0], NULL, BPF_DIAG_MOD_WRITE);
9908 		clear_caller_saved_regs(env, caller->regs);
9909 		invalidate_outgoing_stack_args(env, cur_func(env));
9910 
9911 		/* All non-void global functions return a 64-bit SCALAR_VALUE. */
9912 		if (!returns_void) {
9913 			mark_reg_unknown(env, caller->regs, BPF_REG_0);
9914 			bpf_diag_mod_end(env);
9915 		}
9916 
9917 		if (env->subprog_info[subprog].might_throw) {
9918 			struct bpf_verifier_state *branch;
9919 
9920 			branch = push_stack(env, *insn_idx + 1, *insn_idx, false);
9921 			if (IS_ERR(branch)) {
9922 				verbose(env, "failed to push state for global subprog exception path\n");
9923 				return PTR_ERR(branch);
9924 			}
9925 			return process_bpf_exit_full(env, NULL, true);
9926 		}
9927 
9928 		/* continue with next insn after call */
9929 		return 0;
9930 	}
9931 
9932 	/*
9933 	 * Track caller's total stack arg count (incoming + max outgoing).
9934 	 * This is needed so the JIT knows how much stack arg space to allocate.
9935 	 */
9936 	caller_info = &env->subprog_info[caller->subprogno];
9937 	callee_incoming = bpf_in_stack_arg_cnt(&env->subprog_info[subprog]);
9938 	stack_arg_cnt = bpf_in_stack_arg_cnt(caller_info) + callee_incoming;
9939 	if (stack_arg_cnt > caller_info->stack_arg_cnt)
9940 		caller_info->stack_arg_cnt = stack_arg_cnt;
9941 
9942 	/* for regular function entry setup new frame and continue
9943 	 * from that frame.
9944 	 */
9945 	err = setup_func_entry(env, subprog, *insn_idx, set_callee_state, state);
9946 	if (err)
9947 		return err;
9948 
9949 	bpf_diag_record_scrub(env, &caller->regs[BPF_REG_0], BPF_DIAG_MOD_CALLER_SAVED);
9950 	clear_caller_saved_regs(env, caller->regs);
9951 
9952 	/* and go analyze first insn of the callee */
9953 	*insn_idx = env->subprog_info[subprog].start - 1;
9954 
9955 	if (env->log.level & BPF_LOG_LEVEL) {
9956 		verbose(env, "caller:\n");
9957 		print_verifier_state(env, state, caller->frameno, true);
9958 		verbose(env, "callee:\n");
9959 		print_verifier_state(env, state, state->curframe, true);
9960 	}
9961 
9962 	return 0;
9963 }
9964 
map_set_for_each_callback_args(struct bpf_verifier_env * env,struct bpf_func_state * caller,struct bpf_func_state * callee)9965 int map_set_for_each_callback_args(struct bpf_verifier_env *env,
9966 				   struct bpf_func_state *caller,
9967 				   struct bpf_func_state *callee)
9968 {
9969 	/* bpf_for_each_map_elem(struct bpf_map *map, void *callback_fn,
9970 	 *      void *callback_ctx, u64 flags);
9971 	 * callback_fn(struct bpf_map *map, void *key, void *value,
9972 	 *      void *callback_ctx);
9973 	 */
9974 	callee->regs[BPF_REG_1] = caller->regs[BPF_REG_1];
9975 
9976 	callee->regs[BPF_REG_2].type = PTR_TO_MAP_KEY;
9977 	__mark_reg_known_zero(&callee->regs[BPF_REG_2]);
9978 	callee->regs[BPF_REG_2].map_ptr = caller->regs[BPF_REG_1].map_ptr;
9979 
9980 	callee->regs[BPF_REG_3].type = PTR_TO_MAP_VALUE;
9981 	__mark_reg_known_zero(&callee->regs[BPF_REG_3]);
9982 	callee->regs[BPF_REG_3].map_ptr = caller->regs[BPF_REG_1].map_ptr;
9983 
9984 	/* pointer to stack or null */
9985 	callee->regs[BPF_REG_4] = caller->regs[BPF_REG_3];
9986 
9987 	/* unused */
9988 	bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_5]);
9989 	return 0;
9990 }
9991 
set_callee_state(struct bpf_verifier_env * env,struct bpf_func_state * caller,struct bpf_func_state * callee,int insn_idx)9992 static int set_callee_state(struct bpf_verifier_env *env,
9993 			    struct bpf_func_state *caller,
9994 			    struct bpf_func_state *callee, int insn_idx)
9995 {
9996 	int i;
9997 
9998 	/* copy r1 - r5 args that callee can access.  The copy includes parent
9999 	 * pointers, which connects us up to the liveness chain
10000 	 */
10001 	for (i = BPF_REG_1; i <= BPF_REG_5; i++)
10002 		callee->regs[i] = caller->regs[i];
10003 	return 0;
10004 }
10005 
set_map_elem_callback_state(struct bpf_verifier_env * env,struct bpf_func_state * caller,struct bpf_func_state * callee,int insn_idx)10006 static int set_map_elem_callback_state(struct bpf_verifier_env *env,
10007 				       struct bpf_func_state *caller,
10008 				       struct bpf_func_state *callee,
10009 				       int insn_idx)
10010 {
10011 	struct bpf_insn_aux_data *insn_aux = &env->insn_aux_data[insn_idx];
10012 	struct bpf_map *map;
10013 	int err;
10014 
10015 	/* valid map_ptr and poison value does not matter */
10016 	map = insn_aux->map_ptr_state.map_ptr;
10017 	if (!map->ops->map_set_for_each_callback_args ||
10018 	    !map->ops->map_for_each_callback) {
10019 		verbose(env, "callback function not allowed for map\n");
10020 		return -ENOTSUPP;
10021 	}
10022 
10023 	err = map->ops->map_set_for_each_callback_args(env, caller, callee);
10024 	if (err)
10025 		return err;
10026 
10027 	callee->in_callback_fn = true;
10028 	callee->callback_ret_range = retval_range(0, 1);
10029 	return 0;
10030 }
10031 
set_loop_callback_state(struct bpf_verifier_env * env,struct bpf_func_state * caller,struct bpf_func_state * callee,int insn_idx)10032 static int set_loop_callback_state(struct bpf_verifier_env *env,
10033 				   struct bpf_func_state *caller,
10034 				   struct bpf_func_state *callee,
10035 				   int insn_idx)
10036 {
10037 	/* bpf_loop(u32 nr_loops, void *callback_fn, void *callback_ctx,
10038 	 *	    u64 flags);
10039 	 * callback_fn(u64 index, void *callback_ctx);
10040 	 */
10041 	callee->regs[BPF_REG_1].type = SCALAR_VALUE;
10042 	callee->regs[BPF_REG_2] = caller->regs[BPF_REG_3];
10043 
10044 	/* unused */
10045 	bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_3]);
10046 	bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_4]);
10047 	bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_5]);
10048 
10049 	callee->in_callback_fn = true;
10050 	callee->callback_ret_range = retval_range(0, 1);
10051 	return 0;
10052 }
10053 
set_timer_callback_state(struct bpf_verifier_env * env,struct bpf_func_state * caller,struct bpf_func_state * callee,int insn_idx)10054 static int set_timer_callback_state(struct bpf_verifier_env *env,
10055 				    struct bpf_func_state *caller,
10056 				    struct bpf_func_state *callee,
10057 				    int insn_idx)
10058 {
10059 	struct bpf_map *map_ptr = caller->regs[BPF_REG_1].map_ptr;
10060 
10061 	/* bpf_timer_set_callback(struct bpf_timer *timer, void *callback_fn);
10062 	 * callback_fn(struct bpf_map *map, void *key, void *value);
10063 	 */
10064 	callee->regs[BPF_REG_1].type = CONST_PTR_TO_MAP;
10065 	__mark_reg_known_zero(&callee->regs[BPF_REG_1]);
10066 	callee->regs[BPF_REG_1].map_ptr = map_ptr;
10067 
10068 	callee->regs[BPF_REG_2].type = PTR_TO_MAP_KEY;
10069 	__mark_reg_known_zero(&callee->regs[BPF_REG_2]);
10070 	callee->regs[BPF_REG_2].map_ptr = map_ptr;
10071 
10072 	callee->regs[BPF_REG_3].type = PTR_TO_MAP_VALUE;
10073 	__mark_reg_known_zero(&callee->regs[BPF_REG_3]);
10074 	callee->regs[BPF_REG_3].map_ptr = map_ptr;
10075 
10076 	/* unused */
10077 	bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_4]);
10078 	bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_5]);
10079 	callee->in_async_callback_fn = true;
10080 	callee->callback_ret_range = retval_range(0, 0);
10081 	return 0;
10082 }
10083 
set_find_vma_callback_state(struct bpf_verifier_env * env,struct bpf_func_state * caller,struct bpf_func_state * callee,int insn_idx)10084 static int set_find_vma_callback_state(struct bpf_verifier_env *env,
10085 				       struct bpf_func_state *caller,
10086 				       struct bpf_func_state *callee,
10087 				       int insn_idx)
10088 {
10089 	/* bpf_find_vma(struct task_struct *task, u64 addr,
10090 	 *               void *callback_fn, void *callback_ctx, u64 flags)
10091 	 * (callback_fn)(struct task_struct *task,
10092 	 *               struct vm_area_struct *vma, void *callback_ctx);
10093 	 */
10094 	callee->regs[BPF_REG_1] = caller->regs[BPF_REG_1];
10095 
10096 	callee->regs[BPF_REG_2].type = PTR_TO_BTF_ID;
10097 	__mark_reg_known_zero(&callee->regs[BPF_REG_2]);
10098 	callee->regs[BPF_REG_2].btf =  btf_vmlinux;
10099 	callee->regs[BPF_REG_2].btf_id = btf_tracing_ids[BTF_TRACING_TYPE_VMA];
10100 
10101 	/* pointer to stack or null */
10102 	callee->regs[BPF_REG_3] = caller->regs[BPF_REG_4];
10103 
10104 	/* unused */
10105 	bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_4]);
10106 	bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_5]);
10107 	callee->in_callback_fn = true;
10108 	callee->callback_ret_range = retval_range(0, 1);
10109 	return 0;
10110 }
10111 
set_user_ringbuf_callback_state(struct bpf_verifier_env * env,struct bpf_func_state * caller,struct bpf_func_state * callee,int insn_idx)10112 static int set_user_ringbuf_callback_state(struct bpf_verifier_env *env,
10113 					   struct bpf_func_state *caller,
10114 					   struct bpf_func_state *callee,
10115 					   int insn_idx)
10116 {
10117 	/* bpf_user_ringbuf_drain(struct bpf_map *map, void *callback_fn, void
10118 	 *			  callback_ctx, u64 flags);
10119 	 * callback_fn(const struct bpf_dynptr_t* dynptr, void *callback_ctx);
10120 	 */
10121 	bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_0]);
10122 	mark_dynptr_cb_reg(env, &callee->regs[BPF_REG_1], BPF_DYNPTR_TYPE_LOCAL);
10123 	callee->regs[BPF_REG_2] = caller->regs[BPF_REG_3];
10124 
10125 	/* unused */
10126 	bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_3]);
10127 	bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_4]);
10128 	bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_5]);
10129 
10130 	callee->in_callback_fn = true;
10131 	callee->callback_ret_range = retval_range(0, 1);
10132 	return 0;
10133 }
10134 
set_rbtree_add_callback_state(struct bpf_verifier_env * env,struct bpf_func_state * caller,struct bpf_func_state * callee,int insn_idx)10135 static int set_rbtree_add_callback_state(struct bpf_verifier_env *env,
10136 					 struct bpf_func_state *caller,
10137 					 struct bpf_func_state *callee,
10138 					 int insn_idx)
10139 {
10140 	/* void bpf_rbtree_add_impl(struct bpf_rb_root *root, struct bpf_rb_node *node,
10141 	 *                     bool (less)(struct bpf_rb_node *a, const struct bpf_rb_node *b));
10142 	 *
10143 	 * 'struct bpf_rb_node *node' arg to bpf_rbtree_add_impl is the same PTR_TO_BTF_ID w/ offset
10144 	 * that 'less' callback args will be receiving. However, 'node' arg was release_reference'd
10145 	 * by this point, so look at 'root'
10146 	 */
10147 	struct btf_field *field;
10148 
10149 	field = reg_find_field_offset(&caller->regs[BPF_REG_1],
10150 				      caller->regs[BPF_REG_1].var_off.value,
10151 				      BPF_RB_ROOT);
10152 	if (!field || !field->graph_root.value_btf_id)
10153 		return -EFAULT;
10154 
10155 	mark_reg_graph_node(callee->regs, BPF_REG_1, &field->graph_root);
10156 	ref_set_non_owning(env, &callee->regs[BPF_REG_1]);
10157 	mark_reg_graph_node(callee->regs, BPF_REG_2, &field->graph_root);
10158 	ref_set_non_owning(env, &callee->regs[BPF_REG_2]);
10159 
10160 	bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_3]);
10161 	bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_4]);
10162 	bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_5]);
10163 	callee->in_callback_fn = true;
10164 	callee->callback_ret_range = retval_range(0, 1);
10165 	return 0;
10166 }
10167 
set_task_work_schedule_callback_state(struct bpf_verifier_env * env,struct bpf_func_state * caller,struct bpf_func_state * callee,int insn_idx)10168 static int set_task_work_schedule_callback_state(struct bpf_verifier_env *env,
10169 						 struct bpf_func_state *caller,
10170 						 struct bpf_func_state *callee,
10171 						 int insn_idx)
10172 {
10173 	struct bpf_map *map_ptr = caller->regs[BPF_REG_3].map_ptr;
10174 
10175 	/*
10176 	 * callback_fn(struct bpf_map *map, void *key, void *value);
10177 	 */
10178 	callee->regs[BPF_REG_1].type = CONST_PTR_TO_MAP;
10179 	__mark_reg_known_zero(&callee->regs[BPF_REG_1]);
10180 	callee->regs[BPF_REG_1].map_ptr = map_ptr;
10181 
10182 	callee->regs[BPF_REG_2].type = PTR_TO_MAP_KEY;
10183 	__mark_reg_known_zero(&callee->regs[BPF_REG_2]);
10184 	callee->regs[BPF_REG_2].map_ptr = map_ptr;
10185 
10186 	callee->regs[BPF_REG_3].type = PTR_TO_MAP_VALUE;
10187 	__mark_reg_known_zero(&callee->regs[BPF_REG_3]);
10188 	callee->regs[BPF_REG_3].map_ptr = map_ptr;
10189 
10190 	/* unused */
10191 	bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_4]);
10192 	bpf_mark_reg_not_init(env, &callee->regs[BPF_REG_5]);
10193 	callee->in_async_callback_fn = true;
10194 	callee->callback_ret_range = retval_range(S32_MIN, S32_MAX);
10195 	return 0;
10196 }
10197 
10198 static bool is_rbtree_lock_required_kfunc(u32 btf_id);
10199 
account_processed_insn(struct bpf_verifier_env * env)10200 static void account_processed_insn(struct bpf_verifier_env *env)
10201 {
10202 	struct bpf_func_state *frame = cur_func(env);
10203 
10204 	env->insn_processed++;
10205 	frame->insns_subtotal++;
10206 	env->subprog_info[frame->subprogno].insns_self++;
10207 }
10208 
account_processed_insns(struct bpf_verifier_env * env,struct bpf_func_state * callee,struct bpf_func_state * caller)10209 static void account_processed_insns(struct bpf_verifier_env *env,
10210 				    struct bpf_func_state *callee,
10211 				    struct bpf_func_state *caller)
10212 {
10213 	u32 insns;
10214 
10215 	if (!callee)
10216 		return;
10217 
10218 	insns = callee->insns_subtotal;
10219 
10220 	env->subprog_info[callee->subprogno].insns_total += insns;
10221 	if (caller)
10222 		caller->insns_subtotal += insns;
10223 	callee->insns_subtotal = 0;
10224 }
10225 
account_current_path(struct bpf_verifier_env * env)10226 static void account_current_path(struct bpf_verifier_env *env)
10227 {
10228 	struct bpf_verifier_state *state = env->cur_state;
10229 	int frame;
10230 
10231 	for (frame = state->curframe; frame >= 0; frame--)
10232 		account_processed_insns(env, state->frame[frame],
10233 					frame ? state->frame[frame - 1] : NULL);
10234 }
10235 
10236 /* Are we currently verifying the callback for a rbtree helper that must
10237  * be called with lock held? If so, no need to complain about unreleased
10238  * lock
10239  */
in_rbtree_lock_required_cb(struct bpf_verifier_env * env)10240 static bool in_rbtree_lock_required_cb(struct bpf_verifier_env *env)
10241 {
10242 	struct bpf_verifier_state *state = env->cur_state;
10243 	struct bpf_insn *insn = env->prog->insnsi;
10244 	struct bpf_func_state *callee;
10245 	int kfunc_btf_id;
10246 
10247 	if (!state->curframe)
10248 		return false;
10249 
10250 	callee = state->frame[state->curframe];
10251 
10252 	if (!callee->in_callback_fn)
10253 		return false;
10254 
10255 	kfunc_btf_id = insn[callee->callsite].imm;
10256 	return is_rbtree_lock_required_kfunc(kfunc_btf_id);
10257 }
10258 
retval_range_within(struct bpf_retval_range range,const struct bpf_reg_state * reg)10259 static bool retval_range_within(struct bpf_retval_range range, const struct bpf_reg_state *reg)
10260 {
10261 	if (range.return_32bit)
10262 		return range.minval <= reg_s32_min(reg) && reg_s32_max(reg) <= range.maxval;
10263 	else
10264 		return range.minval <= reg_smin(reg) && reg_smax(reg) <= range.maxval;
10265 }
10266 
prepare_func_exit(struct bpf_verifier_env * env,int * insn_idx)10267 static int prepare_func_exit(struct bpf_verifier_env *env, int *insn_idx)
10268 {
10269 	struct bpf_verifier_state *state = env->cur_state, *prev_st;
10270 	struct bpf_func_state *caller, *callee;
10271 	struct bpf_reg_state *r0;
10272 	bool in_callback_fn;
10273 	int err;
10274 
10275 	callee = state->frame[state->curframe];
10276 	r0 = &callee->regs[BPF_REG_0];
10277 	if (r0->type == PTR_TO_STACK) {
10278 		/* technically it's ok to return caller's stack pointer
10279 		 * (or caller's caller's pointer) back to the caller,
10280 		 * since these pointers are valid. Only current stack
10281 		 * pointer will be invalid as soon as function exits,
10282 		 * but let's be conservative
10283 		 */
10284 		verbose(env, "cannot return stack pointer to the caller\n");
10285 		return -EINVAL;
10286 	}
10287 
10288 	caller = state->frame[state->curframe - 1];
10289 	if (callee->in_callback_fn) {
10290 		if (r0->type != SCALAR_VALUE) {
10291 			verbose(env, "R0 not a scalar value\n");
10292 			return -EACCES;
10293 		}
10294 
10295 		/* we are going to rely on register's precise value */
10296 		err = mark_chain_precision(env, BPF_REG_0);
10297 		if (err)
10298 			return err;
10299 
10300 		/* enforce R0 return value range, and bpf_callback_t returns 64bit */
10301 		if (!retval_range_within(callee->callback_ret_range, r0)) {
10302 			verbose_invalid_scalar(env, r0, callee->callback_ret_range,
10303 					       "At callback return", "R0");
10304 			return -EINVAL;
10305 		}
10306 		if (!bpf_calls_callback(env, callee->callsite)) {
10307 			verifier_bug(env, "in callback at %d, callsite %d !calls_callback",
10308 				     *insn_idx, callee->callsite);
10309 			return -EFAULT;
10310 		}
10311 	} else {
10312 		/* return to the caller whatever r0 had in the callee */
10313 		bpf_diag_mod_begin(env, &caller->regs[BPF_REG_0], r0, BPF_DIAG_MOD_WRITE);
10314 		caller->regs[BPF_REG_0] = *r0;
10315 		bpf_diag_mod_end(env);
10316 	}
10317 
10318 	/* for callbacks like bpf_loop or bpf_for_each_map_elem go back to callsite,
10319 	 * there function call logic would reschedule callback visit. If iteration
10320 	 * converges is_state_visited() would prune that visit eventually.
10321 	 */
10322 	in_callback_fn = callee->in_callback_fn;
10323 	if (in_callback_fn)
10324 		*insn_idx = callee->callsite;
10325 	else
10326 		*insn_idx = callee->callsite + 1;
10327 
10328 	if (env->log.level & BPF_LOG_LEVEL) {
10329 		verbose(env, "returning from callee:\n");
10330 		print_verifier_state(env, state, callee->frameno, true);
10331 		verbose(env, "to caller at %d:\n", *insn_idx);
10332 		print_verifier_state(env, state, caller->frameno, true);
10333 	}
10334 	account_processed_insns(env, callee, caller);
10335 	/* clear everything in the callee. In case of exceptional exits using
10336 	 * bpf_throw, this will be done by copy_verifier_state for extra frames. */
10337 	free_func_state(callee);
10338 	state->frame[state->curframe--] = NULL;
10339 	invalidate_outgoing_stack_args(env, caller);
10340 
10341 	/* for callbacks widen imprecise scalars to make programs like below verify:
10342 	 *
10343 	 *   struct ctx { int i; }
10344 	 *   void cb(int idx, struct ctx *ctx) { ctx->i++; ... }
10345 	 *   ...
10346 	 *   struct ctx = { .i = 0; }
10347 	 *   bpf_loop(100, cb, &ctx, 0);
10348 	 *
10349 	 * This is similar to what is done in process_iter_next_call() for open
10350 	 * coded iterators.
10351 	 */
10352 	prev_st = in_callback_fn ? find_prev_entry(env, state, *insn_idx) : NULL;
10353 	if (prev_st) {
10354 		err = widen_imprecise_scalars(env, prev_st, state);
10355 		if (err)
10356 			return err;
10357 	}
10358 	return 0;
10359 }
10360 
do_refine_retval_range(struct bpf_verifier_env * env,struct bpf_reg_state * regs,int ret_type,int func_id,struct bpf_call_arg_meta * meta)10361 static int do_refine_retval_range(struct bpf_verifier_env *env,
10362 				  struct bpf_reg_state *regs, int ret_type,
10363 				  int func_id,
10364 				  struct bpf_call_arg_meta *meta)
10365 {
10366 	struct bpf_retval_range range;
10367 	struct bpf_reg_state *ret_reg = &regs[BPF_REG_0];
10368 	enum bpf_prog_type prog_type = resolve_prog_type(env->prog);
10369 
10370 	if (ret_type != RET_INTEGER)
10371 		return 0;
10372 
10373 	switch (func_id) {
10374 	case BPF_FUNC_get_stack:
10375 	case BPF_FUNC_get_task_stack:
10376 	case BPF_FUNC_probe_read_str:
10377 	case BPF_FUNC_probe_read_kernel_str:
10378 	case BPF_FUNC_probe_read_user_str:
10379 		reg_set_srange64(ret_reg, -MAX_ERRNO, meta->msize_max_value);
10380 		reg_set_srange32(ret_reg, -MAX_ERRNO, meta->msize_max_value);
10381 		reg_bounds_sync(ret_reg);
10382 		break;
10383 	case BPF_FUNC_get_smp_processor_id:
10384 		reg_set_urange64(ret_reg, 0, nr_cpu_ids - 1);
10385 		reg_set_urange32(ret_reg, 0, nr_cpu_ids - 1);
10386 		reg_bounds_sync(ret_reg);
10387 		break;
10388 	case BPF_FUNC_get_retval:
10389 		/*
10390 		 * bpf_get_retval may see arbitrary value passed by bpf_prog_run_array_cg for
10391 		 * CGROUP_GETSOCKOPT type.
10392 		 */
10393 		if (prog_type == BPF_PROG_TYPE_CGROUP_SOCKOPT &&
10394 		    env->prog->expected_attach_type == BPF_CGROUP_GETSOCKOPT)
10395 			break;
10396 
10397 		if (prog_type == BPF_PROG_TYPE_LSM &&
10398 		    env->prog->expected_attach_type == BPF_LSM_CGROUP) {
10399 			if (!env->prog->aux->attach_func_proto->type)
10400 				break;
10401 			bpf_lsm_get_retval_range(env->prog, &range);
10402 		} else {
10403 			range.minval = -MAX_ERRNO;
10404 			range.maxval = 0;
10405 		}
10406 
10407 		reg_set_srange64(ret_reg, range.minval, range.maxval);
10408 		reg_set_srange32(ret_reg, range.minval, range.maxval);
10409 		reg_bounds_sync(ret_reg);
10410 		break;
10411 	}
10412 
10413 	return reg_bounds_sanity_check(env, ret_reg, "retval");
10414 }
10415 
10416 static int
record_func_map(struct bpf_verifier_env * env,struct bpf_call_arg_meta * meta,int func_id,int insn_idx)10417 record_func_map(struct bpf_verifier_env *env, struct bpf_call_arg_meta *meta,
10418 		int func_id, int insn_idx)
10419 {
10420 	struct bpf_insn_aux_data *aux = &env->insn_aux_data[insn_idx];
10421 	struct bpf_map *map = meta->map.ptr;
10422 
10423 	if (func_id != BPF_FUNC_tail_call &&
10424 	    func_id != BPF_FUNC_map_lookup_elem &&
10425 	    func_id != BPF_FUNC_map_update_elem &&
10426 	    func_id != BPF_FUNC_map_delete_elem &&
10427 	    func_id != BPF_FUNC_map_push_elem &&
10428 	    func_id != BPF_FUNC_map_pop_elem &&
10429 	    func_id != BPF_FUNC_map_peek_elem &&
10430 	    func_id != BPF_FUNC_for_each_map_elem &&
10431 	    func_id != BPF_FUNC_redirect_map &&
10432 	    func_id != BPF_FUNC_map_lookup_percpu_elem)
10433 		return 0;
10434 
10435 	if (map == NULL) {
10436 		verifier_bug(env, "expected map for helper call");
10437 		return -EFAULT;
10438 	}
10439 
10440 	/* In case of read-only, some additional restrictions
10441 	 * need to be applied in order to prevent altering the
10442 	 * state of the map from program side.
10443 	 */
10444 	if ((map->map_flags & BPF_F_RDONLY_PROG) &&
10445 	    (func_id == BPF_FUNC_map_delete_elem ||
10446 	     func_id == BPF_FUNC_map_update_elem ||
10447 	     func_id == BPF_FUNC_map_push_elem ||
10448 	     func_id == BPF_FUNC_map_pop_elem)) {
10449 		verbose(env, "write into map forbidden\n");
10450 		return -EACCES;
10451 	}
10452 
10453 	if (!aux->map_ptr_state.map_ptr)
10454 		bpf_map_ptr_store(aux, meta->map.ptr,
10455 				  !meta->map.ptr->bypass_spec_v1, false);
10456 	else if (aux->map_ptr_state.map_ptr != meta->map.ptr)
10457 		bpf_map_ptr_store(aux, meta->map.ptr,
10458 				  !meta->map.ptr->bypass_spec_v1, true);
10459 	return 0;
10460 }
10461 
10462 static int
record_func_key(struct bpf_verifier_env * env,struct bpf_call_arg_meta * meta,int func_id,int insn_idx)10463 record_func_key(struct bpf_verifier_env *env, struct bpf_call_arg_meta *meta,
10464 		int func_id, int insn_idx)
10465 {
10466 	struct bpf_insn_aux_data *aux = &env->insn_aux_data[insn_idx];
10467 	struct bpf_reg_state *reg;
10468 	struct bpf_map *map = meta->map.ptr;
10469 	u64 val, max;
10470 	int err;
10471 
10472 	if (func_id != BPF_FUNC_tail_call)
10473 		return 0;
10474 	if (!map || map->map_type != BPF_MAP_TYPE_PROG_ARRAY) {
10475 		verbose(env, "expected prog array map for tail call");
10476 		return -EINVAL;
10477 	}
10478 
10479 	reg = reg_state(env, BPF_REG_3);
10480 	val = reg->var_off.value;
10481 	max = map->max_entries;
10482 
10483 	if (!(is_reg_const(reg, false) && val < max)) {
10484 		bpf_map_key_store(aux, BPF_MAP_KEY_POISON);
10485 		return 0;
10486 	}
10487 
10488 	err = mark_chain_precision(env, BPF_REG_3);
10489 	if (err)
10490 		return err;
10491 	if (bpf_map_key_unseen(aux))
10492 		bpf_map_key_store(aux, val);
10493 	else if (!bpf_map_key_poisoned(aux) &&
10494 		  bpf_map_key_immediate(aux) != val)
10495 		bpf_map_key_store(aux, BPF_MAP_KEY_POISON);
10496 	return 0;
10497 }
10498 
check_reference_leak(struct bpf_verifier_env * env,bool exception_exit)10499 static int check_reference_leak(struct bpf_verifier_env *env, bool exception_exit)
10500 {
10501 	struct bpf_verifier_state *state = env->cur_state;
10502 	enum bpf_prog_type type = resolve_prog_type(env->prog);
10503 	struct bpf_reg_state *reg = reg_state(env, BPF_REG_0);
10504 	bool refs_lingering = false;
10505 	int i;
10506 
10507 	if (!exception_exit && cur_func(env)->frameno)
10508 		return 0;
10509 
10510 	for (i = 0; i < state->acquired_refs; i++) {
10511 		if (state->refs[i].type != REF_TYPE_PTR)
10512 			continue;
10513 		/* Allow struct_ops programs to return a referenced kptr back to
10514 		 * kernel. Type checks are performed later in check_return_code.
10515 		 */
10516 		if (type == BPF_PROG_TYPE_STRUCT_OPS && !exception_exit &&
10517 		    reg->id == state->refs[i].id)
10518 			continue;
10519 		verbose(env, "Unreleased reference id=%d alloc_insn=%d\n",
10520 			state->refs[i].id, state->refs[i].insn_idx);
10521 		bpf_diag_leak(env, state->refs[i].id, state->refs[i].insn_idx, env->insn_idx);
10522 		refs_lingering = true;
10523 	}
10524 	return refs_lingering ? -EINVAL : 0;
10525 }
10526 
check_resource_leak(struct bpf_verifier_env * env,bool exception_exit,bool check_lock,const char * prefix)10527 static int check_resource_leak(struct bpf_verifier_env *env, bool exception_exit, bool check_lock, const char *prefix)
10528 {
10529 	int err;
10530 
10531 	if (check_lock && env->cur_state->active_locks) {
10532 		verbose(env, "%s cannot be used inside bpf_spin_lock-ed region\n", prefix);
10533 		bpf_diag_ctx_active(env, env->insn_idx, prefix, BPF_DIAG_CONTEXT_LOCK,
10534 				    "Release the BPF spin lock before this operation on every path.");
10535 		return -EINVAL;
10536 	}
10537 
10538 	err = check_reference_leak(env, exception_exit);
10539 	if (err) {
10540 		verbose(env, "%s would lead to reference leak\n", prefix);
10541 		return err;
10542 	}
10543 
10544 	if (check_lock && env->cur_state->active_irq_id) {
10545 		verbose(env, "%s cannot be used inside bpf_local_irq_save-ed region\n", prefix);
10546 		bpf_diag_ctx_active(env, env->insn_idx, prefix, BPF_DIAG_CONTEXT_IRQ,
10547 				    "Restore the saved IRQ state before this operation on every path.");
10548 		return -EINVAL;
10549 	}
10550 
10551 	if (check_lock && env->cur_state->active_rcu_locks) {
10552 		verbose(env, "%s cannot be used inside bpf_rcu_read_lock-ed region\n", prefix);
10553 		bpf_diag_ctx_active(env, env->insn_idx, prefix, BPF_DIAG_CONTEXT_RCU,
10554 				    "Call bpf_rcu_read_unlock() before this operation on every path.");
10555 		return -EINVAL;
10556 	}
10557 
10558 	if (check_lock && env->cur_state->active_preempt_locks) {
10559 		verbose(env, "%s cannot be used inside bpf_preempt_disable-ed region\n", prefix);
10560 		bpf_diag_ctx_active(
10561 			env, env->insn_idx, prefix, BPF_DIAG_CONTEXT_PREEMPT,
10562 			"Call bpf_preempt_enable() before this operation on every path.");
10563 		return -EINVAL;
10564 	}
10565 
10566 	return 0;
10567 }
10568 
check_bpf_snprintf_call(struct bpf_verifier_env * env,struct bpf_reg_state * regs)10569 static int check_bpf_snprintf_call(struct bpf_verifier_env *env,
10570 				   struct bpf_reg_state *regs)
10571 {
10572 	struct bpf_reg_state *fmt_reg = &regs[BPF_REG_3];
10573 	struct bpf_reg_state *data_len_reg = &regs[BPF_REG_5];
10574 	struct bpf_map *fmt_map = fmt_reg->map_ptr;
10575 	struct bpf_bprintf_data data = {};
10576 	int err, fmt_map_off, num_args;
10577 	u64 fmt_addr;
10578 	char *fmt;
10579 
10580 	/* data must be an array of u64 */
10581 	if (data_len_reg->var_off.value % 8)
10582 		return -EINVAL;
10583 	num_args = data_len_reg->var_off.value / 8;
10584 
10585 	/* fmt being ARG_PTR_TO_CONST_STR guarantees that var_off is const
10586 	 * and map_direct_value_addr is set.
10587 	 */
10588 	fmt_map_off = fmt_reg->var_off.value;
10589 	err = fmt_map->ops->map_direct_value_addr(fmt_map, &fmt_addr,
10590 						  fmt_map_off);
10591 	if (err) {
10592 		verbose(env, "failed to retrieve map value address\n");
10593 		return -EFAULT;
10594 	}
10595 	fmt = (char *)(long)fmt_addr + fmt_map_off;
10596 
10597 	/* We are also guaranteed that fmt+fmt_map_off is NULL terminated, we
10598 	 * can focus on validating the format specifiers.
10599 	 */
10600 	err = bpf_bprintf_prepare(fmt, UINT_MAX, NULL, num_args, &data);
10601 	if (err < 0)
10602 		verbose(env, "Invalid format string\n");
10603 
10604 	return err;
10605 }
10606 
check_get_func_ip(struct bpf_verifier_env * env)10607 static int check_get_func_ip(struct bpf_verifier_env *env)
10608 {
10609 	enum bpf_prog_type type = resolve_prog_type(env->prog);
10610 	int func_id = BPF_FUNC_get_func_ip;
10611 
10612 	if (type == BPF_PROG_TYPE_TRACING) {
10613 		if (!bpf_prog_has_trampoline(env->prog)) {
10614 			verbose(env, "func %s#%d supported only for fentry/fexit/fsession/fmod_ret programs\n",
10615 				func_id_name(func_id), func_id);
10616 			return -ENOTSUPP;
10617 		}
10618 		return 0;
10619 	} else if (type == BPF_PROG_TYPE_KPROBE) {
10620 		return 0;
10621 	}
10622 
10623 	verbose(env, "func %s#%d not supported for program type %d\n",
10624 		func_id_name(func_id), func_id, type);
10625 	return -ENOTSUPP;
10626 }
10627 
cur_aux(const struct bpf_verifier_env * env)10628 static struct bpf_insn_aux_data *cur_aux(const struct bpf_verifier_env *env)
10629 {
10630 	return &env->insn_aux_data[env->insn_idx];
10631 }
10632 
loop_flag_is_zero(struct bpf_verifier_env * env)10633 static bool loop_flag_is_zero(struct bpf_verifier_env *env)
10634 {
10635 	struct bpf_reg_state *reg = reg_state(env, BPF_REG_4);
10636 	bool reg_is_null = bpf_register_is_null(reg);
10637 
10638 	if (reg_is_null)
10639 		mark_chain_precision(env, BPF_REG_4);
10640 
10641 	return reg_is_null;
10642 }
10643 
update_loop_inline_state(struct bpf_verifier_env * env,u32 subprogno)10644 static void update_loop_inline_state(struct bpf_verifier_env *env, u32 subprogno)
10645 {
10646 	struct bpf_loop_inline_state *state = &cur_aux(env)->loop_inline_state;
10647 
10648 	if (!state->initialized) {
10649 		state->initialized = 1;
10650 		state->fit_for_inline = loop_flag_is_zero(env);
10651 		state->callback_subprogno = subprogno;
10652 		return;
10653 	}
10654 
10655 	if (!state->fit_for_inline)
10656 		return;
10657 
10658 	state->fit_for_inline = (loop_flag_is_zero(env) &&
10659 				 state->callback_subprogno == subprogno);
10660 }
10661 
10662 /* Returns whether or not the given map can potentially elide
10663  * lookup return value nullness check. This is possible if the key
10664  * is statically known.
10665  */
can_elide_value_nullness(const struct bpf_map * map)10666 static bool can_elide_value_nullness(const struct bpf_map *map)
10667 {
10668 	if (map->map_flags & BPF_F_INNER_MAP)
10669 		return false;
10670 
10671 	switch (map->map_type) {
10672 	case BPF_MAP_TYPE_ARRAY:
10673 	case BPF_MAP_TYPE_PERCPU_ARRAY:
10674 		return true;
10675 	default:
10676 		return false;
10677 	}
10678 }
10679 
bpf_get_helper_proto(struct bpf_verifier_env * env,int func_id,const struct bpf_func_proto ** ptr)10680 int bpf_get_helper_proto(struct bpf_verifier_env *env, int func_id,
10681 			 const struct bpf_func_proto **ptr)
10682 {
10683 	if (func_id < 0 || func_id >= __BPF_FUNC_MAX_ID)
10684 		return -ERANGE;
10685 
10686 	if (!env->ops->get_func_proto)
10687 		return -EINVAL;
10688 
10689 	*ptr = env->ops->get_func_proto(func_id, env->prog);
10690 	return *ptr && (*ptr)->func ? 0 : -EINVAL;
10691 }
10692 
10693 /* Check if we're in a sleepable context. */
in_sleepable_context(struct bpf_verifier_env * env)10694 static inline bool in_sleepable_context(struct bpf_verifier_env *env)
10695 {
10696 	return !env->cur_state->active_rcu_locks &&
10697 	       !env->cur_state->active_preempt_locks &&
10698 	       !env->cur_state->active_locks &&
10699 	       !env->cur_state->active_irq_id &&
10700 	       in_sleepable(env);
10701 }
10702 
non_sleepable_context_description(struct bpf_verifier_env * env)10703 static const char *non_sleepable_context_description(struct bpf_verifier_env *env)
10704 {
10705 	if (env->cur_state->active_rcu_locks)
10706 		return "rcu_read_lock region";
10707 	if (env->cur_state->active_preempt_locks)
10708 		return "non-preemptible region";
10709 	if (env->cur_state->active_irq_id)
10710 		return "IRQ-disabled region";
10711 	if (env->cur_state->active_locks)
10712 		return "lock region";
10713 	return "non-sleepable prog";
10714 }
10715 
release_reg(struct bpf_verifier_env * env,struct bpf_reg_state * reg,bool convert_rcu,bool release_dynptr)10716 static int release_reg(struct bpf_verifier_env *env, struct bpf_reg_state *reg,
10717 		       bool convert_rcu, bool release_dynptr)
10718 {
10719 	int err = -EINVAL;
10720 
10721 	if (bpf_register_is_null(reg))
10722 		return 0;
10723 
10724 	if (release_dynptr)
10725 		err = unmark_stack_slots_dynptr(env, reg);
10726 	else if (convert_rcu)
10727 		err = ref_convert_alloc_rcu_protected(env, reg->id);
10728 	else if (reg_is_referenced(env, reg))
10729 		err = release_reference(env, reg->id);
10730 
10731 	return err;
10732 }
10733 
check_helper_call(struct bpf_verifier_env * env,struct bpf_insn * insn,int * insn_idx_p)10734 static int check_helper_call(struct bpf_verifier_env *env, struct bpf_insn *insn,
10735 			     int *insn_idx_p)
10736 {
10737 	enum bpf_prog_type prog_type = resolve_prog_type(env->prog);
10738 	bool returns_cpu_specific_alloc_ptr = false;
10739 	const struct bpf_func_proto *fn = NULL;
10740 	enum bpf_return_type ret_type;
10741 	enum bpf_type_flag ret_flag;
10742 	struct bpf_reg_state *regs;
10743 	struct bpf_call_arg_meta meta;
10744 	const char *operation;
10745 	int insn_idx = *insn_idx_p;
10746 	bool changes_data;
10747 	int i, err, func_id;
10748 
10749 	/* find function prototype */
10750 	func_id = insn->imm;
10751 	err = bpf_get_helper_proto(env, insn->imm, &fn);
10752 	if (err == -ERANGE) {
10753 		verbose(env, "invalid func %s#%d\n", func_id_name(func_id), func_id);
10754 		return -EINVAL;
10755 	}
10756 
10757 	if (err) {
10758 		verbose(env, "program of this type cannot use helper %s#%d\n",
10759 			func_id_name(func_id), func_id);
10760 		operation = bpf_diag_fmt(env, "helper %s#%d", func_id_name(func_id), func_id);
10761 		bpf_diag_policy(
10762 			env, insn_idx, operation, "this program type does not allow the helper",
10763 			"Use a helper allowed for this program type, or move the logic to a compatible program type.");
10764 		return err;
10765 	}
10766 
10767 	/* eBPF programs must be GPL compatible to use GPL-ed functions */
10768 	if (!env->prog->gpl_compatible && fn->gpl_only) {
10769 		verbose(env, "cannot call GPL-restricted function from non-GPL compatible program\n");
10770 		operation = bpf_diag_fmt(env, "helper %s#%d", func_id_name(func_id), func_id);
10771 		bpf_diag_policy(
10772 			env, insn_idx, operation,
10773 			"this helper is restricted to GPL-compatible programs",
10774 			"Use a GPL-compatible license, or replace the helper with one that is available to non-GPL programs.");
10775 		return -EINVAL;
10776 	}
10777 
10778 	if (fn->allowed && !fn->allowed(env->prog)) {
10779 		verbose(env, "helper call is not allowed in probe\n");
10780 		operation = bpf_diag_fmt(env, "helper %s#%d", func_id_name(func_id), func_id);
10781 		bpf_diag_policy(
10782 			env, insn_idx, operation,
10783 			"the helper-specific policy callback rejected this program",
10784 			"Use the helper only from an allowed attach point or program configuration.");
10785 		return -EINVAL;
10786 	}
10787 
10788 	/* With LD_ABS/IND some JITs save/restore skb from r1. */
10789 	changes_data = bpf_helper_changes_pkt_data(func_id);
10790 	if (changes_data && fn->arg1_type != ARG_PTR_TO_CTX) {
10791 		verifier_bug(env, "func %s#%d: r1 != ctx", func_id_name(func_id), func_id);
10792 		return -EFAULT;
10793 	}
10794 
10795 	memset(&meta, 0, sizeof(meta));
10796 
10797 	err = check_func_proto(fn, &meta);
10798 	if (err) {
10799 		verifier_bug(env, "incorrect func proto %s#%d", func_id_name(func_id), func_id);
10800 		return err;
10801 	}
10802 
10803 	if (fn->might_sleep && !in_sleepable_context(env)) {
10804 		verbose(env, "sleepable helper %s#%d in %s\n", func_id_name(func_id), func_id,
10805 			non_sleepable_context_description(env));
10806 		operation = bpf_diag_fmt(env, "sleepable helper %s#%d",
10807 					 func_id_name(func_id), func_id);
10808 		bpf_diag_ctx_forbidden(env, insn_idx, operation,
10809 			"Move the helper call outside the critical section, or use a non-sleepable helper.");
10810 		return -EINVAL;
10811 	}
10812 
10813 	/* Track non-sleepable context for helpers. */
10814 	if (!in_sleepable_context(env))
10815 		env->insn_aux_data[insn_idx].non_sleepable = true;
10816 
10817 	meta.func_id = func_id;
10818 	meta.fn = fn;
10819 	/* check args */
10820 	for (i = 0; i < MAX_BPF_FUNC_REG_ARGS; i++) {
10821 		err = check_func_arg(env, i, &meta, insn_idx);
10822 		if (err)
10823 			return err;
10824 	}
10825 
10826 	err = record_func_map(env, &meta, func_id, insn_idx);
10827 	if (err)
10828 		return err;
10829 
10830 	err = record_func_key(env, &meta, func_id, insn_idx);
10831 	if (err)
10832 		return err;
10833 
10834 	regs = cur_regs(env);
10835 
10836 	/* Mark slots with STACK_MISC in case of raw mode, stack offset
10837 	 * is inferred from register state.
10838 	 */
10839 	for (i = 0; i < meta.arg_raw_mem.size; i++) {
10840 		err = check_mem_access(env, insn_idx, regs + meta.arg_raw_mem.regno,
10841 				       argno_from_reg(meta.arg_raw_mem.regno), i, BPF_B,
10842 				       BPF_WRITE, -1, false, false);
10843 		if (err)
10844 			return err;
10845 	}
10846 
10847 	if (meta.release_regno) {
10848 		struct bpf_reg_state *reg = &regs[meta.release_regno];
10849 		bool convert_rcu = (func_id == BPF_FUNC_kptr_xchg) && in_rcu_cs(env) &&
10850 				   (reg->type & MEM_ALLOC) && (reg->type & MEM_PERCPU);
10851 
10852 		err = release_reg(env, reg, convert_rcu, !!meta.dynptr.id);
10853 		if (err)
10854 			return err;
10855 	}
10856 
10857 	switch (func_id) {
10858 	case BPF_FUNC_tail_call:
10859 		err = check_resource_leak(env, false, true, "tail_call");
10860 		if (err)
10861 			return err;
10862 		break;
10863 	case BPF_FUNC_get_local_storage:
10864 		/* check that flags argument in get_local_storage(map, flags) is 0,
10865 		 * this is required because get_local_storage() can't return an error.
10866 		 */
10867 		if (!bpf_register_is_null(&regs[BPF_REG_2])) {
10868 			verbose(env, "get_local_storage() doesn't support non-zero flags\n");
10869 			return -EINVAL;
10870 		}
10871 		break;
10872 	case BPF_FUNC_for_each_map_elem:
10873 		err = push_callback_call(env, insn, insn_idx, meta.subprogno,
10874 					 set_map_elem_callback_state);
10875 		break;
10876 	case BPF_FUNC_timer_set_callback:
10877 		err = push_callback_call(env, insn, insn_idx, meta.subprogno,
10878 					 set_timer_callback_state);
10879 		break;
10880 	case BPF_FUNC_find_vma:
10881 		err = push_callback_call(env, insn, insn_idx, meta.subprogno,
10882 					 set_find_vma_callback_state);
10883 		break;
10884 	case BPF_FUNC_snprintf:
10885 		err = check_bpf_snprintf_call(env, regs);
10886 		break;
10887 	case BPF_FUNC_loop:
10888 		update_loop_inline_state(env, meta.subprogno);
10889 		/* Verifier relies on R1 value to determine if bpf_loop() iteration
10890 		 * is finished, thus mark it precise.
10891 		 */
10892 		err = mark_chain_precision(env, BPF_REG_1);
10893 		if (err)
10894 			return err;
10895 		if (cur_func(env)->callback_depth < reg_umax(&regs[BPF_REG_1])) {
10896 			err = push_callback_call(env, insn, insn_idx, meta.subprogno,
10897 						 set_loop_callback_state);
10898 		} else {
10899 			cur_func(env)->callback_depth = 0;
10900 			if (env->log.level & BPF_LOG_LEVEL2)
10901 				verbose(env, "frame%d bpf_loop iteration limit reached\n",
10902 					env->cur_state->curframe);
10903 		}
10904 		break;
10905 	case BPF_FUNC_dynptr_from_mem:
10906 		if (regs[BPF_REG_1].type != PTR_TO_MAP_VALUE) {
10907 			verbose(env, "Unsupported reg type %s for bpf_dynptr_from_mem data\n",
10908 				reg_type_str(env, regs[BPF_REG_1].type));
10909 			return -EACCES;
10910 		}
10911 		break;
10912 	case BPF_FUNC_set_retval:
10913 	{
10914 		struct bpf_retval_range range = {
10915 			.minval = -MAX_ERRNO,
10916 			.maxval = 0,
10917 			.return_32bit = true
10918 		};
10919 		struct bpf_reg_state *r1 = &regs[BPF_REG_1];
10920 
10921 		if (r1->type != SCALAR_VALUE) {
10922 			verbose(env, "R1 is not a scalar\n");
10923 			return -EINVAL;
10924 		}
10925 
10926 		/* CGROUP_GETSOCKOPT is allowed to return arbitrary value */
10927 		if (prog_type == BPF_PROG_TYPE_CGROUP_SOCKOPT &&
10928 		    env->prog->expected_attach_type == BPF_CGROUP_GETSOCKOPT)
10929 			break;
10930 
10931 		if (prog_type == BPF_PROG_TYPE_LSM &&
10932 		    env->prog->expected_attach_type == BPF_LSM_CGROUP) {
10933 			if (!env->prog->aux->attach_func_proto->type) {
10934 				/* Make sure programs that attach to void
10935 				 * hooks don't try to modify return value.
10936 				 */
10937 				verbose(env, "BPF_LSM_CGROUP that attach to void LSM hooks can't modify return value!\n");
10938 				return -EINVAL;
10939 			}
10940 			bpf_lsm_get_retval_range(env->prog, &range);
10941 		}
10942 
10943 		err = mark_chain_precision(env, BPF_REG_1);
10944 		if (err)
10945 			return err;
10946 
10947 		if (!retval_range_within(range, r1)) {
10948 			verbose_invalid_scalar(env, r1, range, "At bpf_set_retval", "R1");
10949 			return -EINVAL;
10950 		}
10951 
10952 		break;
10953 	}
10954 	case BPF_FUNC_dynptr_write:
10955 	{
10956 		enum bpf_dynptr_type dynptr_type = meta.dynptr.type;
10957 
10958 		if (dynptr_type == BPF_DYNPTR_TYPE_INVALID)
10959 			return -EFAULT;
10960 
10961 		if (dynptr_type == BPF_DYNPTR_TYPE_SKB ||
10962 		    dynptr_type == BPF_DYNPTR_TYPE_SKB_META)
10963 			/* this will trigger clear_all_pkt_pointers(), which will
10964 			 * invalidate all dynptr slices associated with the skb
10965 			 */
10966 			changes_data = true;
10967 
10968 		break;
10969 	}
10970 	case BPF_FUNC_per_cpu_ptr:
10971 	case BPF_FUNC_this_cpu_ptr:
10972 	{
10973 		struct bpf_reg_state *reg = &regs[BPF_REG_1];
10974 		const struct btf_type *type;
10975 
10976 		if (reg->type & MEM_RCU) {
10977 			type = btf_type_by_id(reg->btf, reg->btf_id);
10978 			if (!type || !btf_type_is_struct(type)) {
10979 				verbose(env, "Helper has invalid btf/btf_id in R1\n");
10980 				return -EFAULT;
10981 			}
10982 			returns_cpu_specific_alloc_ptr = true;
10983 			env->insn_aux_data[insn_idx].call_with_percpu_alloc_ptr = true;
10984 		}
10985 		break;
10986 	}
10987 	case BPF_FUNC_user_ringbuf_drain:
10988 		err = push_callback_call(env, insn, insn_idx, meta.subprogno,
10989 					 set_user_ringbuf_callback_state);
10990 		break;
10991 	}
10992 
10993 	if (err)
10994 		return err;
10995 
10996 	/* reset caller saved regs */
10997 	bpf_diag_record_caller_saved(env, regs);
10998 	bpf_diag_mod_begin(env, &regs[BPF_REG_0], NULL, BPF_DIAG_MOD_WRITE);
10999 	for (i = 0; i < CALLER_SAVED_REGS; i++) {
11000 		bpf_mark_reg_not_init(env, &regs[caller_saved[i]]);
11001 		check_reg_arg(env, caller_saved[i], DST_OP_NO_MARK);
11002 	}
11003 	invalidate_outgoing_stack_args(env, cur_func(env));
11004 
11005 	/* update return register (already marked as written above) */
11006 	ret_type = fn->ret_type;
11007 	ret_flag = type_flag(ret_type);
11008 
11009 	switch (base_type(ret_type)) {
11010 	case RET_INTEGER:
11011 		/* sets type to SCALAR_VALUE */
11012 		mark_reg_unknown(env, regs, BPF_REG_0);
11013 		break;
11014 	case RET_VOID:
11015 		regs[BPF_REG_0].type = NOT_INIT;
11016 		break;
11017 	case RET_PTR_TO_MAP_VALUE:
11018 		/* There is no offset yet applied, variable or fixed */
11019 		mark_reg_known_zero(env, regs, BPF_REG_0);
11020 		/* remember map_ptr, so that check_map_access()
11021 		 * can check 'value_size' boundary of memory access
11022 		 * to map element returned from bpf_map_lookup_elem()
11023 		 */
11024 		if (meta.map.ptr == NULL) {
11025 			verifier_bug(env, "unexpected null map_ptr");
11026 			return -EFAULT;
11027 		}
11028 
11029 		if (func_id == BPF_FUNC_map_lookup_elem &&
11030 		    can_elide_value_nullness(meta.map.ptr) &&
11031 		    meta.const_map_key >= 0 &&
11032 		    meta.const_map_key < meta.map.ptr->max_entries)
11033 			ret_flag &= ~PTR_MAYBE_NULL;
11034 
11035 		regs[BPF_REG_0].map_ptr = meta.map.ptr;
11036 		regs[BPF_REG_0].map_uid = meta.map.uid;
11037 		regs[BPF_REG_0].type = PTR_TO_MAP_VALUE | ret_flag;
11038 		if (type_may_be_null(ret_flag) ||
11039 		    btf_record_has_field(meta.map.ptr->record, BPF_SPIN_LOCK | BPF_RES_SPIN_LOCK)) {
11040 			regs[BPF_REG_0].id = ++env->id_gen;
11041 		}
11042 		/* requires regs[BPF_REG_0].id to be set because of the map-in-map case */
11043 		refine_map_lookup_value(&regs[BPF_REG_0]);
11044 		break;
11045 	case RET_PTR_TO_SOCKET:
11046 		mark_reg_known_zero(env, regs, BPF_REG_0);
11047 		regs[BPF_REG_0].type = PTR_TO_SOCKET | ret_flag;
11048 		break;
11049 	case RET_PTR_TO_SOCK_COMMON:
11050 		mark_reg_known_zero(env, regs, BPF_REG_0);
11051 		regs[BPF_REG_0].type = PTR_TO_SOCK_COMMON | ret_flag;
11052 		break;
11053 	case RET_PTR_TO_TCP_SOCK:
11054 		mark_reg_known_zero(env, regs, BPF_REG_0);
11055 		regs[BPF_REG_0].type = PTR_TO_TCP_SOCK | ret_flag;
11056 		break;
11057 	case RET_PTR_TO_MEM:
11058 		mark_reg_known_zero(env, regs, BPF_REG_0);
11059 		regs[BPF_REG_0].type = PTR_TO_MEM | ret_flag;
11060 		regs[BPF_REG_0].mem_size = meta.ret_mem.size;
11061 		break;
11062 	case RET_PTR_TO_MEM_OR_BTF_ID:
11063 	{
11064 		const struct btf_type *t;
11065 
11066 		mark_reg_known_zero(env, regs, BPF_REG_0);
11067 		t = btf_type_skip_modifiers(meta.ret_btf, meta.ret_btf_id, NULL);
11068 		if (!btf_type_is_struct(t)) {
11069 			u32 tsize;
11070 			const struct btf_type *ret;
11071 			const char *tname;
11072 
11073 			/* resolve the type size of ksym. */
11074 			ret = btf_resolve_size(meta.ret_btf, t, &tsize);
11075 			if (IS_ERR(ret)) {
11076 				tname = btf_name_by_offset(meta.ret_btf, t->name_off);
11077 				verbose(env, "unable to resolve the size of type '%s': %ld\n",
11078 					tname, PTR_ERR(ret));
11079 				return -EINVAL;
11080 			}
11081 			regs[BPF_REG_0].type = PTR_TO_MEM | ret_flag;
11082 			regs[BPF_REG_0].mem_size = tsize;
11083 		} else {
11084 			if (returns_cpu_specific_alloc_ptr) {
11085 				regs[BPF_REG_0].type = PTR_TO_BTF_ID | MEM_ALLOC | MEM_RCU;
11086 			} else {
11087 				/* MEM_RDONLY may be carried from ret_flag, but it
11088 				 * doesn't apply on PTR_TO_BTF_ID. Fold it, otherwise
11089 				 * it will confuse the check of PTR_TO_BTF_ID in
11090 				 * check_mem_access().
11091 				 */
11092 				ret_flag &= ~MEM_RDONLY;
11093 				regs[BPF_REG_0].type = PTR_TO_BTF_ID | ret_flag;
11094 			}
11095 
11096 			regs[BPF_REG_0].btf = meta.ret_btf;
11097 			regs[BPF_REG_0].btf_id = meta.ret_btf_id;
11098 		}
11099 		break;
11100 	}
11101 	case RET_PTR_TO_BTF_ID:
11102 	{
11103 		struct btf *ret_btf;
11104 		int ret_btf_id;
11105 
11106 		mark_reg_known_zero(env, regs, BPF_REG_0);
11107 		regs[BPF_REG_0].type = PTR_TO_BTF_ID | ret_flag;
11108 		if (func_id == BPF_FUNC_kptr_xchg) {
11109 			ret_btf = meta.kptr_field->kptr.btf;
11110 			ret_btf_id = meta.kptr_field->kptr.btf_id;
11111 			if (!btf_is_kernel(ret_btf)) {
11112 				regs[BPF_REG_0].type |= MEM_ALLOC;
11113 				if (meta.kptr_field->type == BPF_KPTR_PERCPU)
11114 					regs[BPF_REG_0].type |= MEM_PERCPU;
11115 			}
11116 		} else {
11117 			if (fn->ret_btf_id == BPF_PTR_POISON) {
11118 				verifier_bug(env, "func %s has non-overwritten BPF_PTR_POISON return type",
11119 					     func_id_name(func_id));
11120 				return -EFAULT;
11121 			}
11122 			ret_btf = btf_vmlinux;
11123 			ret_btf_id = *fn->ret_btf_id;
11124 		}
11125 		if (ret_btf_id == 0) {
11126 			verbose(env, "invalid return type %u of func %s#%d\n",
11127 				base_type(ret_type), func_id_name(func_id),
11128 				func_id);
11129 			return -EINVAL;
11130 		}
11131 		regs[BPF_REG_0].btf = ret_btf;
11132 		regs[BPF_REG_0].btf_id = ret_btf_id;
11133 		break;
11134 	}
11135 	default:
11136 		verbose(env, "unknown return type %u of func %s#%d\n",
11137 			base_type(ret_type), func_id_name(func_id), func_id);
11138 		return -EINVAL;
11139 	}
11140 
11141 	if (type_may_be_null(regs[BPF_REG_0].type) && !regs[BPF_REG_0].id)
11142 		regs[BPF_REG_0].id = ++env->id_gen;
11143 
11144 	if (is_ptr_cast_function(func_id) &&
11145 	    find_reference_state(env->cur_state, meta.ref_obj.id)) {
11146 		struct bpf_verifier_state *branch;
11147 		struct bpf_reg_state *r0;
11148 
11149 		err = validate_ref_obj(env, &meta.ref_obj);
11150 		if (err)
11151 			return err;
11152 
11153 		bpf_diag_mod_end(env);
11154 
11155 		/*
11156 		 * In order for a release of any of the original or cast pointers
11157 		 * to invalidate all other pointers, reuse the same reference id for
11158 		 * the cast result.
11159 		 * This reference id can't be used for nullness propagation,
11160 		 * as cast might return NULL for a non-NULL input.
11161 		 * Hence, explore the NULL case as a separate branch.
11162 		 */
11163 		branch = push_stack(env, env->insn_idx + 1, env->insn_idx, false);
11164 		if (IS_ERR(branch))
11165 			return PTR_ERR(branch);
11166 
11167 		r0 = &branch->frame[branch->curframe]->regs[BPF_REG_0];
11168 		__mark_reg_known_zero(r0);
11169 		r0->type = SCALAR_VALUE;
11170 
11171 		bpf_diag_mod_begin(env, &regs[BPF_REG_0], NULL, BPF_DIAG_MOD_WRITE);
11172 		regs[BPF_REG_0].type &= ~PTR_MAYBE_NULL;
11173 		regs[BPF_REG_0].id = meta.ref_obj.id;
11174 	} else if (is_acquire_function(func_id, meta.map.ptr)) {
11175 		int id = acquire_reference(env, insn_idx, 0);
11176 
11177 		if (id < 0)
11178 			return id;
11179 
11180 		regs[BPF_REG_0].id = id;
11181 	}
11182 
11183 	if (func_id == BPF_FUNC_dynptr_data)
11184 		regs[BPF_REG_0].parent_id = meta.dynptr.id;
11185 
11186 	err = do_refine_retval_range(env, regs, fn->ret_type, func_id, &meta);
11187 	if (err)
11188 		return err;
11189 
11190 	bpf_diag_mod_end(env);
11191 
11192 	err = check_map_func_compatibility(env, meta.map.ptr, func_id);
11193 	if (err)
11194 		return err;
11195 
11196 	if ((func_id == BPF_FUNC_get_stack ||
11197 	     func_id == BPF_FUNC_get_task_stack) &&
11198 	    !env->prog->has_callchain_buf) {
11199 		const char *err_str;
11200 
11201 #ifdef CONFIG_PERF_EVENTS
11202 		err = get_callchain_buffers(sysctl_perf_event_max_stack);
11203 		err_str = "cannot get callchain buffer for func %s#%d\n";
11204 #else
11205 		err = -ENOTSUPP;
11206 		err_str = "func %s#%d not supported without CONFIG_PERF_EVENTS\n";
11207 #endif
11208 		if (err) {
11209 			verbose(env, err_str, func_id_name(func_id), func_id);
11210 			return err;
11211 		}
11212 
11213 		env->prog->has_callchain_buf = true;
11214 	}
11215 
11216 	if (func_id == BPF_FUNC_get_stackid || func_id == BPF_FUNC_get_stack)
11217 		env->prog->call_get_stack = true;
11218 
11219 	if (func_id == BPF_FUNC_get_func_ip) {
11220 		if (check_get_func_ip(env))
11221 			return -ENOTSUPP;
11222 		env->prog->call_get_func_ip = true;
11223 	}
11224 
11225 	if (func_id == BPF_FUNC_tail_call) {
11226 		if (env->cur_state->curframe) {
11227 			struct bpf_verifier_state *branch;
11228 
11229 			mark_reg_scratched(env, BPF_REG_0);
11230 			branch = push_stack(env, env->insn_idx + 1, env->insn_idx, false);
11231 			if (IS_ERR(branch))
11232 				return PTR_ERR(branch);
11233 			clear_all_pkt_pointers(env);
11234 			mark_reg_unknown(env, regs, BPF_REG_0);
11235 			err = prepare_func_exit(env, &env->insn_idx);
11236 			if (err)
11237 				return err;
11238 			env->insn_idx--;
11239 		} else {
11240 			changes_data = false;
11241 		}
11242 	}
11243 
11244 	if (changes_data)
11245 		clear_all_pkt_pointers(env);
11246 	return 0;
11247 }
11248 
is_kfunc_acquire(struct bpf_call_arg_meta * meta)11249 static bool is_kfunc_acquire(struct bpf_call_arg_meta *meta)
11250 {
11251 	return meta->kfunc_flags & KF_ACQUIRE;
11252 }
11253 
is_kfunc_release(struct bpf_call_arg_meta * meta)11254 static bool is_kfunc_release(struct bpf_call_arg_meta *meta)
11255 {
11256 	return meta->kfunc_flags & KF_RELEASE;
11257 }
11258 
is_kfunc_destructive(struct bpf_call_arg_meta * meta)11259 static bool is_kfunc_destructive(struct bpf_call_arg_meta *meta)
11260 {
11261 	return meta->kfunc_flags & KF_DESTRUCTIVE;
11262 }
11263 
is_kfunc_rcu(struct bpf_call_arg_meta * meta)11264 static bool is_kfunc_rcu(struct bpf_call_arg_meta *meta)
11265 {
11266 	return meta->kfunc_flags & KF_RCU;
11267 }
11268 
is_kfunc_rcu_protected(struct bpf_call_arg_meta * meta)11269 static bool is_kfunc_rcu_protected(struct bpf_call_arg_meta *meta)
11270 {
11271 	return meta->kfunc_flags & KF_RCU_PROTECTED;
11272 }
11273 
is_kfunc_arg_mem_size(const struct btf * btf,const struct btf_param * arg)11274 static bool is_kfunc_arg_mem_size(const struct btf *btf,
11275 				  const struct btf_param *arg)
11276 {
11277 	const struct btf_type *t;
11278 
11279 	t = btf_type_skip_modifiers(btf, arg->type, NULL);
11280 	if (!btf_type_is_scalar(t))
11281 		return false;
11282 
11283 	return btf_param_match_suffix(btf, arg, "__sz");
11284 }
11285 
is_kfunc_arg_const_mem_size(const struct btf * btf,const struct btf_param * arg)11286 static bool is_kfunc_arg_const_mem_size(const struct btf *btf,
11287 					const struct btf_param *arg)
11288 {
11289 	const struct btf_type *t;
11290 
11291 	t = btf_type_skip_modifiers(btf, arg->type, NULL);
11292 	if (!btf_type_is_scalar(t))
11293 		return false;
11294 
11295 	return btf_param_match_suffix(btf, arg, "__szk");
11296 }
11297 
is_kfunc_arg_constant(const struct btf * btf,const struct btf_param * arg)11298 static bool is_kfunc_arg_constant(const struct btf *btf, const struct btf_param *arg)
11299 {
11300 	return btf_param_match_suffix(btf, arg, "__k");
11301 }
11302 
is_kfunc_arg_ignore(const struct btf * btf,const struct btf_param * arg)11303 static bool is_kfunc_arg_ignore(const struct btf *btf, const struct btf_param *arg)
11304 {
11305 	return btf_param_match_suffix(btf, arg, "__ign");
11306 }
11307 
is_kfunc_arg_map(const struct btf * btf,const struct btf_param * arg)11308 static bool is_kfunc_arg_map(const struct btf *btf, const struct btf_param *arg)
11309 {
11310 	return btf_param_match_suffix(btf, arg, "__map");
11311 }
11312 
is_kfunc_arg_const_map(const struct btf * btf,const struct btf_param * arg)11313 static bool is_kfunc_arg_const_map(const struct btf *btf, const struct btf_param *arg)
11314 {
11315 	return btf_param_match_suffix(btf, arg, "__const_map");
11316 }
11317 
is_kfunc_arg_alloc_obj(const struct btf * btf,const struct btf_param * arg)11318 static bool is_kfunc_arg_alloc_obj(const struct btf *btf, const struct btf_param *arg)
11319 {
11320 	return btf_param_match_suffix(btf, arg, "__alloc");
11321 }
11322 
is_kfunc_arg_uninit(const struct btf * btf,const struct btf_param * arg)11323 static bool is_kfunc_arg_uninit(const struct btf *btf, const struct btf_param *arg)
11324 {
11325 	return btf_param_match_suffix(btf, arg, "__uninit");
11326 }
11327 
is_kfunc_arg_refcounted_kptr(const struct btf * btf,const struct btf_param * arg)11328 static bool is_kfunc_arg_refcounted_kptr(const struct btf *btf, const struct btf_param *arg)
11329 {
11330 	return btf_param_match_suffix(btf, arg, "__refcounted_kptr");
11331 }
11332 
is_kfunc_arg_nullable(const struct btf * btf,const struct btf_param * arg)11333 static bool is_kfunc_arg_nullable(const struct btf *btf, const struct btf_param *arg)
11334 {
11335 	return btf_param_match_suffix(btf, arg, "__nullable") ||
11336 	       btf_param_match_suffix(btf, arg, "__arena");
11337 }
11338 
is_kfunc_arg_nonown_allowed(const struct btf * btf,const struct btf_param * arg)11339 static bool is_kfunc_arg_nonown_allowed(const struct btf *btf, const struct btf_param *arg)
11340 {
11341 	return btf_param_match_suffix(btf, arg, "__nonown_allowed");
11342 }
11343 
is_kfunc_arg_const_str(const struct btf * btf,const struct btf_param * arg)11344 static bool is_kfunc_arg_const_str(const struct btf *btf, const struct btf_param *arg)
11345 {
11346 	return btf_param_match_suffix(btf, arg, "__str");
11347 }
11348 
is_kfunc_arg_irq_flag(const struct btf * btf,const struct btf_param * arg)11349 static bool is_kfunc_arg_irq_flag(const struct btf *btf, const struct btf_param *arg)
11350 {
11351 	return btf_param_match_suffix(btf, arg, "__irq_flag");
11352 }
11353 
is_kfunc_arg_arena(const struct btf * btf,const struct btf_param * arg)11354 static bool is_kfunc_arg_arena(const struct btf *btf, const struct btf_param *arg)
11355 {
11356 	return btf_param_match_suffix(btf, arg, "__arena__nullable") ||
11357 	       btf_param_match_suffix(btf, arg, "__arena");
11358 }
11359 
is_kfunc_arg_scalar_with_name(const struct btf * btf,const struct btf_param * arg,const char * name)11360 static bool is_kfunc_arg_scalar_with_name(const struct btf *btf,
11361 					  const struct btf_param *arg,
11362 					  const char *name)
11363 {
11364 	int len, target_len = strlen(name);
11365 	const char *param_name;
11366 
11367 	param_name = btf_name_by_offset(btf, arg->name_off);
11368 	if (str_is_empty(param_name))
11369 		return false;
11370 	len = strlen(param_name);
11371 	if (len != target_len)
11372 		return false;
11373 	if (strcmp(param_name, name))
11374 		return false;
11375 
11376 	return true;
11377 }
11378 
11379 enum {
11380 	KF_ARG_DYNPTR_ID,
11381 	KF_ARG_LIST_HEAD_ID,
11382 	KF_ARG_LIST_NODE_ID,
11383 	KF_ARG_RB_ROOT_ID,
11384 	KF_ARG_RB_NODE_ID,
11385 	KF_ARG_WORKQUEUE_ID,
11386 	KF_ARG_RES_SPIN_LOCK_ID,
11387 	KF_ARG_TASK_WORK_ID,
11388 	KF_ARG_PROG_AUX_ID,
11389 	KF_ARG_TIMER_ID
11390 };
11391 
11392 BTF_ID_LIST(kf_arg_btf_ids)
BTF_ID(struct,bpf_dynptr)11393 BTF_ID(struct, bpf_dynptr)
11394 BTF_ID(struct, bpf_list_head)
11395 BTF_ID(struct, bpf_list_node)
11396 BTF_ID(struct, bpf_rb_root)
11397 BTF_ID(struct, bpf_rb_node)
11398 BTF_ID(struct, bpf_wq)
11399 BTF_ID(struct, bpf_res_spin_lock)
11400 BTF_ID(struct, bpf_task_work)
11401 BTF_ID(struct, bpf_prog_aux)
11402 BTF_ID(struct, bpf_timer)
11403 
11404 static bool __is_kfunc_ptr_arg_type(const struct btf *btf,
11405 				    const struct btf_param *arg, int type)
11406 {
11407 	const struct btf_type *t;
11408 	u32 res_id;
11409 
11410 	t = btf_type_skip_modifiers(btf, arg->type, NULL);
11411 	if (!t)
11412 		return false;
11413 	if (!btf_type_is_ptr(t))
11414 		return false;
11415 	t = btf_type_skip_modifiers(btf, t->type, &res_id);
11416 	if (!t)
11417 		return false;
11418 	return btf_types_are_same(btf, res_id, btf_vmlinux, kf_arg_btf_ids[type]);
11419 }
11420 
is_kfunc_arg_dynptr(const struct btf * btf,const struct btf_param * arg)11421 static bool is_kfunc_arg_dynptr(const struct btf *btf, const struct btf_param *arg)
11422 {
11423 	return __is_kfunc_ptr_arg_type(btf, arg, KF_ARG_DYNPTR_ID);
11424 }
11425 
is_kfunc_arg_list_head(const struct btf * btf,const struct btf_param * arg)11426 static bool is_kfunc_arg_list_head(const struct btf *btf, const struct btf_param *arg)
11427 {
11428 	return __is_kfunc_ptr_arg_type(btf, arg, KF_ARG_LIST_HEAD_ID);
11429 }
11430 
is_kfunc_arg_list_node(const struct btf * btf,const struct btf_param * arg)11431 static bool is_kfunc_arg_list_node(const struct btf *btf, const struct btf_param *arg)
11432 {
11433 	return __is_kfunc_ptr_arg_type(btf, arg, KF_ARG_LIST_NODE_ID);
11434 }
11435 
is_kfunc_arg_rbtree_root(const struct btf * btf,const struct btf_param * arg)11436 static bool is_kfunc_arg_rbtree_root(const struct btf *btf, const struct btf_param *arg)
11437 {
11438 	return __is_kfunc_ptr_arg_type(btf, arg, KF_ARG_RB_ROOT_ID);
11439 }
11440 
is_kfunc_arg_rbtree_node(const struct btf * btf,const struct btf_param * arg)11441 static bool is_kfunc_arg_rbtree_node(const struct btf *btf, const struct btf_param *arg)
11442 {
11443 	return __is_kfunc_ptr_arg_type(btf, arg, KF_ARG_RB_NODE_ID);
11444 }
11445 
is_kfunc_arg_timer(const struct btf * btf,const struct btf_param * arg)11446 static bool is_kfunc_arg_timer(const struct btf *btf, const struct btf_param *arg)
11447 {
11448 	return __is_kfunc_ptr_arg_type(btf, arg, KF_ARG_TIMER_ID);
11449 }
11450 
is_kfunc_arg_wq(const struct btf * btf,const struct btf_param * arg)11451 static bool is_kfunc_arg_wq(const struct btf *btf, const struct btf_param *arg)
11452 {
11453 	return __is_kfunc_ptr_arg_type(btf, arg, KF_ARG_WORKQUEUE_ID);
11454 }
11455 
is_kfunc_arg_task_work(const struct btf * btf,const struct btf_param * arg)11456 static bool is_kfunc_arg_task_work(const struct btf *btf, const struct btf_param *arg)
11457 {
11458 	return __is_kfunc_ptr_arg_type(btf, arg, KF_ARG_TASK_WORK_ID);
11459 }
11460 
is_kfunc_arg_res_spin_lock(const struct btf * btf,const struct btf_param * arg)11461 static bool is_kfunc_arg_res_spin_lock(const struct btf *btf, const struct btf_param *arg)
11462 {
11463 	return __is_kfunc_ptr_arg_type(btf, arg, KF_ARG_RES_SPIN_LOCK_ID);
11464 }
11465 
is_rbtree_node_type(const struct btf_type * t)11466 static bool is_rbtree_node_type(const struct btf_type *t)
11467 {
11468 	return t == btf_type_by_id(btf_vmlinux, kf_arg_btf_ids[KF_ARG_RB_NODE_ID]);
11469 }
11470 
is_list_node_type(const struct btf_type * t)11471 static bool is_list_node_type(const struct btf_type *t)
11472 {
11473 	return t == btf_type_by_id(btf_vmlinux, kf_arg_btf_ids[KF_ARG_LIST_NODE_ID]);
11474 }
11475 
is_kfunc_arg_callback(struct bpf_verifier_env * env,const struct btf * btf,const struct btf_param * arg)11476 static bool is_kfunc_arg_callback(struct bpf_verifier_env *env, const struct btf *btf,
11477 				  const struct btf_param *arg)
11478 {
11479 	const struct btf_type *t;
11480 
11481 	t = btf_type_resolve_func_ptr(btf, arg->type, NULL);
11482 	if (!t)
11483 		return false;
11484 
11485 	return true;
11486 }
11487 
is_kfunc_arg_prog_aux(const struct btf * btf,const struct btf_param * arg)11488 static bool is_kfunc_arg_prog_aux(const struct btf *btf, const struct btf_param *arg)
11489 {
11490 	return __is_kfunc_ptr_arg_type(btf, arg, KF_ARG_PROG_AUX_ID);
11491 }
11492 
11493 /*
11494  * A kfunc with KF_IMPLICIT_ARGS has two prototypes in BTF:
11495  *   - the _impl prototype with full arg list (meta->func_proto)
11496  *   - the BPF API prototype w/o implicit args (func->type in BTF)
11497  * To determine whether an argument is implicit, we compare its position
11498  * against the number of arguments in the prototype w/o implicit args.
11499  */
is_kfunc_arg_implicit(const struct bpf_call_arg_meta * meta,u32 arg_idx)11500 static bool is_kfunc_arg_implicit(const struct bpf_call_arg_meta *meta, u32 arg_idx)
11501 {
11502 	const struct btf_type *func, *func_proto;
11503 	u32 argn;
11504 
11505 	if (!(meta->kfunc_flags & KF_IMPLICIT_ARGS))
11506 		return false;
11507 
11508 	func = btf_type_by_id(meta->btf, meta->func_id);
11509 	func_proto = btf_type_by_id(meta->btf, func->type);
11510 	argn = btf_type_vlen(func_proto);
11511 
11512 	return argn <= arg_idx;
11513 }
11514 
11515 /* Returns true if struct is composed of scalars, 4 levels of nesting allowed */
__btf_type_is_scalar_struct(struct bpf_verifier_env * env,const struct btf * btf,const struct btf_type * t,int rec)11516 static bool __btf_type_is_scalar_struct(struct bpf_verifier_env *env,
11517 					const struct btf *btf,
11518 					const struct btf_type *t, int rec)
11519 {
11520 	const struct btf_type *member_type;
11521 	const struct btf_member *member;
11522 	u32 i;
11523 
11524 	if (!btf_type_is_struct(t))
11525 		return false;
11526 
11527 	for_each_member(i, t, member) {
11528 		const struct btf_array *array;
11529 
11530 		member_type = btf_type_skip_modifiers(btf, member->type, NULL);
11531 		if (btf_type_is_struct(member_type)) {
11532 			if (rec >= 3) {
11533 				verbose(env, "max struct nesting depth exceeded\n");
11534 				return false;
11535 			}
11536 			if (!__btf_type_is_scalar_struct(env, btf, member_type, rec + 1))
11537 				return false;
11538 			continue;
11539 		}
11540 		if (btf_type_is_array(member_type)) {
11541 			array = btf_array(member_type);
11542 			if (!array->nelems)
11543 				return false;
11544 			member_type = btf_type_skip_modifiers(btf, array->type, NULL);
11545 			if (!btf_type_is_scalar(member_type))
11546 				return false;
11547 			continue;
11548 		}
11549 		if (!btf_type_is_scalar(member_type))
11550 			return false;
11551 	}
11552 	return true;
11553 }
11554 
11555 enum kfunc_ptr_arg_type {
11556 	KF_ARG_CONST_MEM_SIZE,
11557 	KF_ARG_MEM_SIZE,
11558 	KF_ARG_CONST,
11559 	KF_ARG_CONST_ALLOC_SIZE_OR_ZERO,
11560 	KF_ARG_ANYTHING,
11561 	KF_ARG_PTR_TO_CTX,
11562 	KF_ARG_PTR_TO_ALLOC_BTF_ID,    /* Allocated object */
11563 	KF_ARG_PTR_TO_REFCOUNTED_KPTR, /* Refcounted local kptr */
11564 	KF_ARG_PTR_TO_DYNPTR,
11565 	KF_ARG_PTR_TO_ITER,
11566 	KF_ARG_PTR_TO_LIST_HEAD,
11567 	KF_ARG_PTR_TO_LIST_NODE,
11568 	KF_ARG_PTR_TO_BTF_ID,	       /* Also covers reg2btf_ids conversions */
11569 	KF_ARG_PTR_TO_MEM,
11570 	KF_ARG_PTR_TO_CALLBACK,
11571 	KF_ARG_PTR_TO_RB_ROOT,
11572 	KF_ARG_PTR_TO_RB_NODE,
11573 	KF_ARG_PTR_TO_CONST_STR,
11574 	KF_ARG_CONST_MAP_PTR,
11575 	KF_ARG_PTR_TO_TIMER,
11576 	KF_ARG_PTR_TO_WORKQUEUE,
11577 	KF_ARG_PTR_TO_IRQ_FLAG,
11578 	KF_ARG_PTR_TO_RES_SPIN_LOCK,
11579 	KF_ARG_PTR_TO_TASK_WORK,
11580 	KF_ARG_PTR_TO_ARENA,
11581 };
11582 
11583 enum special_kfunc_type {
11584 	KF_bpf_obj_new_impl,
11585 	KF_bpf_obj_new,
11586 	KF_bpf_obj_drop_impl,
11587 	KF_bpf_obj_drop,
11588 	KF_bpf_refcount_acquire_impl,
11589 	KF_bpf_refcount_acquire,
11590 	KF_bpf_list_push_front_impl,
11591 	KF_bpf_list_push_front,
11592 	KF_bpf_list_push_back_impl,
11593 	KF_bpf_list_push_back,
11594 	KF_bpf_list_add,
11595 	KF_bpf_list_pop_front,
11596 	KF_bpf_list_pop_back,
11597 	KF_bpf_list_del,
11598 	KF_bpf_list_front,
11599 	KF_bpf_list_back,
11600 	KF_bpf_list_is_first,
11601 	KF_bpf_list_is_last,
11602 	KF_bpf_list_empty,
11603 	KF_bpf_cast_to_kern_ctx,
11604 	KF_bpf_rdonly_cast,
11605 	KF_bpf_rcu_read_lock,
11606 	KF_bpf_rcu_read_unlock,
11607 	KF_bpf_rbtree_remove,
11608 	KF_bpf_rbtree_add_impl,
11609 	KF_bpf_rbtree_add,
11610 	KF_bpf_rbtree_first,
11611 	KF_bpf_rbtree_root,
11612 	KF_bpf_rbtree_left,
11613 	KF_bpf_rbtree_right,
11614 	KF_bpf_dynptr_from_skb,
11615 	KF_bpf_dynptr_from_xdp,
11616 	KF_bpf_dynptr_from_skb_meta,
11617 	KF_bpf_xdp_pull_data,
11618 	KF_bpf_dynptr_slice,
11619 	KF_bpf_dynptr_slice_rdwr,
11620 	KF_bpf_dynptr_clone,
11621 	KF_bpf_percpu_obj_new_impl,
11622 	KF_bpf_percpu_obj_new,
11623 	KF_bpf_percpu_obj_drop_impl,
11624 	KF_bpf_percpu_obj_drop,
11625 	KF_bpf_throw,
11626 	KF_bpf_wq_set_callback,
11627 	KF_bpf_preempt_disable,
11628 	KF_bpf_preempt_enable,
11629 	KF_bpf_iter_css_task_new,
11630 	KF_bpf_session_cookie,
11631 	KF_bpf_get_kmem_cache,
11632 	KF_bpf_local_irq_save,
11633 	KF_bpf_local_irq_restore,
11634 	KF_bpf_iter_num_new,
11635 	KF_bpf_iter_num_next,
11636 	KF_bpf_iter_num_destroy,
11637 	KF_bpf_set_dentry_xattr,
11638 	KF_bpf_remove_dentry_xattr,
11639 	KF_bpf_res_spin_lock,
11640 	KF_bpf_res_spin_unlock,
11641 	KF_bpf_res_spin_lock_irqsave,
11642 	KF_bpf_res_spin_unlock_irqrestore,
11643 	KF_bpf_dynptr_from_file,
11644 	KF_bpf_dynptr_file_discard,
11645 	KF___bpf_trap,
11646 	KF_bpf_task_work_schedule_signal,
11647 	KF_bpf_task_work_schedule_resume,
11648 	KF_bpf_arena_alloc_pages,
11649 	KF_bpf_arena_free_pages,
11650 	KF_bpf_session_is_return,
11651 };
11652 
11653 BTF_ID_LIST(special_kfunc_list)
BTF_ID(func,bpf_obj_new_impl)11654 BTF_ID(func, bpf_obj_new_impl)
11655 BTF_ID(func, bpf_obj_new)
11656 BTF_ID(func, bpf_obj_drop_impl)
11657 BTF_ID(func, bpf_obj_drop)
11658 BTF_ID(func, bpf_refcount_acquire_impl)
11659 BTF_ID(func, bpf_refcount_acquire)
11660 BTF_ID(func, bpf_list_push_front_impl)
11661 BTF_ID(func, bpf_list_push_front)
11662 BTF_ID(func, bpf_list_push_back_impl)
11663 BTF_ID(func, bpf_list_push_back)
11664 BTF_ID(func, bpf_list_add)
11665 BTF_ID(func, bpf_list_pop_front)
11666 BTF_ID(func, bpf_list_pop_back)
11667 BTF_ID(func, bpf_list_del)
11668 BTF_ID(func, bpf_list_front)
11669 BTF_ID(func, bpf_list_back)
11670 BTF_ID(func, bpf_list_is_first)
11671 BTF_ID(func, bpf_list_is_last)
11672 BTF_ID(func, bpf_list_empty)
11673 BTF_ID(func, bpf_cast_to_kern_ctx)
11674 BTF_ID(func, bpf_rdonly_cast)
11675 BTF_ID(func, bpf_rcu_read_lock)
11676 BTF_ID(func, bpf_rcu_read_unlock)
11677 BTF_ID(func, bpf_rbtree_remove)
11678 BTF_ID(func, bpf_rbtree_add_impl)
11679 BTF_ID(func, bpf_rbtree_add)
11680 BTF_ID(func, bpf_rbtree_first)
11681 BTF_ID(func, bpf_rbtree_root)
11682 BTF_ID(func, bpf_rbtree_left)
11683 BTF_ID(func, bpf_rbtree_right)
11684 #ifdef CONFIG_NET
11685 BTF_ID(func, bpf_dynptr_from_skb)
11686 BTF_ID(func, bpf_dynptr_from_xdp)
11687 BTF_ID(func, bpf_dynptr_from_skb_meta)
11688 BTF_ID(func, bpf_xdp_pull_data)
11689 #else
11690 BTF_ID_UNUSED
11691 BTF_ID_UNUSED
11692 BTF_ID_UNUSED
11693 BTF_ID_UNUSED
11694 #endif
11695 BTF_ID(func, bpf_dynptr_slice)
11696 BTF_ID(func, bpf_dynptr_slice_rdwr)
11697 BTF_ID(func, bpf_dynptr_clone)
11698 BTF_ID(func, bpf_percpu_obj_new_impl)
11699 BTF_ID(func, bpf_percpu_obj_new)
11700 BTF_ID(func, bpf_percpu_obj_drop_impl)
11701 BTF_ID(func, bpf_percpu_obj_drop)
11702 BTF_ID(func, bpf_throw)
11703 BTF_ID(func, bpf_wq_set_callback)
11704 BTF_ID(func, bpf_preempt_disable)
11705 BTF_ID(func, bpf_preempt_enable)
11706 #ifdef CONFIG_CGROUPS
11707 BTF_ID(func, bpf_iter_css_task_new)
11708 #else
11709 BTF_ID_UNUSED
11710 #endif
11711 #ifdef CONFIG_BPF_EVENTS
11712 BTF_ID(func, bpf_session_cookie)
11713 #else
11714 BTF_ID_UNUSED
11715 #endif
11716 BTF_ID(func, bpf_get_kmem_cache)
11717 BTF_ID(func, bpf_local_irq_save)
11718 BTF_ID(func, bpf_local_irq_restore)
11719 BTF_ID(func, bpf_iter_num_new)
11720 BTF_ID(func, bpf_iter_num_next)
11721 BTF_ID(func, bpf_iter_num_destroy)
11722 #ifdef CONFIG_BPF_LSM
11723 BTF_ID(func, bpf_set_dentry_xattr)
11724 BTF_ID(func, bpf_remove_dentry_xattr)
11725 #else
11726 BTF_ID_UNUSED
11727 BTF_ID_UNUSED
11728 #endif
11729 BTF_ID(func, bpf_res_spin_lock)
11730 BTF_ID(func, bpf_res_spin_unlock)
11731 BTF_ID(func, bpf_res_spin_lock_irqsave)
11732 BTF_ID(func, bpf_res_spin_unlock_irqrestore)
11733 BTF_ID(func, bpf_dynptr_from_file)
11734 BTF_ID(func, bpf_dynptr_file_discard)
11735 BTF_ID(func, __bpf_trap)
11736 BTF_ID(func, bpf_task_work_schedule_signal)
11737 BTF_ID(func, bpf_task_work_schedule_resume)
11738 BTF_ID(func, bpf_arena_alloc_pages)
11739 BTF_ID(func, bpf_arena_free_pages)
11740 #ifdef CONFIG_BPF_EVENTS
11741 BTF_ID(func, bpf_session_is_return)
11742 #else
11743 BTF_ID_UNUSED
11744 #endif
11745 
11746 static bool is_bpf_obj_new_kfunc(u32 func_id)
11747 {
11748 	return func_id == special_kfunc_list[KF_bpf_obj_new] ||
11749 	       func_id == special_kfunc_list[KF_bpf_obj_new_impl];
11750 }
11751 
is_bpf_percpu_obj_new_kfunc(u32 func_id)11752 static bool is_bpf_percpu_obj_new_kfunc(u32 func_id)
11753 {
11754 	return func_id == special_kfunc_list[KF_bpf_percpu_obj_new] ||
11755 	       func_id == special_kfunc_list[KF_bpf_percpu_obj_new_impl];
11756 }
11757 
is_bpf_obj_drop_kfunc(u32 func_id)11758 static bool is_bpf_obj_drop_kfunc(u32 func_id)
11759 {
11760 	return func_id == special_kfunc_list[KF_bpf_obj_drop] ||
11761 	       func_id == special_kfunc_list[KF_bpf_obj_drop_impl];
11762 }
11763 
is_bpf_percpu_obj_drop_kfunc(u32 func_id)11764 static bool is_bpf_percpu_obj_drop_kfunc(u32 func_id)
11765 {
11766 	return func_id == special_kfunc_list[KF_bpf_percpu_obj_drop] ||
11767 	       func_id == special_kfunc_list[KF_bpf_percpu_obj_drop_impl];
11768 }
11769 
is_bpf_refcount_acquire_kfunc(u32 func_id)11770 static bool is_bpf_refcount_acquire_kfunc(u32 func_id)
11771 {
11772 	return func_id == special_kfunc_list[KF_bpf_refcount_acquire] ||
11773 	       func_id == special_kfunc_list[KF_bpf_refcount_acquire_impl];
11774 }
11775 
is_bpf_list_push_kfunc(u32 func_id)11776 static bool is_bpf_list_push_kfunc(u32 func_id)
11777 {
11778 	return func_id == special_kfunc_list[KF_bpf_list_push_front] ||
11779 	       func_id == special_kfunc_list[KF_bpf_list_push_front_impl] ||
11780 	       func_id == special_kfunc_list[KF_bpf_list_push_back] ||
11781 	       func_id == special_kfunc_list[KF_bpf_list_push_back_impl] ||
11782 	       func_id == special_kfunc_list[KF_bpf_list_add];
11783 }
11784 
is_bpf_rbtree_add_kfunc(u32 func_id)11785 static bool is_bpf_rbtree_add_kfunc(u32 func_id)
11786 {
11787 	return func_id == special_kfunc_list[KF_bpf_rbtree_add] ||
11788 	       func_id == special_kfunc_list[KF_bpf_rbtree_add_impl];
11789 }
11790 
is_task_work_add_kfunc(u32 func_id)11791 static bool is_task_work_add_kfunc(u32 func_id)
11792 {
11793 	return func_id == special_kfunc_list[KF_bpf_task_work_schedule_signal] ||
11794 	       func_id == special_kfunc_list[KF_bpf_task_work_schedule_resume];
11795 }
11796 
is_kfunc_ret_null(struct bpf_call_arg_meta * meta)11797 static bool is_kfunc_ret_null(struct bpf_call_arg_meta *meta)
11798 {
11799 	if (is_bpf_refcount_acquire_kfunc(meta->func_id) && meta->arg_owning_ref)
11800 		return false;
11801 
11802 	return meta->kfunc_flags & KF_RET_NULL;
11803 }
11804 
is_kfunc_bpf_rcu_read_lock(struct bpf_call_arg_meta * meta)11805 static bool is_kfunc_bpf_rcu_read_lock(struct bpf_call_arg_meta *meta)
11806 {
11807 	return meta->func_id == special_kfunc_list[KF_bpf_rcu_read_lock];
11808 }
11809 
is_kfunc_bpf_rcu_read_unlock(struct bpf_call_arg_meta * meta)11810 static bool is_kfunc_bpf_rcu_read_unlock(struct bpf_call_arg_meta *meta)
11811 {
11812 	return meta->func_id == special_kfunc_list[KF_bpf_rcu_read_unlock];
11813 }
11814 
is_kfunc_bpf_preempt_disable(struct bpf_call_arg_meta * meta)11815 static bool is_kfunc_bpf_preempt_disable(struct bpf_call_arg_meta *meta)
11816 {
11817 	return meta->func_id == special_kfunc_list[KF_bpf_preempt_disable];
11818 }
11819 
is_kfunc_bpf_preempt_enable(struct bpf_call_arg_meta * meta)11820 static bool is_kfunc_bpf_preempt_enable(struct bpf_call_arg_meta *meta)
11821 {
11822 	return meta->func_id == special_kfunc_list[KF_bpf_preempt_enable];
11823 }
11824 
bpf_is_kfunc_pkt_changing(struct bpf_call_arg_meta * meta)11825 bool bpf_is_kfunc_pkt_changing(struct bpf_call_arg_meta *meta)
11826 {
11827 	return meta->func_id == special_kfunc_list[KF_bpf_xdp_pull_data];
11828 }
11829 
11830 static int
get_kfunc_arg_type(struct bpf_verifier_env * env,struct bpf_call_arg_meta * meta,const struct btf_param * args,int arg,int nargs)11831 get_kfunc_arg_type(struct bpf_verifier_env *env, struct bpf_call_arg_meta *meta,
11832 		   const struct btf_param *args, int arg, int nargs)
11833 {
11834 	const struct btf_type *t, *ref_t = NULL;
11835 	argno_t argno = argno_from_arg(arg + 1);
11836 	const char *ref_tname = NULL;
11837 	int arg_type;
11838 
11839 	t = btf_type_skip_modifiers(meta->btf, args[arg].type, NULL);
11840 
11841 	/* Scalar arguments are classified from their BTF suffix/name alone. */
11842 	if (btf_type_is_scalar(t)) {
11843 		if (is_kfunc_arg_constant(meta->btf, &args[arg]))
11844 			return KF_ARG_CONST;
11845 		if (is_kfunc_arg_const_mem_size(meta->btf, &args[arg]))
11846 			return KF_ARG_CONST_MEM_SIZE;
11847 		if (is_kfunc_arg_mem_size(meta->btf, &args[arg]))
11848 			return KF_ARG_MEM_SIZE;
11849 		if (is_kfunc_arg_scalar_with_name(meta->btf, &args[arg], "rdonly_buf_size") ||
11850 		    is_kfunc_arg_scalar_with_name(meta->btf, &args[arg], "rdwr_buf_size"))
11851 			return KF_ARG_CONST_ALLOC_SIZE_OR_ZERO;
11852 		return KF_ARG_ANYTHING;
11853 	}
11854 
11855 	if (!btf_type_is_ptr(t)) {
11856 		verbose(env, "Unrecognized %s type %s\n",
11857 			reg_arg_name(env, argno), btf_type_str(t));
11858 		return -EINVAL;
11859 	}
11860 	ref_t = btf_type_skip_modifiers(meta->btf, t->type, NULL);
11861 	ref_tname = btf_name_by_offset(meta->btf, ref_t->name_off);
11862 
11863 	/* In this function, we verify the kfunc's BTF as per the argument type,
11864 	 * leaving the rest of the verification with respect to the register
11865 	 * type to our caller. When a set of conditions hold in the BTF type of
11866 	 * arguments, we resolve it to a known kfunc_ptr_arg_type.
11867 	 */
11868 	if (meta->func_id == special_kfunc_list[KF_bpf_cast_to_kern_ctx] ||
11869 	    meta->func_id == special_kfunc_list[KF_bpf_session_is_return] ||
11870 	    meta->func_id == special_kfunc_list[KF_bpf_session_cookie])
11871 		arg_type = KF_ARG_PTR_TO_CTX;
11872 	else if (btf_is_prog_ctx_type(&env->log, meta->btf, t, resolve_prog_type(env->prog), arg))
11873 		arg_type = KF_ARG_PTR_TO_CTX;
11874 	else if (is_kfunc_arg_alloc_obj(meta->btf, &args[arg]))
11875 		arg_type = KF_ARG_PTR_TO_ALLOC_BTF_ID;
11876 	else if (is_kfunc_arg_refcounted_kptr(meta->btf, &args[arg]))
11877 		arg_type = KF_ARG_PTR_TO_REFCOUNTED_KPTR;
11878 	else if (is_kfunc_arg_dynptr(meta->btf, &args[arg]))
11879 		arg_type = KF_ARG_PTR_TO_DYNPTR;
11880 	else if (is_kfunc_arg_iter(meta, arg, &args[arg]))
11881 		arg_type = KF_ARG_PTR_TO_ITER;
11882 	else if (is_kfunc_arg_list_head(meta->btf, &args[arg]))
11883 		arg_type = KF_ARG_PTR_TO_LIST_HEAD;
11884 	else if (is_kfunc_arg_list_node(meta->btf, &args[arg]))
11885 		arg_type = KF_ARG_PTR_TO_LIST_NODE;
11886 	else if (is_kfunc_arg_rbtree_root(meta->btf, &args[arg]))
11887 		arg_type = KF_ARG_PTR_TO_RB_ROOT;
11888 	else if (is_kfunc_arg_rbtree_node(meta->btf, &args[arg]))
11889 		arg_type = KF_ARG_PTR_TO_RB_NODE;
11890 	else if (is_kfunc_arg_const_str(meta->btf, &args[arg]))
11891 		arg_type = KF_ARG_PTR_TO_CONST_STR;
11892 	else if (is_kfunc_arg_const_map(meta->btf, &args[arg]))
11893 		arg_type = KF_ARG_CONST_MAP_PTR;
11894 	else if (is_kfunc_arg_map(meta->btf, &args[arg]))
11895 		arg_type = KF_ARG_PTR_TO_BTF_ID;
11896 	else if (is_kfunc_arg_wq(meta->btf, &args[arg]))
11897 		arg_type = KF_ARG_PTR_TO_WORKQUEUE;
11898 	else if (is_kfunc_arg_timer(meta->btf, &args[arg]))
11899 		arg_type = KF_ARG_PTR_TO_TIMER;
11900 	else if (is_kfunc_arg_task_work(meta->btf, &args[arg]))
11901 		arg_type = KF_ARG_PTR_TO_TASK_WORK;
11902 	else if (is_kfunc_arg_irq_flag(meta->btf, &args[arg]))
11903 		arg_type = KF_ARG_PTR_TO_IRQ_FLAG;
11904 	else if (is_kfunc_arg_res_spin_lock(meta->btf, &args[arg]))
11905 		arg_type = KF_ARG_PTR_TO_RES_SPIN_LOCK;
11906 	else if (is_kfunc_arg_callback(env, meta->btf, &args[arg]))
11907 		arg_type = KF_ARG_PTR_TO_CALLBACK;
11908 	else if (is_kfunc_arg_arena(meta->btf, &args[arg])) {
11909 		if (!bpf_jit_supports_arena_args()) {
11910 			verbose(env, "JIT does not support kfunc %s() with arena pointer arguments\n",
11911 				meta->func_name);
11912 			return -ENOTSUPP;
11913 		}
11914 		if (!env->prog->aux->arena) {
11915 			verbose(env,
11916 				"%s arena pointer requires a program with an associated arena\n",
11917 				reg_arg_name(env, argno));
11918 			return -EINVAL;
11919 		}
11920 		if (reg_from_argno(argno) < 0) {
11921 			verbose(env, "%s arena pointer cannot be a stack argument\n",
11922 				reg_arg_name(env, argno));
11923 			return -EINVAL;
11924 		}
11925 		/*
11926 		 * Both suffixes accept a constant zero. The function model determines
11927 		 * whether the JIT rebases it to the arena base or preserves NULL.
11928 		 * The common nullable path below records that verifier property.
11929 		 */
11930 		arg_type = KF_ARG_PTR_TO_ARENA;
11931 	} else if (arg + 1 < nargs &&
11932 		 (is_kfunc_arg_mem_size(meta->btf, &args[arg + 1]) ||
11933 		  is_kfunc_arg_const_mem_size(meta->btf, &args[arg + 1]))) {
11934 		if (!btf_type_is_void(ref_t) && !btf_type_is_scalar(ref_t) &&
11935 		    !__btf_type_is_scalar_struct(env, meta->btf, ref_t, 0)) {
11936 			verbose(env, "%s pointer type %s %s must point to void, scalar, or struct with scalar\n",
11937 				reg_arg_name(env, argno), btf_type_str(ref_t), ref_tname);
11938 			return -EINVAL;
11939 		}
11940 		arg_type = KF_ARG_PTR_TO_MEM;
11941 	} else if (btf_type_is_struct(ref_t))
11942 		/* A pointer to a struct without a size argument is classified as KF_ARG_PTR_TO_BTF_ID */
11943 		arg_type = KF_ARG_PTR_TO_BTF_ID;
11944 	else {
11945 		/*
11946 		 * Otherwise this is a fixed-size memory buffer supported by
11947 		 * check_helper_mem_access(): a pointer to a scalar or a struct of
11948 		 * scalars. The access size is derived from the pointed-to BTF type.
11949 		 */
11950 		if (!btf_type_is_scalar(ref_t) &&
11951 		    !__btf_type_is_scalar_struct(env, meta->btf, ref_t, 0)) {
11952 			verbose(env, "%s pointer type %s %s must point to scalar, or struct with scalar\n",
11953 				reg_arg_name(env, argno), btf_type_str(ref_t), ref_tname);
11954 			return -EINVAL;
11955 		}
11956 		arg_type = KF_ARG_PTR_TO_MEM | MEM_FIXED_SIZE;
11957 	}
11958 
11959 	if (is_kfunc_arg_nullable(meta->btf, &args[arg]))
11960 		arg_type |= PTR_MAYBE_NULL;
11961 
11962 	return arg_type;
11963 }
11964 
gen_kfunc_arg_proto(struct bpf_verifier_env * env,struct bpf_call_arg_meta * meta,struct bpf_func_proto * proto)11965 static int gen_kfunc_arg_proto(struct bpf_verifier_env *env, struct bpf_call_arg_meta *meta,
11966 			       struct bpf_func_proto *proto)
11967 {
11968 	const struct btf *btf = meta->btf;
11969 	const struct btf_param *args;
11970 	u32 i, nargs;
11971 	int arg_type;
11972 
11973 	args = (const struct btf_param *)(meta->func_proto + 1);
11974 	nargs = btf_type_vlen(meta->func_proto);
11975 	if (nargs > MAX_BPF_FUNC_ARGS) {
11976 		verbose(env, "Function %s has %d > %d args\n", meta->func_name,
11977 			nargs, MAX_BPF_FUNC_ARGS);
11978 		return -EINVAL;
11979 	}
11980 	if (nargs > MAX_BPF_FUNC_REG_ARGS && !bpf_jit_supports_stack_args()) {
11981 		verbose(env, "JIT does not support kfunc %s() with %d args\n",
11982 			meta->func_name, nargs);
11983 		return -ENOTSUPP;
11984 	}
11985 
11986 	for (i = 0; i < nargs; i++) {
11987 		if (is_kfunc_arg_prog_aux(btf, &args[i]) ||
11988 		    is_kfunc_arg_ignore(btf, &args[i]) ||
11989 		    is_kfunc_arg_implicit(meta, i))
11990 			continue;
11991 
11992 		arg_type = get_kfunc_arg_type(env, meta, args, i, nargs);
11993 		if (arg_type < 0)
11994 			return arg_type;
11995 
11996 		proto->arg_type[i] = arg_type;
11997 	}
11998 
11999 	return 0;
12000 }
12001 
process_kf_arg_ptr_to_btf_id(struct bpf_verifier_env * env,struct bpf_reg_state * reg,const struct btf_type * ref_t,const char * ref_tname,u32 ref_id,struct bpf_call_arg_meta * meta,int arg,argno_t argno)12002 static int process_kf_arg_ptr_to_btf_id(struct bpf_verifier_env *env,
12003 					struct bpf_reg_state *reg,
12004 					const struct btf_type *ref_t,
12005 					const char *ref_tname, u32 ref_id,
12006 					struct bpf_call_arg_meta *meta,
12007 					int arg, argno_t argno)
12008 {
12009 	const struct btf_type *reg_ref_t;
12010 	bool strict_type_match = false;
12011 	const struct btf *reg_btf;
12012 	const char *reg_ref_tname;
12013 	bool taking_projection;
12014 	bool struct_same;
12015 	u32 reg_ref_id;
12016 
12017 	if (base_type(reg->type) == PTR_TO_BTF_ID) {
12018 		reg_btf = reg->btf;
12019 		reg_ref_id = reg->btf_id;
12020 	} else {
12021 		reg_btf = btf_vmlinux;
12022 		reg_ref_id = *reg2btf_ids[base_type(reg->type)];
12023 	}
12024 
12025 	/* Enforce strict type matching for calls to kfuncs that are acquiring
12026 	 * or releasing a reference, or are no-cast aliases. We do _not_
12027 	 * enforce strict matching for kfuncs by default,
12028 	 * as we want to enable BPF programs to pass types that are bitwise
12029 	 * equivalent without forcing them to explicitly cast with something
12030 	 * like bpf_cast_to_kern_ctx().
12031 	 *
12032 	 * For example, say we had a type like the following:
12033 	 *
12034 	 * struct bpf_cpumask {
12035 	 *	cpumask_t cpumask;
12036 	 *	refcount_t usage;
12037 	 * };
12038 	 *
12039 	 * Note that as specified in <linux/cpumask.h>, cpumask_t is typedef'ed
12040 	 * to a struct cpumask, so it would be safe to pass a struct
12041 	 * bpf_cpumask * to a kfunc expecting a struct cpumask *.
12042 	 *
12043 	 * The philosophy here is similar to how we allow scalars of different
12044 	 * types to be passed to kfuncs as long as the size is the same. The
12045 	 * only difference here is that we're simply allowing
12046 	 * btf_struct_ids_match() to walk the struct at the 0th offset, and
12047 	 * resolve types.
12048 	 */
12049 	if ((is_kfunc_release(meta) && reg_is_referenced(env, reg)) ||
12050 	    btf_type_ids_nocast_alias(&env->log, reg_btf, reg_ref_id, meta->btf, ref_id))
12051 		strict_type_match = true;
12052 
12053 	WARN_ON_ONCE(is_kfunc_release(meta) && !tnum_is_const(reg->var_off));
12054 
12055 	reg_ref_t = btf_type_skip_modifiers(reg_btf, reg_ref_id, &reg_ref_id);
12056 	reg_ref_tname = btf_name_by_offset(reg_btf, reg_ref_t->name_off);
12057 	struct_same = btf_struct_ids_match(&env->log, reg_btf, reg_ref_id, reg->var_off.value,
12058 					   meta->btf, ref_id, strict_type_match,
12059 					   !type_is_alloc(reg->type));
12060 	/* If kfunc is accepting a projection type (ie. __sk_buff), it cannot
12061 	 * actually use it -- it must cast to the underlying type. So we allow
12062 	 * caller to pass in the underlying type.
12063 	 */
12064 	taking_projection = btf_is_projection_of(ref_tname, reg_ref_tname);
12065 	if (!taking_projection && !struct_same) {
12066 		verbose(env, "kernel function %s %s expected pointer to %s %s but %s has a pointer to %s %s\n",
12067 			meta->func_name, reg_arg_name(env, argno),
12068 			btf_type_str(ref_t), ref_tname, reg_arg_name(env, argno),
12069 			btf_type_str(reg_ref_t), reg_ref_tname);
12070 		return -EINVAL;
12071 	}
12072 	return 0;
12073 }
12074 
process_irq_flag(struct bpf_verifier_env * env,struct bpf_reg_state * reg,argno_t argno,struct bpf_call_arg_meta * meta)12075 static int process_irq_flag(struct bpf_verifier_env *env, struct bpf_reg_state *reg, argno_t argno,
12076 			     struct bpf_call_arg_meta *meta)
12077 {
12078 	int err, spi, kfunc_class = IRQ_NATIVE_KFUNC;
12079 	bool irq_save;
12080 
12081 	if (meta->func_id == special_kfunc_list[KF_bpf_local_irq_save] ||
12082 	    meta->func_id == special_kfunc_list[KF_bpf_res_spin_lock_irqsave]) {
12083 		irq_save = true;
12084 		if (meta->func_id == special_kfunc_list[KF_bpf_res_spin_lock_irqsave])
12085 			kfunc_class = IRQ_LOCK_KFUNC;
12086 	} else if (meta->func_id == special_kfunc_list[KF_bpf_local_irq_restore] ||
12087 		   meta->func_id == special_kfunc_list[KF_bpf_res_spin_unlock_irqrestore]) {
12088 		irq_save = false;
12089 		if (meta->func_id == special_kfunc_list[KF_bpf_res_spin_unlock_irqrestore])
12090 			kfunc_class = IRQ_LOCK_KFUNC;
12091 	} else {
12092 		verifier_bug(env, "unknown irq flags kfunc");
12093 		return -EFAULT;
12094 	}
12095 
12096 	if (irq_save) {
12097 		if (!is_irq_flag_reg_valid_uninit(env, reg)) {
12098 			verbose(env, "expected uninitialized irq flag as %s\n",
12099 				reg_arg_name(env, argno));
12100 			bpf_diag_res(env, env->insn_idx, "IRQ flag is already initialized",
12101 				     "Saving IRQ state requires an uninitialized stack slot for "
12102 				     "the IRQ flag, but this slot already contains tracked IRQ "
12103 				     "flag state.",
12104 				     "Use a fresh stack slot for this save operation, or restore "
12105 				     "the existing IRQ flag before reusing the slot.");
12106 			return -EINVAL;
12107 		}
12108 
12109 		err = check_mem_access(env, env->insn_idx, reg, argno, 0, BPF_DW,
12110 				       BPF_WRITE, -1, false, false);
12111 		if (err)
12112 			return err;
12113 
12114 		err = mark_stack_slot_irq_flag(env, meta, reg, env->insn_idx, kfunc_class);
12115 		if (err)
12116 			return err;
12117 	} else {
12118 		err = is_irq_flag_reg_valid_init(env, reg);
12119 		if (err) {
12120 			verbose(env, "expected an initialized irq flag as %s\n",
12121 				reg_arg_name(env, argno));
12122 			bpf_diag_res(env, env->insn_idx, "uninitialized IRQ flag restore",
12123 				     "Restoring IRQ state requires a stack slot that was "
12124 				     "initialized by a matching IRQ save operation on this path.",
12125 				     "Pass the same stack slot that was previously initialized by "
12126 				     "the matching IRQ save kfunc.");
12127 			return err;
12128 		}
12129 
12130 		spi = irq_flag_get_spi(env, reg);
12131 		if (spi < 0)
12132 			return spi;
12133 
12134 		mark_stack_slots_scratched(env, spi, 1);
12135 
12136 		err = unmark_stack_slot_irq_flag(env, reg, kfunc_class);
12137 		if (err)
12138 			return err;
12139 
12140 		if (!in_rcu_cs(env))
12141 			invalidate_rcu_protected_refs(env);
12142 	}
12143 	return 0;
12144 }
12145 
ref_set_non_owning(struct bpf_verifier_env * env,struct bpf_reg_state * reg)12146 static int ref_set_non_owning(struct bpf_verifier_env *env, struct bpf_reg_state *reg)
12147 {
12148 	struct btf_record *rec = reg_btf_record(reg);
12149 
12150 	if (!env->cur_state->active_locks) {
12151 		verifier_bug(env, "%s w/o active lock", __func__);
12152 		return -EFAULT;
12153 	}
12154 
12155 	if (type_flag(reg->type) & NON_OWN_REF) {
12156 		verifier_bug(env, "NON_OWN_REF already set");
12157 		return -EFAULT;
12158 	}
12159 
12160 	reg->type |= NON_OWN_REF;
12161 	if (rec->refcount_off >= 0)
12162 		reg->type |= MEM_RCU;
12163 
12164 	return 0;
12165 }
12166 
ref_convert_owning_non_owning(struct bpf_verifier_env * env,u32 id)12167 static void ref_convert_owning_non_owning(struct bpf_verifier_env *env, u32 id)
12168 {
12169 	struct bpf_func_state *unused;
12170 	struct bpf_reg_state *reg;
12171 	int err;
12172 
12173 	err = release_reference_nomark(env, id);
12174 	WARN_ON_ONCE(err);
12175 
12176 	bpf_for_each_reg_in_vstate(env->cur_state, unused, reg, ({
12177 		if (reg->id == id) {
12178 			reg->id = 0;
12179 			ref_set_non_owning(env, reg);
12180 		}
12181 	}));
12182 
12183 	return;
12184 }
12185 
12186 /* Implementation details:
12187  *
12188  * Each register points to some region of memory, which we define as an
12189  * allocation. Each allocation may embed a bpf_spin_lock which protects any
12190  * special BPF objects (bpf_list_head, bpf_rb_root, etc.) part of the same
12191  * allocation. The lock and the data it protects are colocated in the same
12192  * memory region.
12193  *
12194  * Hence, everytime a register holds a pointer value pointing to such
12195  * allocation, the verifier preserves a unique reg->id for it.
12196  *
12197  * The verifier remembers the lock 'ptr' and the lock 'id' whenever
12198  * bpf_spin_lock is called.
12199  *
12200  * To enable this, lock state in the verifier captures two values:
12201  *	active_lock.ptr = Register's type specific pointer
12202  *	active_lock.id  = A unique ID for each register pointer value
12203  *
12204  * Currently, PTR_TO_MAP_VALUE and PTR_TO_BTF_ID | MEM_ALLOC are the two
12205  * supported register types.
12206  *
12207  * The active_lock.ptr in case of map values is the reg->map_ptr, and in case of
12208  * allocated objects is the reg->btf pointer.
12209  *
12210  * The active_lock.id is non-unique for maps supporting direct_value_addr, as we
12211  * can establish the provenance of the map value statically for each distinct
12212  * lookup into such maps. They always contain a single map value hence unique
12213  * IDs for each pseudo load pessimizes the algorithm and rejects valid programs.
12214  *
12215  * So, in case of global variables, they use array maps with max_entries = 1,
12216  * hence their active_lock.ptr becomes map_ptr and id = 0 (since they all point
12217  * into the same map value as max_entries is 1, as described above).
12218  *
12219  * In case of inner map lookups, the inner map pointer has same map_ptr as the
12220  * outer map pointer (in verifier context), but each lookup into an inner map
12221  * assigns a fresh reg->id to the lookup, so while lookups into distinct inner
12222  * maps from the same outer map share the same map_ptr as active_lock.ptr, they
12223  * will get different reg->id assigned to each lookup, hence different
12224  * active_lock.id.
12225  *
12226  * In case of allocated objects, active_lock.ptr is the reg->btf, and the
12227  * reg->id is a unique ID preserved after the NULL pointer check on the pointer
12228  * returned from bpf_obj_new. Each allocation receives a new reg->id.
12229  */
check_reg_allocation_locked(struct bpf_verifier_env * env,struct bpf_reg_state * reg)12230 static int check_reg_allocation_locked(struct bpf_verifier_env *env, struct bpf_reg_state *reg)
12231 {
12232 	struct bpf_reference_state *s;
12233 	void *ptr;
12234 	u32 id;
12235 
12236 	switch ((int)reg->type) {
12237 	case PTR_TO_MAP_VALUE:
12238 		ptr = reg->map_ptr;
12239 		break;
12240 	case PTR_TO_BTF_ID | MEM_ALLOC:
12241 		ptr = reg->btf;
12242 		break;
12243 	default:
12244 		verifier_bug(env, "unknown reg type for lock check");
12245 		return -EFAULT;
12246 	}
12247 	id = reg->id;
12248 
12249 	if (!env->cur_state->active_locks)
12250 		return -EINVAL;
12251 	s = find_lock_state(env->cur_state, REF_TYPE_LOCK_MASK, id, ptr);
12252 	if (!s) {
12253 		verbose(env, "held lock and object are not in the same allocation\n");
12254 		return -EINVAL;
12255 	}
12256 	return 0;
12257 }
12258 
is_bpf_list_api_kfunc(u32 btf_id)12259 static bool is_bpf_list_api_kfunc(u32 btf_id)
12260 {
12261 	return is_bpf_list_push_kfunc(btf_id) ||
12262 	       btf_id == special_kfunc_list[KF_bpf_list_pop_front] ||
12263 	       btf_id == special_kfunc_list[KF_bpf_list_pop_back] ||
12264 	       btf_id == special_kfunc_list[KF_bpf_list_del] ||
12265 	       btf_id == special_kfunc_list[KF_bpf_list_front] ||
12266 	       btf_id == special_kfunc_list[KF_bpf_list_back] ||
12267 	       btf_id == special_kfunc_list[KF_bpf_list_is_first] ||
12268 	       btf_id == special_kfunc_list[KF_bpf_list_is_last] ||
12269 	       btf_id == special_kfunc_list[KF_bpf_list_empty];
12270 }
12271 
is_bpf_rbtree_api_kfunc(u32 btf_id)12272 static bool is_bpf_rbtree_api_kfunc(u32 btf_id)
12273 {
12274 	return is_bpf_rbtree_add_kfunc(btf_id) ||
12275 	       btf_id == special_kfunc_list[KF_bpf_rbtree_remove] ||
12276 	       btf_id == special_kfunc_list[KF_bpf_rbtree_first] ||
12277 	       btf_id == special_kfunc_list[KF_bpf_rbtree_root] ||
12278 	       btf_id == special_kfunc_list[KF_bpf_rbtree_left] ||
12279 	       btf_id == special_kfunc_list[KF_bpf_rbtree_right];
12280 }
12281 
is_bpf_res_spin_lock_kfunc(u32 btf_id)12282 static bool is_bpf_res_spin_lock_kfunc(u32 btf_id)
12283 {
12284 	return btf_id == special_kfunc_list[KF_bpf_res_spin_lock] ||
12285 	       btf_id == special_kfunc_list[KF_bpf_res_spin_unlock] ||
12286 	       btf_id == special_kfunc_list[KF_bpf_res_spin_lock_irqsave] ||
12287 	       btf_id == special_kfunc_list[KF_bpf_res_spin_unlock_irqrestore];
12288 }
12289 
kfunc_spin_allowed(struct bpf_verifier_env * env,s32 func_id,s16 offset)12290 static bool kfunc_spin_allowed(struct bpf_verifier_env *env, s32 func_id, s16 offset)
12291 {
12292 	struct bpf_kfunc_meta kfunc;
12293 	int err;
12294 
12295 	err = fetch_kfunc_meta(env, func_id, offset, &kfunc);
12296 	if (err || !kfunc.flags)
12297 		return false;
12298 
12299 	return *kfunc.flags & KF_SPINLOCK_SAFE;
12300 }
12301 
is_sync_callback_calling_kfunc(u32 btf_id)12302 static bool is_sync_callback_calling_kfunc(u32 btf_id)
12303 {
12304 	return is_bpf_rbtree_add_kfunc(btf_id);
12305 }
12306 
is_async_callback_calling_kfunc(u32 btf_id)12307 static bool is_async_callback_calling_kfunc(u32 btf_id)
12308 {
12309 	return is_bpf_wq_set_callback_kfunc(btf_id) ||
12310 	       is_task_work_add_kfunc(btf_id);
12311 }
12312 
bpf_is_throw_kfunc(struct bpf_insn * insn)12313 bool bpf_is_throw_kfunc(struct bpf_insn *insn)
12314 {
12315 	return bpf_pseudo_kfunc_call(insn) && insn->off == 0 &&
12316 	       insn->imm == special_kfunc_list[KF_bpf_throw];
12317 }
12318 
is_bpf_wq_set_callback_kfunc(u32 btf_id)12319 static bool is_bpf_wq_set_callback_kfunc(u32 btf_id)
12320 {
12321 	return btf_id == special_kfunc_list[KF_bpf_wq_set_callback];
12322 }
12323 
is_callback_calling_kfunc(u32 btf_id)12324 static bool is_callback_calling_kfunc(u32 btf_id)
12325 {
12326 	return is_sync_callback_calling_kfunc(btf_id) ||
12327 	       is_async_callback_calling_kfunc(btf_id);
12328 }
12329 
is_rbtree_lock_required_kfunc(u32 btf_id)12330 static bool is_rbtree_lock_required_kfunc(u32 btf_id)
12331 {
12332 	return is_bpf_rbtree_api_kfunc(btf_id);
12333 }
12334 
check_kfunc_is_graph_root_api(struct bpf_verifier_env * env,enum btf_field_type head_field_type,u32 kfunc_btf_id)12335 static bool check_kfunc_is_graph_root_api(struct bpf_verifier_env *env,
12336 					  enum btf_field_type head_field_type,
12337 					  u32 kfunc_btf_id)
12338 {
12339 	bool ret;
12340 
12341 	switch (head_field_type) {
12342 	case BPF_LIST_HEAD:
12343 		ret = is_bpf_list_api_kfunc(kfunc_btf_id);
12344 		break;
12345 	case BPF_RB_ROOT:
12346 		ret = is_bpf_rbtree_api_kfunc(kfunc_btf_id);
12347 		break;
12348 	default:
12349 		verbose(env, "verifier internal error: unexpected graph root argument type %s\n",
12350 			btf_field_type_name(head_field_type));
12351 		return false;
12352 	}
12353 
12354 	if (!ret)
12355 		verbose(env, "verifier internal error: %s head arg for unknown kfunc\n",
12356 			btf_field_type_name(head_field_type));
12357 	return ret;
12358 }
12359 
check_kfunc_is_graph_node_api(struct bpf_verifier_env * env,enum btf_field_type node_field_type,u32 kfunc_btf_id)12360 static bool check_kfunc_is_graph_node_api(struct bpf_verifier_env *env,
12361 					  enum btf_field_type node_field_type,
12362 					  u32 kfunc_btf_id)
12363 {
12364 	bool ret;
12365 
12366 	switch (node_field_type) {
12367 	case BPF_LIST_NODE:
12368 		ret = is_bpf_list_push_kfunc(kfunc_btf_id) ||
12369 		      kfunc_btf_id == special_kfunc_list[KF_bpf_list_del] ||
12370 		      kfunc_btf_id == special_kfunc_list[KF_bpf_list_is_first] ||
12371 		      kfunc_btf_id == special_kfunc_list[KF_bpf_list_is_last];
12372 		break;
12373 	case BPF_RB_NODE:
12374 		ret = (is_bpf_rbtree_add_kfunc(kfunc_btf_id) ||
12375 		       kfunc_btf_id == special_kfunc_list[KF_bpf_rbtree_remove] ||
12376 		       kfunc_btf_id == special_kfunc_list[KF_bpf_rbtree_left] ||
12377 		       kfunc_btf_id == special_kfunc_list[KF_bpf_rbtree_right]);
12378 		break;
12379 	default:
12380 		verbose(env, "verifier internal error: unexpected graph node argument type %s\n",
12381 			btf_field_type_name(node_field_type));
12382 		return false;
12383 	}
12384 
12385 	if (!ret)
12386 		verbose(env, "verifier internal error: %s node arg for unknown kfunc\n",
12387 			btf_field_type_name(node_field_type));
12388 	return ret;
12389 }
12390 
12391 static int
__process_kf_arg_ptr_to_graph_root(struct bpf_verifier_env * env,struct bpf_reg_state * reg,argno_t argno,struct bpf_call_arg_meta * meta,enum btf_field_type head_field_type,struct btf_field ** head_field)12392 __process_kf_arg_ptr_to_graph_root(struct bpf_verifier_env *env,
12393 				   struct bpf_reg_state *reg, argno_t argno,
12394 				   struct bpf_call_arg_meta *meta,
12395 				   enum btf_field_type head_field_type,
12396 				   struct btf_field **head_field)
12397 {
12398 	const char *head_type_name;
12399 	struct btf_field *field;
12400 	struct btf_record *rec;
12401 	u32 head_off;
12402 
12403 	if (meta->btf != btf_vmlinux) {
12404 		verifier_bug(env, "unexpected btf mismatch in kfunc call");
12405 		return -EFAULT;
12406 	}
12407 
12408 	if (!check_kfunc_is_graph_root_api(env, head_field_type, meta->func_id))
12409 		return -EFAULT;
12410 
12411 	head_type_name = btf_field_type_name(head_field_type);
12412 	if (!tnum_is_const(reg->var_off)) {
12413 		verbose(env,
12414 			"%s doesn't have constant offset. %s has to be at the constant offset\n",
12415 			reg_arg_name(env, argno), head_type_name);
12416 		return -EINVAL;
12417 	}
12418 
12419 	rec = reg_btf_record(reg);
12420 	head_off = reg->var_off.value;
12421 	field = btf_record_find(rec, head_off, head_field_type);
12422 	if (!field) {
12423 		verbose(env, "%s not found at offset=%u\n", head_type_name, head_off);
12424 		return -EINVAL;
12425 	}
12426 
12427 	/* All functions require bpf_list_head to be protected using a bpf_spin_lock */
12428 	if (check_reg_allocation_locked(env, reg)) {
12429 		verbose(env, "bpf_spin_lock at off=%d must be held for %s\n",
12430 			rec->spin_lock_off, head_type_name);
12431 		return -EINVAL;
12432 	}
12433 
12434 	if (*head_field) {
12435 		verifier_bug(env, "repeating %s arg", head_type_name);
12436 		return -EFAULT;
12437 	}
12438 	*head_field = field;
12439 	return 0;
12440 }
12441 
process_kf_arg_ptr_to_list_head(struct bpf_verifier_env * env,struct bpf_reg_state * reg,argno_t argno,struct bpf_call_arg_meta * meta)12442 static int process_kf_arg_ptr_to_list_head(struct bpf_verifier_env *env,
12443 					   struct bpf_reg_state *reg, argno_t argno,
12444 					   struct bpf_call_arg_meta *meta)
12445 {
12446 	return __process_kf_arg_ptr_to_graph_root(env, reg, argno, meta, BPF_LIST_HEAD,
12447 							  &meta->arg_list_head.field);
12448 }
12449 
process_kf_arg_ptr_to_rbtree_root(struct bpf_verifier_env * env,struct bpf_reg_state * reg,argno_t argno,struct bpf_call_arg_meta * meta)12450 static int process_kf_arg_ptr_to_rbtree_root(struct bpf_verifier_env *env,
12451 					     struct bpf_reg_state *reg, argno_t argno,
12452 					     struct bpf_call_arg_meta *meta)
12453 {
12454 	return __process_kf_arg_ptr_to_graph_root(env, reg, argno, meta, BPF_RB_ROOT,
12455 							  &meta->arg_rbtree_root.field);
12456 }
12457 
12458 static int
__process_kf_arg_ptr_to_graph_node(struct bpf_verifier_env * env,struct bpf_reg_state * reg,argno_t argno,struct bpf_call_arg_meta * meta,enum btf_field_type head_field_type,enum btf_field_type node_field_type,struct btf_field ** node_field)12459 __process_kf_arg_ptr_to_graph_node(struct bpf_verifier_env *env,
12460 				   struct bpf_reg_state *reg, argno_t argno,
12461 				   struct bpf_call_arg_meta *meta,
12462 				   enum btf_field_type head_field_type,
12463 				   enum btf_field_type node_field_type,
12464 				   struct btf_field **node_field)
12465 {
12466 	const char *node_type_name;
12467 	const struct btf_type *et, *t;
12468 	struct btf_field *field;
12469 	u32 node_off;
12470 
12471 	if (meta->btf != btf_vmlinux) {
12472 		verifier_bug(env, "unexpected btf mismatch in kfunc call");
12473 		return -EFAULT;
12474 	}
12475 
12476 	if (!check_kfunc_is_graph_node_api(env, node_field_type, meta->func_id))
12477 		return -EFAULT;
12478 
12479 	node_type_name = btf_field_type_name(node_field_type);
12480 	if (!tnum_is_const(reg->var_off)) {
12481 		verbose(env,
12482 			"%s doesn't have constant offset. %s has to be at the constant offset\n",
12483 			reg_arg_name(env, argno), node_type_name);
12484 		return -EINVAL;
12485 	}
12486 
12487 	node_off = reg->var_off.value;
12488 	field = reg_find_field_offset(reg, node_off, node_field_type);
12489 	if (!field) {
12490 		verbose(env, "%s not found at offset=%u\n", node_type_name, node_off);
12491 		return -EINVAL;
12492 	}
12493 
12494 	field = *node_field;
12495 
12496 	et = btf_type_by_id(field->graph_root.btf, field->graph_root.value_btf_id);
12497 	t = btf_type_by_id(reg->btf, reg->btf_id);
12498 	if (!btf_struct_ids_match(&env->log, reg->btf, reg->btf_id, 0, field->graph_root.btf,
12499 				  field->graph_root.value_btf_id, true,
12500 				  !type_is_alloc(reg->type))) {
12501 		verbose(env, "operation on %s expects arg#1 %s at offset=%d "
12502 			"in struct %s, but arg is at offset=%d in struct %s\n",
12503 			btf_field_type_name(head_field_type),
12504 			btf_field_type_name(node_field_type),
12505 			field->graph_root.node_offset,
12506 			btf_name_by_offset(field->graph_root.btf, et->name_off),
12507 			node_off, btf_name_by_offset(reg->btf, t->name_off));
12508 		return -EINVAL;
12509 	}
12510 	meta->arg_btf = reg->btf;
12511 	meta->arg_btf_id = reg->btf_id;
12512 
12513 	if (node_off != field->graph_root.node_offset) {
12514 		verbose(env, "arg#1 offset=%d, but expected %s at offset=%d in struct %s\n",
12515 			node_off, btf_field_type_name(node_field_type),
12516 			field->graph_root.node_offset,
12517 			btf_name_by_offset(field->graph_root.btf, et->name_off));
12518 		return -EINVAL;
12519 	}
12520 
12521 	return 0;
12522 }
12523 
process_kf_arg_ptr_to_list_node(struct bpf_verifier_env * env,struct bpf_reg_state * reg,argno_t argno,struct bpf_call_arg_meta * meta)12524 static int process_kf_arg_ptr_to_list_node(struct bpf_verifier_env *env,
12525 					   struct bpf_reg_state *reg, argno_t argno,
12526 					   struct bpf_call_arg_meta *meta)
12527 {
12528 	return __process_kf_arg_ptr_to_graph_node(env, reg, argno, meta,
12529 						  BPF_LIST_HEAD, BPF_LIST_NODE,
12530 						  &meta->arg_list_head.field);
12531 }
12532 
process_kf_arg_ptr_to_rbtree_node(struct bpf_verifier_env * env,struct bpf_reg_state * reg,argno_t argno,struct bpf_call_arg_meta * meta)12533 static int process_kf_arg_ptr_to_rbtree_node(struct bpf_verifier_env *env,
12534 					     struct bpf_reg_state *reg, argno_t argno,
12535 					     struct bpf_call_arg_meta *meta)
12536 {
12537 	return __process_kf_arg_ptr_to_graph_node(env, reg, argno, meta,
12538 						  BPF_RB_ROOT, BPF_RB_NODE,
12539 						  &meta->arg_rbtree_root.field);
12540 }
12541 
12542 /*
12543  * css_task iter allowlist is needed to avoid dead locking on css_set_lock.
12544  * LSM hooks and iters (both sleepable and non-sleepable) are safe.
12545  * Any sleepable progs are also safe since bpf_check_attach_target() enforce
12546  * them can only be attached to some specific hook points.
12547  */
check_css_task_iter_allowlist(struct bpf_verifier_env * env)12548 static bool check_css_task_iter_allowlist(struct bpf_verifier_env *env)
12549 {
12550 	enum bpf_prog_type prog_type = resolve_prog_type(env->prog);
12551 
12552 	switch (prog_type) {
12553 	case BPF_PROG_TYPE_LSM:
12554 		return true;
12555 	case BPF_PROG_TYPE_TRACING:
12556 		if (env->prog->expected_attach_type == BPF_TRACE_ITER)
12557 			return true;
12558 		fallthrough;
12559 	default:
12560 		return in_sleepable(env);
12561 	}
12562 }
12563 
check_kfunc_args(struct bpf_verifier_env * env,struct bpf_call_arg_meta * meta,int insn_idx)12564 static int check_kfunc_args(struct bpf_verifier_env *env, struct bpf_call_arg_meta *meta,
12565 			    int insn_idx)
12566 {
12567 	const char *func_name = meta->func_name, *ref_tname;
12568 	struct bpf_func_state *caller = cur_func(env);
12569 	struct bpf_reg_state *regs = cur_regs(env);
12570 	const struct btf *btf = meta->btf;
12571 	const struct btf_param *args;
12572 	struct btf_record *rec;
12573 	u32 i, nargs;
12574 	int ret;
12575 
12576 	args = (const struct btf_param *)(meta->func_proto + 1);
12577 	nargs = btf_type_vlen(meta->func_proto);
12578 
12579 	ret = check_outgoing_stack_args(env, caller, nargs, func_name, btf, args);
12580 	if (ret)
12581 		return ret;
12582 
12583 	/* Check that BTF function arguments match actual types that the
12584 	 * verifier sees.
12585 	 */
12586 	for (i = 0; i < nargs; i++) {
12587 		struct bpf_reg_state *reg = get_func_arg_reg(caller, regs, i);
12588 		const struct btf_type *t, *ref_t, *resolve_ret;
12589 		enum bpf_arg_type arg_type = ARG_DONTCARE;
12590 		argno_t argno = argno_from_arg(i + 1);
12591 		int regno = reg_from_argno(argno);
12592 		bool btf_id_fixed_off_ok = true;
12593 		u32 ref_id = args[i].type, type_size;
12594 		int kf_arg_type = meta->fn->arg_type[i];
12595 
12596 		if (is_kfunc_arg_prog_aux(btf, &args[i])) {
12597 			/* Reject repeated use bpf_prog_aux */
12598 			if (meta->arg_prog) {
12599 				verifier_bug(env, "Only 1 prog->aux argument supported per-kfunc");
12600 				return -EFAULT;
12601 			}
12602 			if (regno < 0) {
12603 				verbose(env, "%s prog->aux cannot be a stack argument\n",
12604 					reg_arg_name(env, argno));
12605 				return -EINVAL;
12606 			}
12607 			meta->arg_prog = true;
12608 			cur_aux(env)->arg_prog = regno;
12609 			continue;
12610 		}
12611 
12612 		if (is_kfunc_arg_ignore(btf, &args[i]) || is_kfunc_arg_implicit(meta, i))
12613 			continue;
12614 
12615 		t = btf_type_skip_modifiers(btf, args[i].type, NULL);
12616 
12617 		if (btf_type_is_ptr(t)) {
12618 			ref_t = btf_type_skip_modifiers(btf, t->type, &ref_id);
12619 			ref_tname = btf_name_by_offset(btf, ref_t->name_off);
12620 		}
12621 
12622 		if (btf_type_is_ptr(t) &&
12623 		    (bpf_register_is_null(reg) || type_may_be_null(reg->type)) &&
12624 		    !type_may_be_null(kf_arg_type)) {
12625 			const char *expected_type;
12626 
12627 			expected_type = bpf_diag_fmt_btf_type(env, btf, args[i].type);
12628 			verbose(env, "Possibly NULL pointer passed to trusted %s\n",
12629 				reg_arg_name(env, argno));
12630 			bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name,
12631 					      "Add a NULL check and call the kfunc only on the non-NULL path.",
12632 					      "the pointer may be NULL, but this kfunc requires a non-NULL value of type %s",
12633 					      expected_type);
12634 			return -EACCES;
12635 		}
12636 
12637 		if (regno == meta->release_regno && !is_kfunc_arg_dynptr(meta->btf, &args[i]) &&
12638 		    !reg_is_referenced(env, reg) && !bpf_register_is_null(reg)) {
12639 			const char *expected_type;
12640 
12641 			expected_type = bpf_diag_fmt_btf_type(env, btf, ref_id);
12642 			verbose(env, "release kfunc %s expects referenced PTR_TO_BTF_ID passed to %s\n",
12643 				func_name, reg_arg_name(env, argno));
12644 			bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name,
12645 					      "Pass the resource-owning pointer returned by the matching acquire kfunc, and avoid calling the release kfunc after ownership has already been transferred or released.",
12646 					      "release kfuncs require a resource-owning value of type %s returned by a matching acquire kfunc",
12647 					      expected_type);
12648 			return -EINVAL;
12649 		}
12650 
12651 		if (reg_is_referenced(env, reg))
12652 			update_ref_obj(&meta->ref_obj, reg);
12653 
12654 		if (bpf_register_is_null(reg) && type_may_be_null(kf_arg_type))
12655 			continue;
12656 
12657 		if (is_kfunc_arg_map(btf, &args[i])) {
12658 			ref_id = *reg2btf_ids[CONST_PTR_TO_MAP];
12659 			ref_t = btf_type_by_id(btf_vmlinux, ref_id);
12660 			ref_tname = btf_name_by_offset(btf, ref_t->name_off);
12661 		}
12662 
12663 		switch (base_type(kf_arg_type)) {
12664 		case KF_ARG_CONST:
12665 		case KF_ARG_CONST_MEM_SIZE:
12666 		case KF_ARG_MEM_SIZE:
12667 		case KF_ARG_ANYTHING:
12668 		case KF_ARG_CONST_ALLOC_SIZE_OR_ZERO:
12669 		case KF_ARG_PTR_TO_ALLOC_BTF_ID:
12670 		case KF_ARG_PTR_TO_BTF_ID:
12671 		case KF_ARG_CONST_MAP_PTR:
12672 		case KF_ARG_PTR_TO_ITER:
12673 		case KF_ARG_PTR_TO_LIST_HEAD:
12674 		case KF_ARG_PTR_TO_LIST_NODE:
12675 		case KF_ARG_PTR_TO_RB_ROOT:
12676 		case KF_ARG_PTR_TO_RB_NODE:
12677 		case KF_ARG_PTR_TO_MEM:
12678 		case KF_ARG_PTR_TO_CALLBACK:
12679 		case KF_ARG_PTR_TO_CONST_STR:
12680 		case KF_ARG_PTR_TO_WORKQUEUE:
12681 		case KF_ARG_PTR_TO_TIMER:
12682 		case KF_ARG_PTR_TO_TASK_WORK:
12683 		case KF_ARG_PTR_TO_IRQ_FLAG:
12684 		case KF_ARG_PTR_TO_RES_SPIN_LOCK:
12685 		case KF_ARG_PTR_TO_ARENA:
12686 			break;
12687 		case KF_ARG_PTR_TO_DYNPTR:
12688 			arg_type = ARG_PTR_TO_DYNPTR;
12689 			break;
12690 		case KF_ARG_PTR_TO_CTX:
12691 			arg_type = ARG_PTR_TO_CTX;
12692 			break;
12693 		case KF_ARG_PTR_TO_REFCOUNTED_KPTR:
12694 			arg_type = ARG_PTR_TO_BTF_ID;
12695 			btf_id_fixed_off_ok = false;
12696 			break;
12697 		default:
12698 			verifier_bug(env, "unknown kfunc arg type %d", kf_arg_type);
12699 			return -EFAULT;
12700 		}
12701 
12702 		if (regno == meta->release_regno)
12703 			arg_type |= OBJ_RELEASE;
12704 		ret = __check_func_arg_reg_off(env, reg, argno, arg_type,
12705 					       btf_id_fixed_off_ok);
12706 		if (ret < 0)
12707 			return ret;
12708 
12709 		switch (base_type(kf_arg_type)) {
12710 		case KF_ARG_CONST:
12711 			if (reg->type != SCALAR_VALUE) {
12712 				verbose(env, "%s is not a scalar\n", reg_arg_name(env, argno));
12713 				bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name,
12714 						      "Pass an integer scalar value for this argument, not a pointer or resource object.",
12715 						      "the kfunc expects an integer scalar, but %s is %s",
12716 						      reg_arg_name(env, argno),
12717 						      bpf_diag_reg_type_plain(env, reg->type));
12718 				return -EINVAL;
12719 			}
12720 
12721 			ret = process_const_arg(env, reg, argno, meta);
12722 			if (ret < 0) {
12723 				if (ret == -EINVAL)
12724 					bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name,
12725 							      "Pass a compile-time constant or a value the verifier can prove is constant at this call.",
12726 							      "the kfunc requires this scalar argument to be a verifier-known constant, but %s is variable on this path",
12727 							      reg_arg_name(env, argno));
12728 				return ret;
12729 			}
12730 			break;
12731 		case KF_ARG_ANYTHING:
12732 			if (reg->type != SCALAR_VALUE) {
12733 				verbose(env, "%s is not a scalar\n", reg_arg_name(env, argno));
12734 				bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name,
12735 						      "Pass an integer scalar value for this argument, not a pointer or resource object.",
12736 						      "the kfunc expects an integer scalar, but %s is %s",
12737 						      reg_arg_name(env, argno),
12738 						      bpf_diag_reg_type_plain(env, reg->type));
12739 				return -EINVAL;
12740 			}
12741 			break;
12742 		case KF_ARG_CONST_ALLOC_SIZE_OR_ZERO:
12743 			if (reg->type != SCALAR_VALUE) {
12744 				verbose(env, "%s is not a scalar\n", reg_arg_name(env, argno));
12745 				bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name,
12746 						      "Pass an integer scalar value for this argument, not a pointer or resource object.",
12747 						      "the kfunc expects an integer scalar, but %s is %s",
12748 						      reg_arg_name(env, argno),
12749 						      bpf_diag_reg_type_plain(env, reg->type));
12750 				return -EINVAL;
12751 			}
12752 
12753 			if (is_kfunc_arg_scalar_with_name(btf, &args[i], "rdonly_buf_size"))
12754 				meta->r0_rdonly = true;
12755 			ret = process_const_alloc_mem_size(env, reg, argno, &meta->ret_mem);
12756 			if (ret < 0) {
12757 				if (ret == -EINVAL)
12758 					bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name,
12759 							      "Pass a verifier-known constant size for this kfunc buffer argument.",
12760 							      "the kfunc uses this argument as a return-buffer size, but %s is invalid or variable on this path",
12761 							      reg_arg_name(env, argno));
12762 				return ret;
12763 			}
12764 			break;
12765 		case KF_ARG_PTR_TO_CTX:
12766 			if (reg->type != PTR_TO_CTX) {
12767 				verbose(env, "%s expected pointer to ctx, but got %s\n",
12768 					reg_arg_name(env, argno), reg_type_str(env, reg->type));
12769 				bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name,
12770 						      "Pass the original program context pointer or preserve it before modifying registers.",
12771 						      "the kfunc expects a context pointer, but %s is %s",
12772 						      reg_arg_name(env, argno),
12773 						      bpf_diag_reg_type_plain(env, reg->type));
12774 				return -EINVAL;
12775 			}
12776 
12777 			if (meta->func_id == special_kfunc_list[KF_bpf_cast_to_kern_ctx]) {
12778 				ret = get_kern_ctx_btf_id(&env->log, resolve_prog_type(env->prog));
12779 				if (ret < 0)
12780 					return -EINVAL;
12781 				meta->ret_btf_id  = ret;
12782 			}
12783 			break;
12784 		case KF_ARG_PTR_TO_ARENA:
12785 			if (reg->type != PTR_TO_ARENA && reg->type != SCALAR_VALUE) {
12786 				verbose(env, "%s is not a pointer to arena or scalar\n",
12787 					reg_arg_name(env, argno));
12788 				return -EINVAL;
12789 			}
12790 			break;
12791 		case KF_ARG_PTR_TO_ALLOC_BTF_ID:
12792 			if (reg->type == (PTR_TO_BTF_ID | MEM_ALLOC)) {
12793 				if (!is_bpf_obj_drop_kfunc(meta->func_id)) {
12794 					verbose(env, "%s expected for bpf_obj_drop()\n",
12795 						reg_arg_name(env, argno));
12796 					return -EINVAL;
12797 				}
12798 			} else if (reg->type == (PTR_TO_BTF_ID | MEM_ALLOC | MEM_PERCPU)) {
12799 				if (!is_bpf_percpu_obj_drop_kfunc(meta->func_id)) {
12800 					verbose(env, "%s expected for bpf_percpu_obj_drop()\n",
12801 						reg_arg_name(env, argno));
12802 					return -EINVAL;
12803 				}
12804 			} else {
12805 				verbose(env, "%s expected pointer to allocated object\n",
12806 					reg_arg_name(env, argno));
12807 				bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name,
12808 						      "Pass a pointer returned by the matching BPF object allocation path.",
12809 						      "the kfunc expects an allocated object pointer, but %s is %s",
12810 						      reg_arg_name(env, argno),
12811 						      bpf_diag_reg_type_plain(env, reg->type));
12812 				return -EINVAL;
12813 			}
12814 			if (!reg_is_referenced(env, reg)) {
12815 				verbose(env, "allocated object must be referenced\n");
12816 				bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name,
12817 						      "Pass the owned object pointer before it is released or transferred.",
12818 						      "the allocated object pointer in %s must still carry verifier-tracked ownership, but this pointer no longer owns a live resource",
12819 						      reg_arg_name(env, argno));
12820 				return -EINVAL;
12821 			}
12822 			if (meta->btf == btf_vmlinux) {
12823 				meta->arg_btf = reg->btf;
12824 				meta->arg_btf_id = reg->btf_id;
12825 			}
12826 			break;
12827 		case KF_ARG_PTR_TO_DYNPTR:
12828 		{
12829 			enum bpf_arg_type dynptr_arg_type = ARG_PTR_TO_DYNPTR;
12830 
12831 			if (is_kfunc_arg_uninit(btf, &args[i]))
12832 				dynptr_arg_type |= MEM_UNINIT;
12833 
12834 			if (meta->func_id == special_kfunc_list[KF_bpf_dynptr_from_skb]) {
12835 				dynptr_arg_type |= DYNPTR_TYPE_SKB;
12836 			} else if (meta->func_id == special_kfunc_list[KF_bpf_dynptr_from_xdp]) {
12837 				dynptr_arg_type |= DYNPTR_TYPE_XDP;
12838 			} else if (meta->func_id == special_kfunc_list[KF_bpf_dynptr_from_skb_meta]) {
12839 				dynptr_arg_type |= DYNPTR_TYPE_SKB_META;
12840 			} else if (meta->func_id == special_kfunc_list[KF_bpf_dynptr_from_file]) {
12841 				dynptr_arg_type |= DYNPTR_TYPE_FILE;
12842 			} else if (meta->func_id == special_kfunc_list[KF_bpf_dynptr_file_discard]) {
12843 				dynptr_arg_type |= DYNPTR_TYPE_FILE | OBJ_RELEASE;
12844 			} else if (meta->func_id == special_kfunc_list[KF_bpf_dynptr_clone] &&
12845 				   (dynptr_arg_type & MEM_UNINIT)) {
12846 				enum bpf_dynptr_type parent_type = meta->dynptr.type;
12847 
12848 				if (parent_type == BPF_DYNPTR_TYPE_INVALID) {
12849 					verifier_bug(env, "no dynptr type for parent of clone");
12850 					return -EFAULT;
12851 				}
12852 
12853 				dynptr_arg_type |= (unsigned int)get_dynptr_type_flag(parent_type);
12854 			}
12855 
12856 			ret = process_dynptr_func(env, reg, argno, insn_idx, func_name,
12857 						  dynptr_arg_type, &meta->ref_obj, &meta->dynptr);
12858 			if (ret < 0)
12859 				return ret;
12860 			break;
12861 		}
12862 		case KF_ARG_PTR_TO_ITER:
12863 			if (meta->func_id == special_kfunc_list[KF_bpf_iter_css_task_new]) {
12864 				if (!check_css_task_iter_allowlist(env)) {
12865 					verbose(env, "css_task_iter is only allowed in bpf_lsm, bpf_iter and sleepable progs\n");
12866 					return -EINVAL;
12867 				}
12868 			}
12869 			ret = process_iter_arg(env, reg, argno, insn_idx, meta);
12870 			if (ret < 0)
12871 				return ret;
12872 			break;
12873 		case KF_ARG_PTR_TO_LIST_HEAD:
12874 			if (reg->type != PTR_TO_MAP_VALUE &&
12875 			    reg->type != (PTR_TO_BTF_ID | MEM_ALLOC)) {
12876 				verbose(env, "%s expected pointer to map value or allocated object\n",
12877 					reg_arg_name(env, argno));
12878 				return -EINVAL;
12879 			}
12880 			if (reg->type == (PTR_TO_BTF_ID | MEM_ALLOC) &&
12881 			    !reg_is_referenced(env, reg)) {
12882 				verbose(env, "allocated object must be referenced\n");
12883 				return -EINVAL;
12884 			}
12885 			ret = process_kf_arg_ptr_to_list_head(env, reg, argno, meta);
12886 			if (ret < 0)
12887 				return ret;
12888 			break;
12889 		case KF_ARG_PTR_TO_RB_ROOT:
12890 			if (reg->type != PTR_TO_MAP_VALUE &&
12891 			    reg->type != (PTR_TO_BTF_ID | MEM_ALLOC)) {
12892 				verbose(env, "%s expected pointer to map value or allocated object\n",
12893 					reg_arg_name(env, argno));
12894 				return -EINVAL;
12895 			}
12896 			if (reg->type == (PTR_TO_BTF_ID | MEM_ALLOC) &&
12897 			    !reg_is_referenced(env, reg)) {
12898 				verbose(env, "allocated object must be referenced\n");
12899 				return -EINVAL;
12900 			}
12901 			ret = process_kf_arg_ptr_to_rbtree_root(env, reg, argno, meta);
12902 			if (ret < 0)
12903 				return ret;
12904 			break;
12905 		case KF_ARG_PTR_TO_LIST_NODE:
12906 			if (is_kfunc_arg_nonown_allowed(btf, &args[i]) &&
12907 			    type_is_non_owning_ref(reg->type) && !reg_is_referenced(env, reg)) {
12908 				/* Allow bpf_list_front/back return value for
12909 				 * __nonown_allowed list-node arguments.
12910 				 */
12911 				goto check_ok;
12912 			}
12913 			if (reg->type != (PTR_TO_BTF_ID | MEM_ALLOC)) {
12914 				verbose(env, "%s expected pointer to allocated object\n",
12915 					reg_arg_name(env, argno));
12916 				return -EINVAL;
12917 			}
12918 			if (!reg_is_referenced(env, reg)) {
12919 				verbose(env, "allocated object must be referenced\n");
12920 				return -EINVAL;
12921 			}
12922 check_ok:
12923 			ret = process_kf_arg_ptr_to_list_node(env, reg, argno, meta);
12924 			if (ret < 0)
12925 				return ret;
12926 			break;
12927 		case KF_ARG_PTR_TO_RB_NODE:
12928 			if (is_bpf_rbtree_add_kfunc(meta->func_id)) {
12929 				if (reg->type != (PTR_TO_BTF_ID | MEM_ALLOC)) {
12930 					verbose(env, "%s expected pointer to allocated object\n",
12931 						reg_arg_name(env, argno));
12932 					return -EINVAL;
12933 				}
12934 				if (!reg_is_referenced(env, reg)) {
12935 					verbose(env, "allocated object must be referenced\n");
12936 					return -EINVAL;
12937 				}
12938 			} else {
12939 				if (!type_is_non_owning_ref(reg->type) &&
12940 				    !reg_is_referenced(env, reg)) {
12941 					verbose(env, "%s can only take non-owning or refcounted bpf_rb_node pointer\n", func_name);
12942 					return -EINVAL;
12943 				}
12944 				if (in_rbtree_lock_required_cb(env)) {
12945 					verbose(env, "%s not allowed in rbtree cb\n", func_name);
12946 					return -EINVAL;
12947 				}
12948 			}
12949 
12950 			ret = process_kf_arg_ptr_to_rbtree_node(env, reg, argno, meta);
12951 			if (ret < 0)
12952 				return ret;
12953 			break;
12954 		case KF_ARG_CONST_MAP_PTR:
12955 			if (base_type(reg->type) != CONST_PTR_TO_MAP ||
12956 			    type_may_be_null(reg->type)) {
12957 				verbose(env, "pointer in %s isn't map pointer\n",
12958 					reg_arg_name(env, argno));
12959 				return -EINVAL;
12960 			}
12961 			ret = process_map_ptr_arg(env, reg, argno, meta);
12962 			if (ret < 0)
12963 				return ret;
12964 			break;
12965 		case KF_ARG_PTR_TO_BTF_ID:
12966 			/* Only base_type is checked, further checks are done here */
12967 			if (base_type(reg->type) == PTR_TO_BTF_ID ||
12968 			    reg2btf_ids[base_type(reg->type)]) {
12969 				if (!is_trusted_reg(env, reg) ||
12970 				    bpf_type_has_unsafe_modifiers(reg->type)) {
12971 					if (!is_kfunc_rcu(meta)) {
12972 						const char *expected_type;
12973 
12974 						expected_type = bpf_diag_fmt_btf_type(env, btf, ref_id);
12975 						verbose(env, "%s must be referenced or trusted\n",
12976 							reg_arg_name(env, argno));
12977 						bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name,
12978 								      "Pass a pointer acquired from a verifier-tracked source, or call this kfunc only inside the required protection if it accepts RCU pointers.",
12979 								      "the kfunc requires a trusted or resource-owning pointer to %s, but %s is %s",
12980 								      expected_type,
12981 								      reg_arg_name(env, argno),
12982 								      bpf_diag_reg_type_plain(env, reg->type));
12983 						return -EINVAL;
12984 					}
12985 					if (!is_rcu_reg(reg)) {
12986 						const char *expected_type;
12987 
12988 						expected_type = bpf_diag_fmt_btf_type(env, btf, ref_id);
12989 						verbose(env, "%s must be a rcu pointer\n",
12990 							reg_arg_name(env, argno));
12991 						bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name,
12992 								      "Use this kfunc with a pointer that is valid in an RCU read lock region.",
12993 								      "the kfunc requires an RCU-protected pointer to %s, but %s is %s",
12994 								      expected_type,
12995 								      reg_arg_name(env, argno),
12996 								      bpf_diag_reg_type_plain(env, reg->type));
12997 						return -EINVAL;
12998 					}
12999 				}
13000 
13001 				ret = process_kf_arg_ptr_to_btf_id(env, reg, ref_t, ref_tname, ref_id, meta, i, argno);
13002 				if (ret < 0)
13003 					return ret;
13004 				break;
13005 			}
13006 
13007 			if (!__btf_type_is_scalar_struct(env, meta->btf, ref_t, 0)) {
13008 				enum bpf_reg_type reg2btf_type = lookup_reg2btf_ids(ref_id);
13009 				const char *expected_type;
13010 
13011 				verbose(env, "%s is %s expected %s %s",
13012 					reg_arg_name(env, argno), reg_type_str(env, reg->type),
13013 					btf_type_str(ref_t), ref_tname);
13014 				if (reg2btf_type != NOT_INIT)
13015 					verbose(env, " or %s", reg_type_str(env, reg2btf_type));
13016 				verbose(env, "\n");
13017 				expected_type = bpf_diag_fmt_btf_type(env, btf, ref_id);
13018 				bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name,
13019 						      "Pass a verifier-tracked pointer to the expected kernel object type, not a pointer to stack storage or another memory buffer.",
13020 						      "the kfunc expects a pointer to %s, but this argument is %s and cannot be used as that kernel object pointer",
13021 						      expected_type,
13022 						      bpf_diag_reg_type_plain(env, reg->type));
13023 				return -EINVAL;
13024 			}
13025 
13026 			/*
13027 			 * If the register does not contain btf id but the argument type is a pointer to
13028 			 * scalar-only struct, allow verifying it as a fixed size memory.
13029 			 */
13030 			kf_arg_type = KF_ARG_PTR_TO_MEM | MEM_FIXED_SIZE;
13031 			fallthrough;
13032 		case KF_ARG_PTR_TO_MEM:
13033 			if (kf_arg_type & MEM_FIXED_SIZE) {
13034 				bool known_memory;
13035 
13036 				resolve_ret = btf_resolve_size(btf, ref_t, &type_size);
13037 				if (IS_ERR(resolve_ret)) {
13038 					verbose(env, "%s reference type('%s %s') size cannot be determined: %ld\n",
13039 						reg_arg_name(env, argno), btf_type_str(ref_t),
13040 						ref_tname, PTR_ERR(resolve_ret));
13041 					return -EINVAL;
13042 				}
13043 				ret = check_mem_reg(env, reg, argno, type_size, BPF_READ | BPF_WRITE,
13044 						    meta, &known_memory);
13045 				if (ret < 0) {
13046 					const char *expected_type;
13047 
13048 					expected_type = bpf_diag_fmt_btf_type(env, btf, ref_id);
13049 					if (known_memory)
13050 						bpf_diag_call_arg_fmt(
13051 							env, insn_idx, argno, func_name,
13052 							"Pass memory with at least the required number of accessible bytes and suitable read and write access.",
13053 							"the kfunc expects %u bytes of memory for %s, but the verifier cannot prove that %s provides a readable and writable range of that size",
13054 							type_size, expected_type,
13055 							bpf_diag_reg_type_plain(env, reg->type));
13056 					else
13057 						bpf_diag_call_arg_fmt(
13058 							env, insn_idx, argno, func_name,
13059 							"Pass stack, map, context, or other verifier-known memory of the expected type and size, not an integer cast to a pointer.",
13060 							"the kfunc expects %u bytes of memory for %s, but it is %s and not verifier-known memory",
13061 							type_size, expected_type,
13062 							bpf_diag_reg_type_plain(env, reg->type));
13063 					return ret;
13064 				}
13065 			}
13066 			break;
13067 		case KF_ARG_CONST_MEM_SIZE:
13068 			ret = process_const_arg(env, reg, argno, meta);
13069 			if (ret < 0) {
13070 				if (ret == -EINVAL)
13071 					bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name,
13072 							      "Pass a compile-time constant or a value the verifier can prove is constant at this call.",
13073 							      "the kfunc requires this memory size to be a verifier-known constant, but %s is variable on this path",
13074 							      reg_arg_name(env, argno));
13075 				return ret;
13076 			}
13077 			fallthrough;
13078 		case KF_ARG_MEM_SIZE:
13079 		{
13080 			struct bpf_reg_state *buff_reg = get_func_arg_reg(caller, regs, i - 1);
13081 			struct bpf_reg_state *size_reg = reg;
13082 			argno_t buff_argno = argno_from_arg(i);
13083 			enum bpf_mem_size_failure failure;
13084 
13085 			if (reg->type != SCALAR_VALUE) {
13086 				verbose(env, "%s is not a scalar\n", reg_arg_name(env, argno));
13087 				bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name,
13088 						      "Pass an integer scalar length for this memory argument.",
13089 						      "the kfunc expects a scalar memory size, but %s is %s",
13090 						      reg_arg_name(env, argno),
13091 						      bpf_diag_reg_type_plain(env, reg->type));
13092 				return -EINVAL;
13093 			}
13094 
13095 			if (bpf_register_is_null(buff_reg))
13096 				break;
13097 
13098 			ret = check_mem_size_reg(env, buff_reg, size_reg, buff_argno, argno,
13099 						 BPF_READ | BPF_WRITE, true, meta, &failure);
13100 			if (ret < 0) {
13101 				const char *buff_arg, *size_arg;
13102 
13103 				buff_arg = bpf_diag_arg_name(env, buff_argno);
13104 				size_arg = bpf_diag_arg_name(env, argno);
13105 				verbose(env, "%s and ", reg_arg_name(env, buff_argno));
13106 				verbose(env, "%s memory, len pair leads to invalid memory access\n",
13107 					reg_arg_name(env, argno));
13108 				if (failure == BPF_MEM_SIZE_FAIL_MEMORY) {
13109 					bpf_diag_call_arg_fmt(env, insn_idx, buff_argno, func_name,
13110 							      "Pass a stack, map, context, or other verifier-known memory pointer, and keep the paired length within that object.",
13111 							      "it is the memory pointer in a memory/length pair with %s, but %s does not describe verifier-readable memory for the requested length",
13112 							      size_arg, buff_arg);
13113 				} else if (failure == BPF_MEM_SIZE_FAIL_SIZE) {
13114 					if (reg_smin(size_reg) < 0)
13115 						bpf_diag_call_arg_fmt(
13116 							env, insn_idx, argno, func_name,
13117 							"Constrain the memory size to a non-negative value smaller than BPF_MAX_VAR_SIZ before this call.",
13118 							"the memory size in %s may be negative because its signed minimum is %lld",
13119 							size_arg, reg_smin(size_reg));
13120 					else
13121 						bpf_diag_call_arg_fmt(
13122 							env, insn_idx, argno, func_name,
13123 							"Constrain the memory size to a non-negative value smaller than BPF_MAX_VAR_SIZ before this call.",
13124 							"the memory size in %s may reach %llu bytes, but variable memory accesses must stay below %u bytes",
13125 							size_arg, reg_umax(size_reg), BPF_MAX_VAR_SIZ);
13126 				}
13127 				return ret;
13128 			}
13129 			break;
13130 		}
13131 		case KF_ARG_PTR_TO_CALLBACK:
13132 			if (reg->type != PTR_TO_FUNC) {
13133 				verbose(env, "%s expected pointer to func\n", reg_arg_name(env, argno));
13134 				return -EINVAL;
13135 			}
13136 			meta->subprogno = reg->subprogno;
13137 			break;
13138 		case KF_ARG_PTR_TO_REFCOUNTED_KPTR:
13139 			if (!type_is_ptr_alloc_obj(reg->type)) {
13140 				verbose(env, "%s is neither owning or non-owning ref\n",
13141 					reg_arg_name(env, argno));
13142 				bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name,
13143 						      "Pass an owning or non-owning pointer to a BPF-managed object containing a bpf_refcount field.",
13144 						      "the kfunc expects a pointer to a BPF-managed refcounted object, but %s is %s",
13145 						      reg_arg_name(env, argno),
13146 						      bpf_diag_reg_type_plain(env, reg->type));
13147 				return -EINVAL;
13148 			}
13149 			if (!type_is_non_owning_ref(reg->type))
13150 				meta->arg_owning_ref = true;
13151 
13152 			rec = reg_btf_record(reg);
13153 			if (!rec) {
13154 				verifier_bug(env, "Couldn't find btf_record");
13155 				return -EFAULT;
13156 			}
13157 
13158 			if (rec->refcount_off < 0) {
13159 				verbose(env, "%s doesn't point to a type with bpf_refcount field\n",
13160 					reg_arg_name(env, argno));
13161 				return -EINVAL;
13162 			}
13163 
13164 			meta->arg_btf = reg->btf;
13165 			meta->arg_btf_id = reg->btf_id;
13166 			break;
13167 		case KF_ARG_PTR_TO_CONST_STR:
13168 			if (reg->type != PTR_TO_MAP_VALUE) {
13169 				verbose(env, "%s doesn't point to a const string\n",
13170 					reg_arg_name(env, argno));
13171 				bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name,
13172 						      "Pass a constant string pointer that the verifier recognizes, such as a string stored in a read-only map value.",
13173 						      "the kfunc expects a pointer to a constant string stored in verifier-known memory, but %s is %s",
13174 						      reg_arg_name(env, argno),
13175 						      bpf_diag_reg_type_plain(env, reg->type));
13176 				return -EINVAL;
13177 			}
13178 			ret = check_arg_const_str(env, reg, argno);
13179 			if (ret)
13180 				return ret;
13181 			break;
13182 		case KF_ARG_PTR_TO_WORKQUEUE:
13183 			if (reg->type != PTR_TO_MAP_VALUE) {
13184 				verbose(env, "%s doesn't point to a map value\n",
13185 					reg_arg_name(env, argno));
13186 				return -EINVAL;
13187 			}
13188 			ret = check_map_field_pointer(env, reg, argno, BPF_WORKQUEUE, &meta->map);
13189 			if (ret < 0)
13190 				return ret;
13191 			break;
13192 		case KF_ARG_PTR_TO_TIMER:
13193 			if (reg->type != PTR_TO_MAP_VALUE) {
13194 				verbose(env, "%s doesn't point to a map value\n",
13195 					reg_arg_name(env, argno));
13196 				return -EINVAL;
13197 			}
13198 			ret = process_timer_func(env, reg, argno, &meta->map);
13199 			if (ret < 0)
13200 				return ret;
13201 			break;
13202 		case KF_ARG_PTR_TO_TASK_WORK:
13203 			if (reg->type != PTR_TO_MAP_VALUE) {
13204 				verbose(env, "%s doesn't point to a map value\n",
13205 					reg_arg_name(env, argno));
13206 				return -EINVAL;
13207 			}
13208 			ret = check_map_field_pointer(env, reg, argno, BPF_TASK_WORK, &meta->map);
13209 			if (ret < 0)
13210 				return ret;
13211 			break;
13212 		case KF_ARG_PTR_TO_IRQ_FLAG:
13213 			if (reg->type != PTR_TO_STACK) {
13214 				verbose(env, "%s doesn't point to an irq flag on stack\n",
13215 					reg_arg_name(env, argno));
13216 				bpf_diag_call_arg_fmt(env, insn_idx, argno, func_name,
13217 						      "Pass the same stack slot used by bpf_local_irq_save() or bpf_res_spin_lock_irqsave().",
13218 						      "the kfunc expects a stack pointer to an IRQ flag slot, but %s is %s",
13219 						      reg_arg_name(env, argno),
13220 						      bpf_diag_reg_type_plain(env, reg->type));
13221 				return -EINVAL;
13222 			}
13223 			ret = process_irq_flag(env, reg, argno, meta);
13224 			if (ret < 0)
13225 				return ret;
13226 			break;
13227 		case KF_ARG_PTR_TO_RES_SPIN_LOCK:
13228 		{
13229 			int flags = PROCESS_RES_LOCK;
13230 
13231 			if (reg->type != PTR_TO_MAP_VALUE && reg->type != (PTR_TO_BTF_ID | MEM_ALLOC)) {
13232 				verbose(env, "%s doesn't point to map value or allocated object\n",
13233 					reg_arg_name(env, argno));
13234 				return -EINVAL;
13235 			}
13236 
13237 			if (!is_bpf_res_spin_lock_kfunc(meta->func_id))
13238 				return -EFAULT;
13239 			if (meta->func_id == special_kfunc_list[KF_bpf_res_spin_lock] ||
13240 			    meta->func_id == special_kfunc_list[KF_bpf_res_spin_lock_irqsave])
13241 				flags |= PROCESS_SPIN_LOCK;
13242 			if (meta->func_id == special_kfunc_list[KF_bpf_res_spin_lock_irqsave] ||
13243 			    meta->func_id == special_kfunc_list[KF_bpf_res_spin_unlock_irqrestore])
13244 				flags |= PROCESS_LOCK_IRQ;
13245 			ret = process_spin_lock(env, reg, argno, flags);
13246 			if (ret < 0)
13247 				return ret;
13248 			break;
13249 		}
13250 		}
13251 	}
13252 
13253 	return 0;
13254 }
13255 
bpf_fetch_kfunc_arg_meta(struct bpf_verifier_env * env,s32 func_id,s16 offset,struct bpf_call_arg_meta * meta)13256 int bpf_fetch_kfunc_arg_meta(struct bpf_verifier_env *env,
13257 			     s32 func_id,
13258 			     s16 offset,
13259 			     struct bpf_call_arg_meta *meta)
13260 {
13261 	struct bpf_kfunc_meta kfunc;
13262 	int err;
13263 
13264 	memset(meta, 0, sizeof(*meta));
13265 
13266 	err = fetch_kfunc_meta(env, func_id, offset, &kfunc);
13267 	if (err)
13268 		return err;
13269 
13270 	meta->btf = kfunc.btf;
13271 	meta->func_id = kfunc.id;
13272 	meta->func_proto = kfunc.proto;
13273 	meta->func_name = kfunc.name;
13274 
13275 	if (!kfunc.flags || !btf_kfunc_is_allowed(kfunc.btf, kfunc.id, env->prog))
13276 		return -EACCES;
13277 
13278 	meta->kfunc_flags = *kfunc.flags;
13279 
13280 	/* Only support release referenced argument passed by register */
13281 	if (is_kfunc_release(meta))
13282 		meta->release_regno = BPF_REG_1;
13283 
13284 	return 0;
13285 }
13286 
13287 /*
13288  * Determine how many bytes a helper accesses through a stack pointer at
13289  * argument position @arg (0-based, corresponding to R1-R5).
13290  *
13291  * Returns:
13292  *   > 0   known read access size in bytes
13293  *     0   doesn't read anything directly
13294  * S64_MIN unknown
13295  *   < 0   known write access of (-return) bytes
13296  */
bpf_helper_stack_access_bytes(struct bpf_verifier_env * env,struct bpf_insn * insn,int arg,int insn_idx)13297 s64 bpf_helper_stack_access_bytes(struct bpf_verifier_env *env, struct bpf_insn *insn,
13298 				  int arg, int insn_idx)
13299 {
13300 	struct bpf_insn_aux_data *aux = &env->insn_aux_data[insn_idx];
13301 	const struct bpf_func_proto *fn;
13302 	enum bpf_arg_type at;
13303 	s64 size;
13304 
13305 	if (bpf_get_helper_proto(env, insn->imm, &fn) < 0)
13306 		return S64_MIN;
13307 
13308 	at = fn->arg_type[arg];
13309 
13310 	switch (base_type(at)) {
13311 	case ARG_PTR_TO_MAP_KEY:
13312 	case ARG_PTR_TO_MAP_VALUE: {
13313 		bool is_key = base_type(at) == ARG_PTR_TO_MAP_KEY;
13314 		u64 val;
13315 		int i, map_reg;
13316 
13317 		for (i = 0; i < arg; i++) {
13318 			if (base_type(fn->arg_type[i]) == ARG_CONST_MAP_PTR)
13319 				break;
13320 		}
13321 		if (i >= arg)
13322 			goto scan_all_maps;
13323 
13324 		map_reg = BPF_REG_1 + i;
13325 
13326 		if (!(aux->const_reg_map_mask & BIT(map_reg)))
13327 			goto scan_all_maps;
13328 
13329 		i = aux->const_reg_vals[map_reg];
13330 		if (i < env->used_map_cnt) {
13331 			size = is_key ? env->used_maps[i]->key_size
13332 				      : env->used_maps[i]->value_size;
13333 			goto out;
13334 		}
13335 scan_all_maps:
13336 		/*
13337 		 * Map pointer is not known at this call site (e.g. different
13338 		 * maps on merged paths).  Conservatively return the largest
13339 		 * key_size or value_size across all maps used by the program.
13340 		 */
13341 		val = 0;
13342 		for (i = 0; i < env->used_map_cnt; i++) {
13343 			struct bpf_map *map = env->used_maps[i];
13344 			u32 sz = is_key ? map->key_size : map->value_size;
13345 
13346 			if (sz > val)
13347 				val = sz;
13348 			if (map->inner_map_meta) {
13349 				sz = is_key ? map->inner_map_meta->key_size
13350 					    : map->inner_map_meta->value_size;
13351 				if (sz > val)
13352 					val = sz;
13353 			}
13354 		}
13355 		if (!val)
13356 			return S64_MIN;
13357 		size = val;
13358 		goto out;
13359 	}
13360 	case ARG_PTR_TO_MEM:
13361 		if (at & MEM_FIXED_SIZE) {
13362 			size = fn->arg_size[arg];
13363 			goto out;
13364 		}
13365 		if (arg + 1 < ARRAY_SIZE(fn->arg_type) &&
13366 		    arg_type_is_mem_size(fn->arg_type[arg + 1])) {
13367 			int size_reg = BPF_REG_1 + arg + 1;
13368 
13369 			if (aux->const_reg_mask & BIT(size_reg)) {
13370 				size = (s64)aux->const_reg_vals[size_reg];
13371 				goto out;
13372 			}
13373 			/*
13374 			 * Size arg is const on each path but differs across merged
13375 			 * paths. MAX_BPF_STACK is a safe upper bound for reads.
13376 			 */
13377 			if (at & MEM_UNINIT)
13378 				return 0;
13379 			return MAX_BPF_STACK;
13380 		}
13381 		return S64_MIN;
13382 	case ARG_PTR_TO_DYNPTR:
13383 		size = BPF_DYNPTR_SIZE;
13384 		break;
13385 	case ARG_PTR_TO_STACK:
13386 		/*
13387 		 * Only used by bpf_calls_callback() helpers. The helper itself
13388 		 * doesn't access stack. The callback subprog does and it's
13389 		 * analyzed separately.
13390 		 */
13391 		return 0;
13392 	default:
13393 		return S64_MIN;
13394 	}
13395 out:
13396 	/*
13397 	 * MEM_UNINIT args are write-only: the helper initializes the
13398 	 * buffer without reading it.
13399 	 */
13400 	if (at & MEM_UNINIT)
13401 		return -size;
13402 	return size;
13403 }
13404 
13405 /*
13406  * Determine how many bytes a kfunc accesses through a stack pointer at
13407  * argument position @arg (0-based, corresponding to R1-R5).
13408  *
13409  * Returns:
13410  *   > 0      known read access size in bytes
13411  *     0      doesn't access memory through that argument (ex: not a pointer)
13412  *   S64_MIN  unknown
13413  *   < 0      known write access of (-return) bytes
13414  */
bpf_kfunc_stack_access_bytes(struct bpf_verifier_env * env,struct bpf_insn * insn,int arg,int insn_idx)13415 s64 bpf_kfunc_stack_access_bytes(struct bpf_verifier_env *env, struct bpf_insn *insn,
13416 				 int arg, int insn_idx)
13417 {
13418 	struct bpf_insn_aux_data *aux = &env->insn_aux_data[insn_idx];
13419 	struct bpf_call_arg_meta meta;
13420 	const struct btf_param *args;
13421 	const struct btf_type *t, *ref_t;
13422 	const struct btf *btf;
13423 	u32 nargs, type_size;
13424 	s64 size;
13425 
13426 	if (bpf_fetch_kfunc_arg_meta(env, insn->imm, insn->off, &meta) < 0)
13427 		return S64_MIN;
13428 
13429 	btf = meta.btf;
13430 	args = btf_params(meta.func_proto);
13431 	nargs = btf_type_vlen(meta.func_proto);
13432 	if (arg >= nargs)
13433 		return 0;
13434 
13435 	t = btf_type_skip_modifiers(btf, args[arg].type, NULL);
13436 	if (!btf_type_is_ptr(t))
13437 		return 0;
13438 
13439 	/* dynptr: fixed 16-byte on-stack representation */
13440 	if (is_kfunc_arg_dynptr(btf, &args[arg])) {
13441 		size = BPF_DYNPTR_SIZE;
13442 		goto out;
13443 	}
13444 
13445 	/* ptr + __sz/__szk pair: size is in the next register */
13446 	if (arg + 1 < nargs &&
13447 	    (btf_param_match_suffix(btf, &args[arg + 1], "__sz") ||
13448 	     btf_param_match_suffix(btf, &args[arg + 1], "__szk"))) {
13449 		int size_reg = BPF_REG_1 + arg + 1;
13450 
13451 		if (aux->const_reg_mask & BIT(size_reg)) {
13452 			size = (s64)aux->const_reg_vals[size_reg];
13453 			goto out;
13454 		}
13455 		return MAX_BPF_STACK;
13456 	}
13457 
13458 	/* fixed-size pointed-to type: resolve via BTF */
13459 	ref_t = btf_type_skip_modifiers(btf, t->type, NULL);
13460 	if (!IS_ERR(btf_resolve_size(btf, ref_t, &type_size))) {
13461 		size = type_size;
13462 		goto out;
13463 	}
13464 
13465 	return S64_MIN;
13466 out:
13467 	/* KF_ITER_NEW kfuncs initialize the iterator state at arg 0 */
13468 	if (arg == 0 && meta.kfunc_flags & KF_ITER_NEW)
13469 		return -size;
13470 	if (is_kfunc_arg_uninit(btf, &args[arg]))
13471 		return -size;
13472 	return size;
13473 }
13474 
13475 /* check special kfuncs and return:
13476  *  1  - not fall-through to 'else' branch, continue verification
13477  *  0  - fall-through to 'else' branch
13478  * < 0 - not fall-through to 'else' branch, return error
13479  */
check_special_kfunc(struct bpf_verifier_env * env,struct bpf_call_arg_meta * meta,struct bpf_reg_state * regs,struct bpf_insn_aux_data * insn_aux,const struct btf_type * ptr_type,struct btf * desc_btf)13480 static int check_special_kfunc(struct bpf_verifier_env *env, struct bpf_call_arg_meta *meta,
13481 			       struct bpf_reg_state *regs, struct bpf_insn_aux_data *insn_aux,
13482 			       const struct btf_type *ptr_type, struct btf *desc_btf)
13483 {
13484 	const struct btf_type *ret_t;
13485 	int err = 0;
13486 
13487 	if (meta->btf != btf_vmlinux)
13488 		return 0;
13489 
13490 	if (is_bpf_obj_new_kfunc(meta->func_id) || is_bpf_percpu_obj_new_kfunc(meta->func_id)) {
13491 		struct btf_struct_meta *struct_meta;
13492 		struct btf *ret_btf;
13493 		u32 ret_btf_id;
13494 
13495 		if (is_bpf_obj_new_kfunc(meta->func_id) && !bpf_global_ma_set)
13496 			return -ENOMEM;
13497 
13498 		if (((u64)(u32)meta->arg_constant.value) != meta->arg_constant.value) {
13499 			verbose(env, "local type ID argument must be in range [0, U32_MAX]\n");
13500 			return -EINVAL;
13501 		}
13502 
13503 		ret_btf = env->prog->aux->btf;
13504 		ret_btf_id = meta->arg_constant.value;
13505 
13506 		/* This may be NULL due to user not supplying a BTF */
13507 		if (!ret_btf) {
13508 			verbose(env, "bpf_obj_new/bpf_percpu_obj_new requires prog BTF\n");
13509 			return -EINVAL;
13510 		}
13511 
13512 		ret_t = btf_type_by_id(ret_btf, ret_btf_id);
13513 		if (!ret_t || !__btf_type_is_struct(ret_t)) {
13514 			verbose(env, "bpf_obj_new/bpf_percpu_obj_new type ID argument must be of a struct\n");
13515 			return -EINVAL;
13516 		}
13517 
13518 		if (is_bpf_percpu_obj_new_kfunc(meta->func_id)) {
13519 			if (ret_t->size > BPF_GLOBAL_PERCPU_MA_MAX_SIZE) {
13520 				verbose(env, "bpf_percpu_obj_new type size (%d) is greater than %d\n",
13521 					ret_t->size, BPF_GLOBAL_PERCPU_MA_MAX_SIZE);
13522 				return -EINVAL;
13523 			}
13524 
13525 			if (!bpf_global_percpu_ma_set) {
13526 				mutex_lock(&bpf_percpu_ma_lock);
13527 				if (!bpf_global_percpu_ma_set) {
13528 					/* Charge memory allocated with bpf_global_percpu_ma to
13529 					 * root memcg. The obj_cgroup for root memcg is NULL.
13530 					 */
13531 					err = bpf_mem_alloc_percpu_init(&bpf_global_percpu_ma, NULL);
13532 					if (!err)
13533 						bpf_global_percpu_ma_set = true;
13534 				}
13535 				mutex_unlock(&bpf_percpu_ma_lock);
13536 				if (err)
13537 					return err;
13538 			}
13539 
13540 			mutex_lock(&bpf_percpu_ma_lock);
13541 			err = bpf_mem_alloc_percpu_unit_init(&bpf_global_percpu_ma, ret_t->size);
13542 			mutex_unlock(&bpf_percpu_ma_lock);
13543 			if (err)
13544 				return err;
13545 		}
13546 
13547 		struct_meta = btf_find_struct_meta(ret_btf, ret_btf_id);
13548 		if (is_bpf_percpu_obj_new_kfunc(meta->func_id)) {
13549 			if (!__btf_type_is_scalar_struct(env, ret_btf, ret_t, 0)) {
13550 				verbose(env, "bpf_percpu_obj_new type ID argument must be of a struct of scalars\n");
13551 				return -EINVAL;
13552 			}
13553 
13554 			if (struct_meta) {
13555 				verbose(env, "bpf_percpu_obj_new type ID argument must not contain special fields\n");
13556 				return -EINVAL;
13557 			}
13558 		}
13559 
13560 		mark_reg_known_zero(env, regs, BPF_REG_0);
13561 		regs[BPF_REG_0].type = PTR_TO_BTF_ID | MEM_ALLOC;
13562 		regs[BPF_REG_0].btf = ret_btf;
13563 		regs[BPF_REG_0].btf_id = ret_btf_id;
13564 		if (is_bpf_percpu_obj_new_kfunc(meta->func_id))
13565 			regs[BPF_REG_0].type |= MEM_PERCPU;
13566 
13567 		insn_aux->obj_new_size = ret_t->size;
13568 		insn_aux->kptr_struct_meta = struct_meta;
13569 	} else if (is_bpf_refcount_acquire_kfunc(meta->func_id)) {
13570 		mark_reg_known_zero(env, regs, BPF_REG_0);
13571 		regs[BPF_REG_0].type = PTR_TO_BTF_ID | MEM_ALLOC;
13572 		regs[BPF_REG_0].btf = meta->arg_btf;
13573 		regs[BPF_REG_0].btf_id = meta->arg_btf_id;
13574 
13575 		insn_aux->kptr_struct_meta =
13576 			btf_find_struct_meta(meta->arg_btf,
13577 					     meta->arg_btf_id);
13578 	} else if (is_list_node_type(ptr_type)) {
13579 		struct btf_field *field = meta->arg_list_head.field;
13580 
13581 		mark_reg_graph_node(regs, BPF_REG_0, &field->graph_root);
13582 	} else if (is_rbtree_node_type(ptr_type)) {
13583 		struct btf_field *field = meta->arg_rbtree_root.field;
13584 
13585 		mark_reg_graph_node(regs, BPF_REG_0, &field->graph_root);
13586 	} else if (meta->func_id == special_kfunc_list[KF_bpf_cast_to_kern_ctx]) {
13587 		mark_reg_known_zero(env, regs, BPF_REG_0);
13588 		regs[BPF_REG_0].type = PTR_TO_BTF_ID | PTR_TRUSTED;
13589 		regs[BPF_REG_0].btf = desc_btf;
13590 		regs[BPF_REG_0].btf_id = meta->ret_btf_id;
13591 	} else if (meta->func_id == special_kfunc_list[KF_bpf_rdonly_cast]) {
13592 		ret_t = btf_type_by_id(desc_btf, meta->arg_constant.value);
13593 		if (!ret_t) {
13594 			verbose(env, "Unknown type ID %lld passed to kfunc bpf_rdonly_cast\n",
13595 				meta->arg_constant.value);
13596 			return -EINVAL;
13597 		} else if (btf_type_is_struct(ret_t)) {
13598 			mark_reg_known_zero(env, regs, BPF_REG_0);
13599 			regs[BPF_REG_0].type = PTR_TO_BTF_ID | PTR_UNTRUSTED;
13600 			regs[BPF_REG_0].btf = desc_btf;
13601 			regs[BPF_REG_0].btf_id = meta->arg_constant.value;
13602 		} else if (btf_type_is_void(ret_t)) {
13603 			mark_reg_known_zero(env, regs, BPF_REG_0);
13604 			regs[BPF_REG_0].type = PTR_TO_MEM | MEM_RDONLY | PTR_UNTRUSTED;
13605 			regs[BPF_REG_0].mem_size = 0;
13606 		} else {
13607 			verbose(env,
13608 				"kfunc bpf_rdonly_cast type ID argument must be of a struct or void\n");
13609 			return -EINVAL;
13610 		}
13611 	} else if (meta->func_id == special_kfunc_list[KF_bpf_dynptr_slice] ||
13612 		   meta->func_id == special_kfunc_list[KF_bpf_dynptr_slice_rdwr]) {
13613 		enum bpf_type_flag type_flag = get_dynptr_type_flag(meta->dynptr.type);
13614 
13615 		mark_reg_known_zero(env, regs, BPF_REG_0);
13616 
13617 		if (!meta->arg_constant.found) {
13618 			verifier_bug(env, "bpf_dynptr_slice(_rdwr) no constant size");
13619 			return -EFAULT;
13620 		}
13621 
13622 		regs[BPF_REG_0].mem_size = meta->arg_constant.value;
13623 
13624 		/* PTR_MAYBE_NULL will be added when is_kfunc_ret_null is checked */
13625 		regs[BPF_REG_0].type = PTR_TO_MEM | type_flag;
13626 
13627 		if (meta->func_id == special_kfunc_list[KF_bpf_dynptr_slice]) {
13628 			regs[BPF_REG_0].type |= MEM_RDONLY;
13629 		} else {
13630 			/* this will set env->seen_direct_write to true */
13631 			if (!may_access_direct_pkt_data(env, NULL, BPF_WRITE)) {
13632 				verbose(env, "the prog does not allow writes to packet data\n");
13633 				return -EINVAL;
13634 			}
13635 		}
13636 
13637 		if (!meta->dynptr.id) {
13638 			verifier_bug(env, "no dynptr id");
13639 			return -EFAULT;
13640 		}
13641 		regs[BPF_REG_0].parent_id = meta->dynptr.id;
13642 	} else {
13643 		return 0;
13644 	}
13645 
13646 	return 1;
13647 }
13648 
13649 static int check_return_code(struct bpf_verifier_env *env, int regno, const char *reg_name);
13650 
check_kfunc_call(struct bpf_verifier_env * env,struct bpf_insn * insn,int * insn_idx_p)13651 static int check_kfunc_call(struct bpf_verifier_env *env, struct bpf_insn *insn,
13652 			    int *insn_idx_p)
13653 {
13654 	bool sleepable, rcu_lock, rcu_unlock, preempt_disable, preempt_enable;
13655 	enum bpf_prog_type prog_type = resolve_prog_type(env->prog);
13656 	struct bpf_reg_state *regs = cur_regs(env);
13657 	const char *func_name, *ptr_type_name;
13658 	const struct btf_type *t, *ptr_type;
13659 	struct bpf_call_arg_meta meta;
13660 	struct bpf_insn_aux_data *insn_aux;
13661 	const char *operation;
13662 	int err, insn_idx = *insn_idx_p;
13663 	u32 i, nargs, ptr_type_id;
13664 	struct bpf_kfunc_desc *desc;
13665 	struct btf *desc_btf;
13666 	int id;
13667 
13668 	/* skip for now, but return error when we find this in fixup_kfunc_call */
13669 	if (!insn->imm)
13670 		return 0;
13671 
13672 	err = bpf_fetch_kfunc_arg_meta(env, insn->imm, insn->off, &meta);
13673 	if (err == -EACCES && meta.func_name) {
13674 		verbose(env, "calling kernel function %s is not allowed\n", meta.func_name);
13675 		operation = bpf_diag_fmt(env, "kfunc %s", meta.func_name);
13676 		bpf_diag_policy(
13677 			env, insn_idx, operation, "this program cannot call the kfunc",
13678 			"Use a kfunc allowed for this program type and attach point, or change the program context.");
13679 	}
13680 	if (err)
13681 		return err;
13682 	desc_btf = meta.btf;
13683 	func_name = meta.func_name;
13684 	insn_aux = &env->insn_aux_data[insn_idx];
13685 
13686 	desc = find_kfunc_desc(env->prog, insn->imm, insn->off);
13687 	if (!desc) {
13688 		verifier_bug(env, "kfunc descriptor not found for func_id %u", insn->imm);
13689 		return -EFAULT;
13690 	}
13691 	meta.fn = &desc->proto;
13692 
13693 	insn_aux->is_iter_next = bpf_is_iter_next_kfunc(&meta);
13694 
13695 	if (!insn->off &&
13696 	    (insn->imm == special_kfunc_list[KF_bpf_res_spin_lock] ||
13697 	     insn->imm == special_kfunc_list[KF_bpf_res_spin_lock_irqsave])) {
13698 		struct bpf_verifier_state *branch;
13699 		struct bpf_reg_state *regs;
13700 
13701 		branch = push_stack(env, env->insn_idx + 1, env->insn_idx, false);
13702 		if (IS_ERR(branch)) {
13703 			verbose(env, "failed to push state for failed lock acquisition\n");
13704 			return PTR_ERR(branch);
13705 		}
13706 
13707 		regs = branch->frame[branch->curframe]->regs;
13708 
13709 		/* Clear r0-r5 registers in forked state */
13710 		for (i = 0; i < CALLER_SAVED_REGS; i++)
13711 			bpf_mark_reg_not_init(env, &regs[caller_saved[i]]);
13712 
13713 		mark_reg_unknown(env, regs, BPF_REG_0);
13714 		err = __mark_reg_s32_range(env, regs, BPF_REG_0, -MAX_ERRNO, -1);
13715 		if (err) {
13716 			verbose(env, "failed to mark s32 range for retval in forked state for lock\n");
13717 			return err;
13718 		}
13719 	} else if (!insn->off && insn->imm == special_kfunc_list[KF___bpf_trap]) {
13720 		verbose(env, "unexpected __bpf_trap() due to uninitialized variable?\n");
13721 		return -EFAULT;
13722 	}
13723 
13724 	if (is_kfunc_destructive(&meta) && !capable(CAP_SYS_BOOT)) {
13725 		verbose(env, "destructive kfunc calls require CAP_SYS_BOOT capability\n");
13726 		operation = bpf_diag_fmt(env, "destructive kfunc %s", meta.func_name);
13727 		bpf_diag_policy(
13728 			env, insn_idx, operation, "destructive kfuncs require CAP_SYS_BOOT",
13729 			"Load the program with CAP_SYS_BOOT, or avoid destructive kfuncs.");
13730 		return -EACCES;
13731 	}
13732 
13733 	sleepable = bpf_is_kfunc_sleepable(&meta);
13734 	if (sleepable && !in_sleepable(env)) {
13735 		verbose(env, "program must be sleepable to call sleepable kfunc %s\n", func_name);
13736 		operation = bpf_diag_fmt(env, "sleepable kfunc %s", func_name);
13737 		bpf_diag_ctx_forbidden(env, insn_idx, operation,
13738 			"Mark the program sleepable if the program type allows it, or use a non-sleepable kfunc.");
13739 		return -EACCES;
13740 	}
13741 
13742 	/* Track non-sleepable context for kfuncs, same as for helpers. */
13743 	if (!in_sleepable_context(env))
13744 		insn_aux->non_sleepable = true;
13745 
13746 	/* Check the arguments */
13747 	err = check_kfunc_args(env, &meta, insn_idx);
13748 	if (err < 0)
13749 		return err;
13750 
13751 	if ((is_bpf_obj_drop_kfunc(meta.func_id) ||
13752 	     is_bpf_percpu_obj_drop_kfunc(meta.func_id)) && (is_tracing_prog_type(prog_type) ||
13753 	     /* is_tracing_prog_type() for now doesn't cover non-iterator tracing progs. */
13754 	     (prog_type == BPF_PROG_TYPE_TRACING && env->prog->expected_attach_type != BPF_TRACE_ITER
13755 	      && !env->prog->sleepable))) {
13756 		struct btf_struct_meta *struct_meta;
13757 
13758 		struct_meta = btf_find_struct_meta(meta.arg_btf, meta.arg_btf_id);
13759 		if (struct_meta && btf_record_has_nmi_unsafe_fields(struct_meta->record)) {
13760 			verbose(env, "%s cannot be used in tracing programs on types with NMI unsafe fields\n",
13761 				func_name);
13762 			return -EINVAL;
13763 		}
13764 	}
13765 
13766 	if (is_bpf_rbtree_add_kfunc(meta.func_id)) {
13767 		err = push_callback_call(env, insn, insn_idx, meta.subprogno,
13768 					 set_rbtree_add_callback_state);
13769 		if (err) {
13770 			verbose(env, "kfunc %s#%d failed callback verification\n",
13771 				func_name, meta.func_id);
13772 			return err;
13773 		}
13774 	}
13775 
13776 	if (is_bpf_wq_set_callback_kfunc(meta.func_id)) {
13777 		err = push_callback_call(env, insn, insn_idx, meta.subprogno,
13778 					 set_timer_callback_state);
13779 		if (err) {
13780 			verbose(env, "kfunc %s#%d failed callback verification\n",
13781 				func_name, meta.func_id);
13782 			return err;
13783 		}
13784 	}
13785 
13786 	if (is_task_work_add_kfunc(meta.func_id)) {
13787 		err = push_callback_call(env, insn, insn_idx, meta.subprogno,
13788 					 set_task_work_schedule_callback_state);
13789 		if (err) {
13790 			verbose(env, "kfunc %s#%d failed callback verification\n",
13791 				func_name, meta.func_id);
13792 			return err;
13793 		}
13794 	}
13795 
13796 	rcu_lock = is_kfunc_bpf_rcu_read_lock(&meta);
13797 	rcu_unlock = is_kfunc_bpf_rcu_read_unlock(&meta);
13798 
13799 	preempt_disable = is_kfunc_bpf_preempt_disable(&meta);
13800 	preempt_enable = is_kfunc_bpf_preempt_enable(&meta);
13801 
13802 	if (rcu_lock) {
13803 		env->cur_state->active_rcu_locks++;
13804 		bpf_diag_record_context(env, insn_idx, BPF_DIAG_CONTEXT_RCU, true,
13805 					env->cur_state->active_rcu_locks);
13806 	} else if (rcu_unlock) {
13807 		if (env->cur_state->active_rcu_locks == 0) {
13808 			verbose(env, "unmatched rcu read unlock (kernel function %s)\n", func_name);
13809 			bpf_diag_ctx_underflow(
13810 				env, insn_idx, func_name, BPF_DIAG_CONTEXT_RCU,
13811 				"Remove the extra bpf_rcu_read_unlock() call, or ensure this path first enters an RCU read lock region.");
13812 			return -EINVAL;
13813 		}
13814 		env->cur_state->active_rcu_locks--;
13815 		bpf_diag_record_context(env, insn_idx, BPF_DIAG_CONTEXT_RCU, false,
13816 					env->cur_state->active_rcu_locks);
13817 		if (!in_rcu_cs(env))
13818 			invalidate_rcu_protected_refs(env);
13819 	} else if (preempt_disable) {
13820 		env->cur_state->active_preempt_locks++;
13821 		bpf_diag_record_context(env, insn_idx, BPF_DIAG_CONTEXT_PREEMPT, true,
13822 					env->cur_state->active_preempt_locks);
13823 	} else if (preempt_enable) {
13824 		if (env->cur_state->active_preempt_locks == 0) {
13825 			verbose(env, "unmatched attempt to enable preemption (kernel function %s)\n", func_name);
13826 			bpf_diag_ctx_underflow(
13827 				env, insn_idx, func_name, BPF_DIAG_CONTEXT_PREEMPT,
13828 				"Remove the extra bpf_preempt_enable() call, or ensure this path first disables preemption.");
13829 			return -EINVAL;
13830 		}
13831 		env->cur_state->active_preempt_locks--;
13832 		bpf_diag_record_context(env, insn_idx, BPF_DIAG_CONTEXT_PREEMPT, false,
13833 					env->cur_state->active_preempt_locks);
13834 		if (!in_rcu_cs(env))
13835 			invalidate_rcu_protected_refs(env);
13836 	}
13837 
13838 	if (sleepable && !in_sleepable_context(env)) {
13839 		verbose(env, "kernel func %s is sleepable within %s\n",
13840 			func_name, non_sleepable_context_description(env));
13841 		operation = bpf_diag_fmt(env, "sleepable kfunc %s", func_name);
13842 		bpf_diag_ctx_forbidden(env, insn_idx, operation,
13843 			"Move the kfunc call outside the critical section, or use a non-sleepable kfunc.");
13844 		return -EACCES;
13845 	}
13846 
13847 	if (in_rbtree_lock_required_cb(env) && (rcu_lock || rcu_unlock)) {
13848 		verbose(env, "Calling bpf_rcu_read_{lock,unlock} in unnecessary rbtree callback\n");
13849 		return -EACCES;
13850 	}
13851 
13852 	if (is_kfunc_rcu_protected(&meta) && !in_rcu_cs(env)) {
13853 		verbose(env, "kernel func %s requires RCU critical section protection\n", func_name);
13854 		bpf_diag_ctx_required(
13855 			env, insn_idx, func_name, BPF_DIAG_CONTEXT_RCU,
13856 			"Call this kfunc between bpf_rcu_read_lock() and bpf_rcu_read_unlock(), keeping all exit paths balanced.");
13857 		return -EACCES;
13858 	}
13859 
13860 	/* In case of release function, we get register number of refcounted
13861 	 * PTR_TO_BTF_ID in bpf_kfunc_arg_meta, do the release now.
13862 	 */
13863 	if (meta.release_regno) {
13864 		err = release_reg(env, &regs[meta.release_regno], false, !!meta.dynptr.id);
13865 		if (err)
13866 			return err;
13867 	}
13868 
13869 	if (is_bpf_list_push_kfunc(meta.func_id) || is_bpf_rbtree_add_kfunc(meta.func_id)) {
13870 		id = regs[BPF_REG_2].id;
13871 		insn_aux->insert_off = regs[BPF_REG_2].var_off.value;
13872 		insn_aux->kptr_struct_meta = btf_find_struct_meta(meta.arg_btf, meta.arg_btf_id);
13873 		ref_convert_owning_non_owning(env, id);
13874 	}
13875 
13876 	if (meta.func_id == special_kfunc_list[KF_bpf_throw]) {
13877 		if (!bpf_jit_supports_exceptions()) {
13878 			verbose(env, "JIT does not support calling kfunc %s#%d\n",
13879 				func_name, meta.func_id);
13880 			return -ENOTSUPP;
13881 		}
13882 		env->seen_exception = true;
13883 
13884 		/* In the case of the default callback, the cookie value passed
13885 		 * to bpf_throw becomes the return value of the program.
13886 		 */
13887 		if (!env->exception_callback_subprog) {
13888 			err = check_return_code(env, BPF_REG_1, "R1");
13889 			if (err < 0)
13890 				return err;
13891 		}
13892 	}
13893 
13894 	bpf_diag_record_caller_saved(env, regs);
13895 	bpf_diag_mod_begin(env, &regs[BPF_REG_0], NULL, BPF_DIAG_MOD_WRITE);
13896 	for (i = 0; i < CALLER_SAVED_REGS; i++) {
13897 		u32 regno = caller_saved[i];
13898 
13899 		bpf_mark_reg_not_init(env, &regs[regno]);
13900 	}
13901 	invalidate_outgoing_stack_args(env, cur_func(env));
13902 
13903 	/* Check return type */
13904 	t = btf_type_skip_modifiers(desc_btf, meta.func_proto->type, NULL);
13905 
13906 	if (is_kfunc_acquire(&meta) && !btf_type_is_struct_ptr(meta.btf, t)) {
13907 		if (meta.btf != btf_vmlinux ||
13908 		    (!is_bpf_obj_new_kfunc(meta.func_id) &&
13909 		     !is_bpf_percpu_obj_new_kfunc(meta.func_id) &&
13910 		     !is_bpf_refcount_acquire_kfunc(meta.func_id))) {
13911 			verbose(env, "acquire kernel function does not return PTR_TO_BTF_ID\n");
13912 			return -EINVAL;
13913 		}
13914 	}
13915 
13916 	if (btf_type_is_scalar(t)) {
13917 		mark_reg_unknown(env, regs, BPF_REG_0);
13918 		if (meta.btf == btf_vmlinux && (meta.func_id == special_kfunc_list[KF_bpf_res_spin_lock] ||
13919 		    meta.func_id == special_kfunc_list[KF_bpf_res_spin_lock_irqsave]))
13920 			__mark_reg_const_zero(env, &regs[BPF_REG_0]);
13921 	} else if (btf_type_is_ptr(t)) {
13922 		ptr_type = btf_type_skip_modifiers(desc_btf, t->type, &ptr_type_id);
13923 		err = check_special_kfunc(env, &meta, regs, insn_aux, ptr_type, desc_btf);
13924 		if (err) {
13925 			if (err < 0)
13926 				return err;
13927 		} else if (btf_type_is_void(ptr_type)) {
13928 			/* kfunc returning 'void *' is equivalent to returning scalar */
13929 			mark_reg_unknown(env, regs, BPF_REG_0);
13930 		} else if (!__btf_type_is_struct(ptr_type)) {
13931 			if (!meta.ret_mem.found) {
13932 				__u32 sz;
13933 
13934 				if (!IS_ERR(btf_resolve_size(desc_btf, ptr_type, &sz))) {
13935 					meta.ret_mem.found = true;
13936 					meta.ret_mem.size = sz;
13937 					meta.r0_rdonly = true;
13938 				}
13939 
13940 				if (meta.func_id == special_kfunc_list[KF_bpf_session_cookie])
13941 					meta.r0_rdonly = false;
13942 			}
13943 			if (!meta.ret_mem.found) {
13944 				ptr_type_name = btf_name_by_offset(desc_btf,
13945 								   ptr_type->name_off);
13946 				verbose(env,
13947 					"kernel function %s returns pointer type %s %s is not supported\n",
13948 					func_name,
13949 					btf_type_str(ptr_type),
13950 					ptr_type_name);
13951 				return -EINVAL;
13952 			}
13953 
13954 			mark_reg_known_zero(env, regs, BPF_REG_0);
13955 			regs[BPF_REG_0].type = PTR_TO_MEM;
13956 			regs[BPF_REG_0].mem_size = meta.ret_mem.size;
13957 
13958 			if (meta.r0_rdonly)
13959 				regs[BPF_REG_0].type |= MEM_RDONLY;
13960 
13961 			/* Ensures we don't access the memory after a release_reference() */
13962 			if (meta.ref_obj.id) {
13963 				err = validate_ref_obj(env, &meta.ref_obj);
13964 				if (err)
13965 					return err;
13966 				regs[BPF_REG_0].parent_id = meta.ref_obj.id;
13967 			}
13968 
13969 			if (is_kfunc_rcu_protected(&meta))
13970 				regs[BPF_REG_0].type |= MEM_RCU;
13971 		} else {
13972 			enum bpf_reg_type type = PTR_TO_BTF_ID;
13973 
13974 			if (meta.func_id == special_kfunc_list[KF_bpf_get_kmem_cache])
13975 				type |= PTR_UNTRUSTED;
13976 			else if (is_kfunc_rcu_protected(&meta) ||
13977 				 (bpf_is_iter_next_kfunc(&meta) &&
13978 				  (get_iter_from_state(env->cur_state, &meta)
13979 					   ->type & MEM_RCU))) {
13980 				/*
13981 				 * If the iterator's constructor (the _new
13982 				 * function e.g., bpf_iter_task_new) has been
13983 				 * annotated with BPF kfunc flag
13984 				 * KF_RCU_PROTECTED and was called within a RCU
13985 				 * read-side critical section, also propagate
13986 				 * the MEM_RCU flag to the pointer returned from
13987 				 * the iterator's next function (e.g.,
13988 				 * bpf_iter_task_next).
13989 				 */
13990 				type |= MEM_RCU;
13991 			} else {
13992 				/*
13993 				 * Any PTR_TO_BTF_ID that is returned from a BPF
13994 				 * kfunc should by default be treated as
13995 				 * implicitly trusted.
13996 				 */
13997 				type |= PTR_TRUSTED;
13998 			}
13999 
14000 			mark_reg_known_zero(env, regs, BPF_REG_0);
14001 			regs[BPF_REG_0].btf = desc_btf;
14002 			regs[BPF_REG_0].type = type;
14003 			regs[BPF_REG_0].btf_id = ptr_type_id;
14004 		}
14005 
14006 		if (is_kfunc_ret_null(&meta)) {
14007 			regs[BPF_REG_0].type |= PTR_MAYBE_NULL;
14008 			/* For mark_ptr_or_null_reg, see 93c230e3f5bd6 */
14009 			regs[BPF_REG_0].id = ++env->id_gen;
14010 		}
14011 		if (is_kfunc_acquire(&meta)) {
14012 			id = acquire_reference(env, insn_idx, 0);
14013 			if (id < 0)
14014 				return id;
14015 			regs[BPF_REG_0].id = id;
14016 		} else if (is_rbtree_node_type(ptr_type) || is_list_node_type(ptr_type)) {
14017 			ref_set_non_owning(env, &regs[BPF_REG_0]);
14018 		}
14019 
14020 		if (reg_may_point_to_spin_lock(&regs[BPF_REG_0]) && !regs[BPF_REG_0].id)
14021 			regs[BPF_REG_0].id = ++env->id_gen;
14022 	} else if (btf_type_is_void(t)) {
14023 		if (meta.btf == btf_vmlinux) {
14024 			if (is_bpf_obj_drop_kfunc(meta.func_id) ||
14025 			    is_bpf_percpu_obj_drop_kfunc(meta.func_id)) {
14026 				insn_aux->kptr_struct_meta =
14027 					btf_find_struct_meta(meta.arg_btf,
14028 							     meta.arg_btf_id);
14029 			}
14030 		}
14031 	}
14032 
14033 	if (bpf_is_kfunc_pkt_changing(&meta))
14034 		clear_all_pkt_pointers(env);
14035 
14036 	nargs = btf_type_vlen(meta.func_proto);
14037 	if (nargs > MAX_BPF_FUNC_REG_ARGS) {
14038 		struct bpf_func_state *caller = cur_func(env);
14039 		struct bpf_subprog_info *caller_info = &env->subprog_info[caller->subprogno];
14040 		u16 out_stack_arg_cnt = nargs - MAX_BPF_FUNC_REG_ARGS;
14041 		u16 stack_arg_cnt = bpf_in_stack_arg_cnt(caller_info) + out_stack_arg_cnt;
14042 
14043 		if (stack_arg_cnt > caller_info->stack_arg_cnt)
14044 			caller_info->stack_arg_cnt = stack_arg_cnt;
14045 	}
14046 
14047 	/*
14048 	 * Record R0 before process_iter_next_call() snapshots the alternate
14049 	 * iterator path's diagnostic position.
14050 	 */
14051 	bpf_diag_mod_end(env);
14052 
14053 	if (bpf_is_iter_next_kfunc(&meta)) {
14054 		err = process_iter_next_call(env, insn_idx, &meta);
14055 		if (err)
14056 			return err;
14057 	}
14058 
14059 	if (meta.func_id == special_kfunc_list[KF_bpf_session_cookie])
14060 		env->prog->call_session_cookie = true;
14061 
14062 	if (bpf_is_throw_kfunc(insn))
14063 		return process_bpf_exit_full(env, NULL, true);
14064 
14065 	return 0;
14066 }
14067 
check_reg_sane_offset_scalar(struct bpf_verifier_env * env,const struct bpf_reg_state * reg,enum bpf_reg_type type)14068 static bool check_reg_sane_offset_scalar(struct bpf_verifier_env *env,
14069 					 const struct bpf_reg_state *reg,
14070 					 enum bpf_reg_type type)
14071 {
14072 	bool known = tnum_is_const(reg->var_off);
14073 	s64 val = reg->var_off.value;
14074 	s64 smin = reg_smin(reg);
14075 
14076 	if (known && (val >= BPF_MAX_VAR_OFF || val <= -BPF_MAX_VAR_OFF)) {
14077 		verbose(env, "math between %s pointer and %lld is not allowed\n",
14078 			reg_type_str(env, type), val);
14079 		return false;
14080 	}
14081 
14082 	if (smin == S64_MIN) {
14083 		verbose(env, "math between %s pointer and register with unbounded min value is not allowed\n",
14084 			reg_type_str(env, type));
14085 		return false;
14086 	}
14087 
14088 	if (smin >= BPF_MAX_VAR_OFF || smin <= -BPF_MAX_VAR_OFF) {
14089 		verbose(env, "value %lld makes %s pointer be out of bounds\n",
14090 			smin, reg_type_str(env, type));
14091 		return false;
14092 	}
14093 
14094 	return true;
14095 }
14096 
check_reg_sane_offset_ptr(struct bpf_verifier_env * env,const struct bpf_reg_state * reg,enum bpf_reg_type type)14097 static bool check_reg_sane_offset_ptr(struct bpf_verifier_env *env,
14098 				      const struct bpf_reg_state *reg,
14099 				      enum bpf_reg_type type)
14100 {
14101 	bool known = tnum_is_const(reg->var_off);
14102 	s64 val = reg->var_off.value;
14103 	s64 smin = reg_smin(reg);
14104 
14105 	if (known && (val >= BPF_MAX_VAR_OFF || val <= -BPF_MAX_VAR_OFF)) {
14106 		verbose(env, "%s pointer offset %lld is not allowed\n",
14107 			reg_type_str(env, type), val);
14108 		return false;
14109 	}
14110 
14111 	if (smin >= BPF_MAX_VAR_OFF || smin <= -BPF_MAX_VAR_OFF) {
14112 		verbose(env, "%s pointer offset %lld is not allowed\n",
14113 			reg_type_str(env, type), smin);
14114 		return false;
14115 	}
14116 
14117 	return true;
14118 }
14119 
14120 enum {
14121 	REASON_BOUNDS	= -1,
14122 	REASON_TYPE	= -2,
14123 	REASON_PATHS	= -3,
14124 	REASON_LIMIT	= -4,
14125 	REASON_STACK	= -5,
14126 };
14127 
retrieve_ptr_limit(const struct bpf_reg_state * ptr_reg,u32 * alu_limit,bool mask_to_left)14128 static int retrieve_ptr_limit(const struct bpf_reg_state *ptr_reg,
14129 			      u32 *alu_limit, bool mask_to_left)
14130 {
14131 	u32 max = 0, ptr_limit = 0;
14132 
14133 	switch (ptr_reg->type) {
14134 	case PTR_TO_STACK:
14135 		/* Offset 0 is out-of-bounds, but acceptable start for the
14136 		 * left direction, see BPF_REG_FP. Also, unknown scalar
14137 		 * offset where we would need to deal with min/max bounds is
14138 		 * currently prohibited for unprivileged.
14139 		 */
14140 		max = MAX_BPF_STACK + mask_to_left;
14141 		ptr_limit = -ptr_reg->var_off.value;
14142 		break;
14143 	case PTR_TO_MAP_VALUE:
14144 		max = ptr_reg->map_ptr->value_size;
14145 		ptr_limit = mask_to_left ? reg_smin(ptr_reg) : reg_umax(ptr_reg);
14146 		break;
14147 	default:
14148 		return REASON_TYPE;
14149 	}
14150 
14151 	if (ptr_limit >= max)
14152 		return REASON_LIMIT;
14153 	*alu_limit = ptr_limit;
14154 	return 0;
14155 }
14156 
can_skip_alu_sanitation(const struct bpf_verifier_env * env,const struct bpf_insn * insn)14157 static bool can_skip_alu_sanitation(const struct bpf_verifier_env *env,
14158 				    const struct bpf_insn *insn)
14159 {
14160 	return env->bypass_spec_v1 ||
14161 		BPF_SRC(insn->code) == BPF_K ||
14162 		cur_aux(env)->nospec;
14163 }
14164 
update_alu_sanitation_state(struct bpf_insn_aux_data * aux,u32 alu_state,u32 alu_limit)14165 static int update_alu_sanitation_state(struct bpf_insn_aux_data *aux,
14166 				       u32 alu_state, u32 alu_limit)
14167 {
14168 	/* If we arrived here from different branches with different
14169 	 * state or limits to sanitize, then this won't work.
14170 	 */
14171 	if (aux->alu_state &&
14172 	    (aux->alu_state != alu_state ||
14173 	     aux->alu_limit != alu_limit))
14174 		return REASON_PATHS;
14175 
14176 	/* Corresponding fixup done in do_misc_fixups(). */
14177 	aux->alu_state = alu_state;
14178 	aux->alu_limit = alu_limit;
14179 	return 0;
14180 }
14181 
sanitize_val_alu(struct bpf_verifier_env * env,struct bpf_insn * insn)14182 static int sanitize_val_alu(struct bpf_verifier_env *env,
14183 			    struct bpf_insn *insn)
14184 {
14185 	struct bpf_insn_aux_data *aux = cur_aux(env);
14186 
14187 	if (can_skip_alu_sanitation(env, insn))
14188 		return 0;
14189 
14190 	return update_alu_sanitation_state(aux, BPF_ALU_NON_POINTER, 0);
14191 }
14192 
sanitize_needed(u8 opcode)14193 static bool sanitize_needed(u8 opcode)
14194 {
14195 	return opcode == BPF_ADD || opcode == BPF_SUB;
14196 }
14197 
14198 struct bpf_sanitize_info {
14199 	struct bpf_insn_aux_data aux;
14200 	bool mask_to_left;
14201 };
14202 
sanitize_speculative_path(struct bpf_verifier_env * env,const struct bpf_insn * insn,u32 next_idx,u32 curr_idx)14203 static int sanitize_speculative_path(struct bpf_verifier_env *env,
14204 				     const struct bpf_insn *insn,
14205 				     u32 next_idx, u32 curr_idx)
14206 {
14207 	struct bpf_verifier_state *branch;
14208 	struct bpf_reg_state *regs;
14209 
14210 	branch = push_stack(env, next_idx, curr_idx, true);
14211 	if (!IS_ERR(branch) && insn) {
14212 		regs = branch->frame[branch->curframe]->regs;
14213 		if (BPF_SRC(insn->code) == BPF_K) {
14214 			mark_reg_unknown(env, regs, insn->dst_reg);
14215 		} else if (BPF_SRC(insn->code) == BPF_X) {
14216 			mark_reg_unknown(env, regs, insn->dst_reg);
14217 			mark_reg_unknown(env, regs, insn->src_reg);
14218 		}
14219 	}
14220 	return PTR_ERR_OR_ZERO(branch);
14221 }
14222 
sanitize_ptr_alu(struct bpf_verifier_env * env,struct bpf_insn * insn,const struct bpf_reg_state * ptr_reg,const struct bpf_reg_state * off_reg,struct bpf_reg_state * dst_reg,struct bpf_sanitize_info * info,const bool commit_window)14223 static int sanitize_ptr_alu(struct bpf_verifier_env *env,
14224 			    struct bpf_insn *insn,
14225 			    const struct bpf_reg_state *ptr_reg,
14226 			    const struct bpf_reg_state *off_reg,
14227 			    struct bpf_reg_state *dst_reg,
14228 			    struct bpf_sanitize_info *info,
14229 			    const bool commit_window)
14230 {
14231 	struct bpf_insn_aux_data *aux = commit_window ? cur_aux(env) : &info->aux;
14232 	struct bpf_verifier_state *vstate = env->cur_state;
14233 	bool off_is_imm = tnum_is_const(off_reg->var_off);
14234 	bool off_is_neg = reg_smin(off_reg) < 0;
14235 	bool ptr_is_dst_reg = ptr_reg == dst_reg;
14236 	u8 opcode = BPF_OP(insn->code);
14237 	u32 alu_state, alu_limit;
14238 	struct bpf_reg_state tmp;
14239 	int err;
14240 
14241 	if (can_skip_alu_sanitation(env, insn))
14242 		return 0;
14243 
14244 	/* We already marked aux for masking from non-speculative
14245 	 * paths, thus we got here in the first place. We only care
14246 	 * to explore bad access from here.
14247 	 */
14248 	if (vstate->speculative)
14249 		goto do_sim;
14250 
14251 	if (!commit_window) {
14252 		if (!tnum_is_const(off_reg->var_off) &&
14253 		    (reg_smin(off_reg) < 0) != (reg_smax(off_reg) < 0))
14254 			return REASON_BOUNDS;
14255 
14256 		info->mask_to_left = (opcode == BPF_ADD &&  off_is_neg) ||
14257 				     (opcode == BPF_SUB && !off_is_neg);
14258 	}
14259 
14260 	err = retrieve_ptr_limit(ptr_reg, &alu_limit, info->mask_to_left);
14261 	if (err < 0)
14262 		return err;
14263 
14264 	if (commit_window) {
14265 		/* In commit phase we narrow the masking window based on
14266 		 * the observed pointer move after the simulated operation.
14267 		 */
14268 		alu_state = info->aux.alu_state;
14269 		alu_limit = abs(info->aux.alu_limit - alu_limit);
14270 	} else {
14271 		alu_state  = off_is_neg ? BPF_ALU_NEG_VALUE : 0;
14272 		alu_state |= off_is_imm ? BPF_ALU_IMMEDIATE : 0;
14273 		alu_state |= ptr_is_dst_reg ?
14274 			     BPF_ALU_SANITIZE_SRC : BPF_ALU_SANITIZE_DST;
14275 
14276 		/* Limit pruning on unknown scalars to enable deep search for
14277 		 * potential masking differences from other program paths.
14278 		 */
14279 		if (!off_is_imm)
14280 			env->explore_alu_limits = true;
14281 	}
14282 
14283 	err = update_alu_sanitation_state(aux, alu_state, alu_limit);
14284 	if (err < 0)
14285 		return err;
14286 do_sim:
14287 	/* If we're in commit phase, we're done here given we already
14288 	 * pushed the truncated dst_reg into the speculative verification
14289 	 * stack.
14290 	 *
14291 	 * Also, when register is a known constant, we rewrite register-based
14292 	 * operation to immediate-based, and thus do not need masking (and as
14293 	 * a consequence, do not need to simulate the zero-truncation either).
14294 	 */
14295 	if (commit_window || off_is_imm)
14296 		return 0;
14297 
14298 	/* Simulate and find potential out-of-bounds access under
14299 	 * speculative execution from truncation as a result of
14300 	 * masking when off was not within expected range. If off
14301 	 * sits in dst, then we temporarily need to move ptr there
14302 	 * to simulate dst (== 0) +/-= ptr. Needed, for example,
14303 	 * for cases where we use K-based arithmetic in one direction
14304 	 * and truncated reg-based in the other in order to explore
14305 	 * bad access.
14306 	 */
14307 	if (!ptr_is_dst_reg) {
14308 		tmp = *dst_reg;
14309 		*dst_reg = *ptr_reg;
14310 	}
14311 	err = sanitize_speculative_path(env, NULL, env->insn_idx + 1, env->insn_idx);
14312 	if (err < 0)
14313 		return REASON_STACK;
14314 	if (!ptr_is_dst_reg)
14315 		*dst_reg = tmp;
14316 	return 0;
14317 }
14318 
sanitize_mark_insn_seen(struct bpf_verifier_env * env)14319 static void sanitize_mark_insn_seen(struct bpf_verifier_env *env)
14320 {
14321 	struct bpf_verifier_state *vstate = env->cur_state;
14322 
14323 	/* If we simulate paths under speculation, we don't update the
14324 	 * insn as 'seen' such that when we verify unreachable paths in
14325 	 * the non-speculative domain, sanitize_dead_code() can still
14326 	 * rewrite/sanitize them.
14327 	 */
14328 	if (!vstate->speculative)
14329 		env->insn_aux_data[env->insn_idx].seen = env->pass_cnt;
14330 }
14331 
sanitize_err(struct bpf_verifier_env * env,const struct bpf_insn * insn,int reason)14332 static int sanitize_err(struct bpf_verifier_env *env, const struct bpf_insn *insn, int reason)
14333 {
14334 	static const char *err = "pointer arithmetic with it prohibited for !root";
14335 	const char *op = BPF_OP(insn->code) == BPF_ADD ? "add" : "sub";
14336 	u32 dst = insn->dst_reg, src = insn->src_reg;
14337 	struct bpf_reg_state *regs = cur_regs(env);
14338 
14339 	switch (reason) {
14340 	case REASON_BOUNDS:
14341 		verbose(env, "R%d has unknown scalar with mixed signed bounds, %s\n",
14342 			regs[src].type == SCALAR_VALUE ? src : dst, err);
14343 		break;
14344 	case REASON_TYPE:
14345 		verbose(env, "R%d has pointer with unsupported alu operation, %s\n",
14346 			regs[src].type == SCALAR_VALUE ? dst : src, err);
14347 		break;
14348 	case REASON_PATHS:
14349 		verbose(env, "R%d tried to %s from different maps, paths or scalars, %s\n",
14350 			dst, op, err);
14351 		break;
14352 	case REASON_LIMIT:
14353 		verbose(env, "R%d tried to %s beyond pointer bounds, %s\n",
14354 			dst, op, err);
14355 		break;
14356 	case REASON_STACK:
14357 		verbose(env, "R%d could not be pushed for speculative verification, %s\n",
14358 			dst, err);
14359 		return -ENOMEM;
14360 	default:
14361 		verifier_bug(env, "unknown reason (%d)", reason);
14362 		break;
14363 	}
14364 
14365 	return -EACCES;
14366 }
14367 
14368 /* check that stack access falls within stack limits and that 'reg' doesn't
14369  * have a variable offset.
14370  *
14371  * Variable offset is prohibited for unprivileged mode for simplicity since it
14372  * requires corresponding support in Spectre masking for stack ALU.  See also
14373  * retrieve_ptr_limit().
14374  */
check_stack_access_for_ptr_arithmetic(struct bpf_verifier_env * env,int regno,const struct bpf_reg_state * reg,int off)14375 static int check_stack_access_for_ptr_arithmetic(
14376 				struct bpf_verifier_env *env,
14377 				int regno,
14378 				const struct bpf_reg_state *reg,
14379 				int off)
14380 {
14381 	if (!tnum_is_const(reg->var_off)) {
14382 		char tn_buf[48];
14383 
14384 		tnum_strn(tn_buf, sizeof(tn_buf), reg->var_off);
14385 		verbose(env, "R%d variable stack access prohibited for !root, var_off=%s off=%d\n",
14386 			regno, tn_buf, off);
14387 		return -EACCES;
14388 	}
14389 
14390 	if (off >= 0 || off < -MAX_BPF_STACK) {
14391 		verbose(env, "R%d stack pointer arithmetic goes out of range, "
14392 			"prohibited for !root; off=%d\n", regno, off);
14393 		return -EACCES;
14394 	}
14395 
14396 	return 0;
14397 }
14398 
sanitize_check_bounds(struct bpf_verifier_env * env,const struct bpf_insn * insn,struct bpf_reg_state * dst_reg)14399 static int sanitize_check_bounds(struct bpf_verifier_env *env,
14400 				 const struct bpf_insn *insn,
14401 				 struct bpf_reg_state *dst_reg)
14402 {
14403 	u32 dst = insn->dst_reg;
14404 
14405 	/* For unprivileged we require that resulting offset must be in bounds
14406 	 * in order to be able to sanitize access later on.
14407 	 */
14408 	if (env->bypass_spec_v1)
14409 		return 0;
14410 
14411 	switch (dst_reg->type) {
14412 	case PTR_TO_STACK:
14413 		if (check_stack_access_for_ptr_arithmetic(env, dst, dst_reg,
14414 							  dst_reg->var_off.value))
14415 			return -EACCES;
14416 		break;
14417 	case PTR_TO_MAP_VALUE:
14418 		if (check_map_access(env, dst_reg, argno_from_reg(dst), 0, 1, false, ACCESS_HELPER)) {
14419 			verbose(env, "R%d pointer arithmetic of map value goes out of range, "
14420 				"prohibited for !root\n", dst);
14421 			return -EACCES;
14422 		}
14423 		break;
14424 	default:
14425 		return -EOPNOTSUPP;
14426 	}
14427 
14428 	return 0;
14429 }
14430 
14431 /* Handles arithmetic on a pointer and a scalar: computes new min/max and var_off.
14432  * Caller should also handle BPF_MOV case separately.
14433  * If we return -EACCES, caller may want to try again treating pointer as a
14434  * scalar.  So we only emit a diagnostic if !env->allow_ptr_leaks.
14435  */
adjust_ptr_min_max_vals(struct bpf_verifier_env * env,struct bpf_insn * insn,u32 ptr_regno,const struct bpf_reg_state * ptr_reg,const struct bpf_reg_state * off_reg)14436 static int adjust_ptr_min_max_vals(struct bpf_verifier_env *env, struct bpf_insn *insn,
14437 				   u32 ptr_regno, const struct bpf_reg_state *ptr_reg,
14438 				   const struct bpf_reg_state *off_reg)
14439 {
14440 	struct bpf_verifier_state *vstate = env->cur_state;
14441 	struct bpf_func_state *state = vstate->frame[vstate->curframe];
14442 	struct bpf_reg_state *regs = state->regs, *dst_reg;
14443 	bool known = tnum_is_const(off_reg->var_off);
14444 	s64 smin_val = reg_smin(off_reg), smax_val = reg_smax(off_reg);
14445 	u64 umin_val = reg_umin(off_reg), umax_val = reg_umax(off_reg);
14446 	struct bpf_sanitize_info info = {};
14447 	u8 opcode = BPF_OP(insn->code);
14448 	u32 dst = insn->dst_reg;
14449 	const char *reason;
14450 	int ret, bounds_ret;
14451 
14452 	dst_reg = &regs[dst];
14453 
14454 	if ((known && (smin_val != smax_val || umin_val != umax_val)) ||
14455 	    smin_val > smax_val || umin_val > umax_val) {
14456 		/* Taint dst register if offset had invalid bounds derived from
14457 		 * e.g. dead branches.
14458 		 */
14459 		__mark_reg_unknown(env, dst_reg);
14460 		return 0;
14461 	}
14462 
14463 	if (BPF_CLASS(insn->code) != BPF_ALU64) {
14464 		/* 32-bit ALU ops on pointers produce (meaningless) scalars */
14465 		if (opcode == BPF_SUB && env->allow_ptr_leaks) {
14466 			__mark_reg_unknown(env, dst_reg);
14467 			return 0;
14468 		}
14469 
14470 		verbose(env,
14471 			"R%d 32-bit pointer arithmetic prohibited\n",
14472 			dst);
14473 		reason = bpf_diag_fmt(
14474 			env, "R%d holds %s. 32-bit ALU operations on pointers discard pointer tracking, so the verifier cannot keep the result as a safe pointer.",
14475 			ptr_regno, bpf_diag_reg_type_plain(env, ptr_reg->type));
14476 		bpf_diag_register_type(
14477 			env, env->insn_idx, ptr_regno, "32-bit pointer arithmetic", reason,
14478 			"Use a 64-bit ALU instruction with an allowed, bounded scalar offset.");
14479 		return -EACCES;
14480 	}
14481 
14482 	if (ptr_reg->type & PTR_MAYBE_NULL) {
14483 		verbose(env, "R%d pointer arithmetic on %s prohibited, null-check it first\n",
14484 			dst, reg_type_str(env, ptr_reg->type));
14485 		reason = bpf_diag_fmt(
14486 			env, "R%d may be NULL (%s). Pointer arithmetic is allowed only after the program proves the pointer is non-NULL on this path.",
14487 			ptr_regno, reg_type_str(env, ptr_reg->type));
14488 		bpf_diag_register_type(
14489 			env, env->insn_idx, ptr_regno, "pointer arithmetic before NULL check", reason,
14490 			"Make sure that a NULL check precedes any arithmetic performed on the pointer.");
14491 		return -EACCES;
14492 	}
14493 
14494 	switch (base_type(ptr_reg->type)) {
14495 	case PTR_TO_CTX:
14496 	case PTR_TO_MAP_VALUE:
14497 	case PTR_TO_MAP_KEY:
14498 	case PTR_TO_STACK:
14499 	case PTR_TO_PACKET_META:
14500 	case PTR_TO_PACKET:
14501 	case PTR_TO_TP_BUFFER:
14502 	case PTR_TO_BTF_ID:
14503 	case PTR_TO_MEM:
14504 	case PTR_TO_BUF:
14505 	case PTR_TO_FUNC:
14506 	case CONST_PTR_TO_DYNPTR:
14507 		break;
14508 	case PTR_TO_FLOW_KEYS:
14509 		if (known)
14510 			break;
14511 		fallthrough;
14512 	case CONST_PTR_TO_MAP:
14513 		/* smin_val represents the known value */
14514 		if (known && smin_val == 0 && opcode == BPF_ADD)
14515 			break;
14516 		fallthrough;
14517 	default:
14518 		verbose(env, "R%d pointer arithmetic on %s prohibited\n",
14519 			dst, reg_type_str(env, ptr_reg->type));
14520 		reason = bpf_diag_fmt(
14521 			env, "R%d holds %s. This pointer kind does not allow offset arithmetic.",
14522 			ptr_regno, bpf_diag_reg_type_plain(env, ptr_reg->type));
14523 		bpf_diag_register_type(
14524 			env, env->insn_idx, ptr_regno, "pointer arithmetic is not allowed", reason,
14525 			"Do not change this pointer's offset; use it only in operations accepted for its kind.");
14526 		return -EACCES;
14527 	}
14528 
14529 	/* For 'scalar += pointer', dst_reg inherits the complete pointer
14530 	 * register state. Individual fields may be adjusted later by pointer
14531 	 * arithmetic. Callers guarantee that below does not overwrite off_reg.
14532 	 */
14533 	if (dst_reg != ptr_reg)
14534 		*dst_reg = *ptr_reg;
14535 
14536 	/*
14537 	 * Accesses to untrusted PTR_TO_MEM are done through probe
14538 	 * instructions, hence no need to track offsets.
14539 	 */
14540 	if (base_type(ptr_reg->type) == PTR_TO_MEM && (ptr_reg->type & PTR_UNTRUSTED))
14541 		return 0;
14542 
14543 	if (!check_reg_sane_offset_scalar(env, off_reg, ptr_reg->type)) {
14544 		reason = bpf_diag_fmt(
14545 			env, "The scalar offset used with R%d is unbounded or outside the verifier's safe pointer-offset range [-%u, %u].",
14546 			ptr_regno, BPF_MAX_VAR_OFF, BPF_MAX_VAR_OFF);
14547 		bpf_diag_register_type(
14548 			env, env->insn_idx, ptr_regno, "pointer offset is not safe", reason,
14549 			"Clamp or bounds-check the scalar offset before applying it to the pointer.");
14550 		return -EINVAL;
14551 	}
14552 	if (!check_reg_sane_offset_ptr(env, ptr_reg, ptr_reg->type)) {
14553 		reason = bpf_diag_fmt(
14554 			env, "R%d already has an offset outside the verifier's safe range [-%u, %u] for %s.",
14555 			ptr_regno, BPF_MAX_VAR_OFF, BPF_MAX_VAR_OFF,
14556 			bpf_diag_reg_type_plain(env, ptr_reg->type));
14557 		bpf_diag_register_type(
14558 			env, env->insn_idx, ptr_regno, "pointer offset is not safe", reason,
14559 			"Keep the base pointer within the verifier's allowed offset range before applying more arithmetic.");
14560 		return -EINVAL;
14561 	}
14562 
14563 	/* pointer types do not carry 32-bit bounds at the moment. */
14564 	__mark_reg32_unbounded(dst_reg);
14565 
14566 	if (sanitize_needed(opcode)) {
14567 		ret = sanitize_ptr_alu(env, insn, ptr_reg, off_reg, dst_reg,
14568 				       &info, false);
14569 		if (ret < 0)
14570 			return sanitize_err(env, insn, ret);
14571 	}
14572 
14573 	switch (opcode) {
14574 	case BPF_ADD:
14575 		/*
14576 		 * dst_reg gets the pointer type and since some positive
14577 		 * integer value was added to the pointer, give it a new 'id'
14578 		 * if it's a PTR_TO_PACKET.
14579 		 * this creates a new 'base' pointer, off_reg (variable) gets
14580 		 * added into the variable offset, and we copy the fixed offset
14581 		 * from ptr_reg.
14582 		 */
14583 		dst_reg->r64 = cnum64_add(ptr_reg->r64, off_reg->r64);
14584 		dst_reg->var_off = tnum_add(ptr_reg->var_off, off_reg->var_off);
14585 		dst_reg->raw = ptr_reg->raw;
14586 		if (reg_is_pkt_pointer(ptr_reg)) {
14587 			if (!known)
14588 				dst_reg->id = ++env->id_gen;
14589 			/*
14590 			 * Clear range for unknown addends since we can't know
14591 			 * where the pkt pointer ended up. Also clear AT_PKT_END /
14592 			 * BEYOND_PKT_END from prior comparison as any pointer
14593 			 * arithmetic invalidates them.
14594 			 */
14595 			if (!known || dst_reg->range < 0)
14596 				memset(&dst_reg->raw, 0, sizeof(dst_reg->raw));
14597 		}
14598 		break;
14599 	case BPF_SUB:
14600 		if (dst_reg != ptr_reg) {
14601 			/* scalar -= pointer.  Creates an unknown scalar */
14602 			verbose(env, "R%d tried to subtract pointer from scalar\n",
14603 				dst);
14604 			reason = bpf_diag_fmt(
14605 				env, "This operation subtracts pointer register R%d from scalar register R%d. "
14606 				"The verifier only tracks pointer-minus-scalar arithmetic for allowed pointer types.",
14607 				ptr_regno, dst);
14608 			bpf_diag_register_type(
14609 				env, env->insn_idx, ptr_regno, "pointer subtracted from scalar", reason,
14610 				"Keep the pointer as the base; only add or subtract bounded scalars when permitted.");
14611 			return -EACCES;
14612 		}
14613 		/* We don't allow subtraction from FP, because (according to
14614 		 * test_verifier.c test "invalid fp arithmetic", JITs might not
14615 		 * be able to deal with it.
14616 		 */
14617 		if (ptr_reg->type == PTR_TO_STACK) {
14618 			verbose(env, "R%d subtraction from stack pointer prohibited\n",
14619 				dst);
14620 			reason = bpf_diag_fmt(
14621 				env, "R%d is a stack pointer. The verifier does not allow BPF_SUB to move stack pointers.",
14622 				ptr_regno);
14623 			bpf_diag_register_type(
14624 				env, env->insn_idx, ptr_regno, "subtraction from stack pointer", reason,
14625 				"Use addition from R10 to form stack addresses within the tracked stack frame.");
14626 			return -EACCES;
14627 		}
14628 		dst_reg->r64 = cnum64_add(ptr_reg->r64, cnum64_negate(off_reg->r64));
14629 		dst_reg->var_off = tnum_sub(ptr_reg->var_off, off_reg->var_off);
14630 		dst_reg->raw = ptr_reg->raw;
14631 		if (reg_is_pkt_pointer(ptr_reg)) {
14632 			if (!known)
14633 				dst_reg->id = ++env->id_gen;
14634 			/*
14635 			 * Clear range if the subtrahend may be negative since
14636 			 * pkt pointer could move past its bounds. A positive
14637 			 * subtrahend moves it backwards keeping positive range
14638 			 * intact. Also clear AT_PKT_END / BEYOND_PKT_END from
14639 			 * prior comparison as arithmetic invalidates them.
14640 			 */
14641 			if ((!known && smin_val < 0) || dst_reg->range < 0)
14642 				memset(&dst_reg->raw, 0, sizeof(dst_reg->raw));
14643 		}
14644 		break;
14645 	case BPF_AND:
14646 	case BPF_OR:
14647 	case BPF_XOR:
14648 		/* bitwise ops on pointers are troublesome, prohibit. */
14649 		verbose(env, "R%d bitwise operator %s on pointer prohibited\n",
14650 			dst, bpf_alu_string[opcode >> 4]);
14651 		reason = bpf_diag_fmt(
14652 			env, "R%d holds %s. Bitwise operator %s would destroy the pointer value the verifier is tracking.",
14653 			ptr_regno, bpf_diag_reg_type_plain(env, ptr_reg->type),
14654 			bpf_alu_string[opcode >> 4]);
14655 		bpf_diag_register_type(
14656 			env, env->insn_idx, ptr_regno, "bitwise operation on pointer", reason,
14657 			"Do bitwise operations on scalar values, not on pointer-valued registers.");
14658 		return -EACCES;
14659 	default:
14660 		/* other operators (e.g. MUL,LSH) produce non-pointer results */
14661 		verbose(env, "R%d pointer arithmetic with %s operator prohibited\n",
14662 			dst, bpf_alu_string[opcode >> 4]);
14663 		reason = bpf_diag_fmt(
14664 			env, "R%d holds %s. Operator %s is not one of the limited pointer arithmetic operations the verifier can track.",
14665 			ptr_regno, bpf_diag_reg_type_plain(env, ptr_reg->type),
14666 			bpf_alu_string[opcode >> 4]);
14667 		bpf_diag_register_type(
14668 			env, env->insn_idx, ptr_regno, "invalid pointer arithmetic operator", reason,
14669 			"Use only verifier-supported addition or subtraction with a bounded scalar offset, or perform this operation on a scalar value.");
14670 		return -EACCES;
14671 	}
14672 
14673 	if (!check_reg_sane_offset_ptr(env, dst_reg, ptr_reg->type)) {
14674 		reason = bpf_diag_fmt(
14675 			env, "After this arithmetic, R%d would be outside the verifier's safe offset range [-%u, %u] for %s.",
14676 			dst, BPF_MAX_VAR_OFF, BPF_MAX_VAR_OFF,
14677 			bpf_diag_reg_type_plain(env, ptr_reg->type));
14678 		bpf_diag_register_type(
14679 			env, env->insn_idx, ptr_regno, "pointer offset is not safe", reason,
14680 			"Tighten the scalar bounds before the arithmetic so the resulting pointer remains within the allowed range.");
14681 		return -EINVAL;
14682 	}
14683 	reg_bounds_sync(dst_reg);
14684 	bounds_ret = sanitize_check_bounds(env, insn, dst_reg);
14685 	if (bounds_ret == -EACCES)
14686 		return bounds_ret;
14687 	if (sanitize_needed(opcode)) {
14688 		ret = sanitize_ptr_alu(env, insn, dst_reg, off_reg, dst_reg,
14689 				       &info, true);
14690 		if (verifier_bug_if(!can_skip_alu_sanitation(env, insn)
14691 				    && !env->cur_state->speculative
14692 				    && bounds_ret
14693 				    && !ret,
14694 				    env, "Pointer type unsupported by sanitize_check_bounds() not rejected by retrieve_ptr_limit() as required")) {
14695 			return -EFAULT;
14696 		}
14697 		if (ret < 0)
14698 			return sanitize_err(env, insn, ret);
14699 	}
14700 
14701 	return 0;
14702 }
14703 
scalar32_min_max_add(struct bpf_reg_state * dst_reg,struct bpf_reg_state * src_reg)14704 static void scalar32_min_max_add(struct bpf_reg_state *dst_reg,
14705 				 struct bpf_reg_state *src_reg)
14706 {
14707 	dst_reg->r32 = cnum32_add(dst_reg->r32, src_reg->r32);
14708 }
14709 
scalar_min_max_add(struct bpf_reg_state * dst_reg,struct bpf_reg_state * src_reg)14710 static void scalar_min_max_add(struct bpf_reg_state *dst_reg,
14711 			       struct bpf_reg_state *src_reg)
14712 {
14713 	dst_reg->r64 = cnum64_add(dst_reg->r64, src_reg->r64);
14714 }
14715 
scalar32_min_max_sub(struct bpf_reg_state * dst_reg,struct bpf_reg_state * src_reg)14716 static void scalar32_min_max_sub(struct bpf_reg_state *dst_reg,
14717 				 struct bpf_reg_state *src_reg)
14718 {
14719 	dst_reg->r32 = cnum32_add(dst_reg->r32, cnum32_negate(src_reg->r32));
14720 }
14721 
scalar_min_max_sub(struct bpf_reg_state * dst_reg,struct bpf_reg_state * src_reg)14722 static void scalar_min_max_sub(struct bpf_reg_state *dst_reg,
14723 			       struct bpf_reg_state *src_reg)
14724 {
14725 	dst_reg->r64 = cnum64_add(dst_reg->r64, cnum64_negate(src_reg->r64));
14726 }
14727 
scalar32_min_max_mul(struct bpf_reg_state * dst_reg,struct bpf_reg_state * src_reg)14728 static void scalar32_min_max_mul(struct bpf_reg_state *dst_reg,
14729 				 struct bpf_reg_state *src_reg)
14730 {
14731 	s32 smin = reg_s32_min(dst_reg);
14732 	s32 smax = reg_s32_max(dst_reg);
14733 	u32 umin = reg_u32_min(dst_reg);
14734 	u32 umax = reg_u32_max(dst_reg);
14735 	s32 tmp_prod[4];
14736 
14737 	if (check_mul_overflow(umax, reg_u32_max(src_reg), &umax) ||
14738 	    check_mul_overflow(umin, reg_u32_min(src_reg), &umin)) {
14739 		/* Overflow possible, we know nothing */
14740 		umin = 0;
14741 		umax = U32_MAX;
14742 	}
14743 	if (check_mul_overflow(smin, reg_s32_min(src_reg), &tmp_prod[0]) ||
14744 	    check_mul_overflow(smin, reg_s32_max(src_reg), &tmp_prod[1]) ||
14745 	    check_mul_overflow(smax, reg_s32_min(src_reg), &tmp_prod[2]) ||
14746 	    check_mul_overflow(smax, reg_s32_max(src_reg), &tmp_prod[3])) {
14747 		/* Overflow possible, we know nothing */
14748 		smin = S32_MIN;
14749 		smax = S32_MAX;
14750 	} else {
14751 		smin = min_array(tmp_prod, 4);
14752 		smax = max_array(tmp_prod, 4);
14753 	}
14754 
14755 	dst_reg->r32 = cnum32_intersect(cnum32_from_urange(umin, umax),
14756 					cnum32_from_srange(smin, smax));
14757 }
14758 
scalar_min_max_mul(struct bpf_reg_state * dst_reg,struct bpf_reg_state * src_reg)14759 static void scalar_min_max_mul(struct bpf_reg_state *dst_reg,
14760 			       struct bpf_reg_state *src_reg)
14761 {
14762 	s64 smin = reg_smin(dst_reg);
14763 	s64 smax = reg_smax(dst_reg);
14764 	u64 umin = reg_umin(dst_reg);
14765 	u64 umax = reg_umax(dst_reg);
14766 	s64 tmp_prod[4];
14767 
14768 	if (check_mul_overflow(umax, reg_umax(src_reg), &umax) ||
14769 	    check_mul_overflow(umin, reg_umin(src_reg), &umin)) {
14770 		/* Overflow possible, we know nothing */
14771 		umin = 0;
14772 		umax = U64_MAX;
14773 	}
14774 	if (check_mul_overflow(smin, reg_smin(src_reg), &tmp_prod[0]) ||
14775 	    check_mul_overflow(smin, reg_smax(src_reg), &tmp_prod[1]) ||
14776 	    check_mul_overflow(smax, reg_smin(src_reg), &tmp_prod[2]) ||
14777 	    check_mul_overflow(smax, reg_smax(src_reg), &tmp_prod[3])) {
14778 		/* Overflow possible, we know nothing */
14779 		smin = S64_MIN;
14780 		smax = S64_MAX;
14781 	} else {
14782 		smin = min_array(tmp_prod, 4);
14783 		smax = max_array(tmp_prod, 4);
14784 	}
14785 
14786 	dst_reg->r64 = cnum64_intersect(cnum64_from_urange(umin, umax),
14787 					cnum64_from_srange(smin, smax));
14788 }
14789 
scalar32_min_max_udiv(struct bpf_reg_state * dst_reg,struct bpf_reg_state * src_reg)14790 static void scalar32_min_max_udiv(struct bpf_reg_state *dst_reg,
14791 				  struct bpf_reg_state *src_reg)
14792 {
14793 	u32 src_val = reg_u32_min(src_reg); /* non-zero, const divisor */
14794 
14795 	reg_set_urange32(dst_reg, reg_u32_min(dst_reg) / src_val,
14796 			 reg_u32_max(dst_reg) / src_val);
14797 
14798 	/* Reset other ranges/tnum to unbounded/unknown. */
14799 	reset_reg64_and_tnum(dst_reg);
14800 }
14801 
scalar_min_max_udiv(struct bpf_reg_state * dst_reg,struct bpf_reg_state * src_reg)14802 static void scalar_min_max_udiv(struct bpf_reg_state *dst_reg,
14803 				struct bpf_reg_state *src_reg)
14804 {
14805 	u64 src_val = reg_umin(src_reg); /* non-zero, const divisor */
14806 
14807 	reg_set_urange64(dst_reg, div64_u64(reg_umin(dst_reg), src_val),
14808 			 div64_u64(reg_umax(dst_reg), src_val));
14809 
14810 	/* Reset other ranges/tnum to unbounded/unknown. */
14811 	reset_reg32_and_tnum(dst_reg);
14812 }
14813 
scalar32_min_max_sdiv(struct bpf_reg_state * dst_reg,struct bpf_reg_state * src_reg)14814 static void scalar32_min_max_sdiv(struct bpf_reg_state *dst_reg,
14815 				  struct bpf_reg_state *src_reg)
14816 {
14817 	s32 smin = reg_s32_min(dst_reg);
14818 	s32 smax = reg_s32_max(dst_reg);
14819 	s32 src_val = reg_s32_min(src_reg); /* non-zero, const divisor */
14820 	s32 res1, res2;
14821 
14822 	/* BPF div specification: S32_MIN / -1 = S32_MIN */
14823 	if (smin == S32_MIN && src_val == -1) {
14824 		/*
14825 		 * If the dividend range contains more than just S32_MIN,
14826 		 * we cannot precisely track the result, so it becomes unbounded.
14827 		 * e.g., [S32_MIN, S32_MIN+10]/(-1),
14828 		 *     = {S32_MIN} U [-(S32_MIN+10), -(S32_MIN+1)]
14829 		 *     = {S32_MIN} U [S32_MAX-9, S32_MAX] = [S32_MIN, S32_MAX]
14830 		 * Otherwise (if dividend is exactly S32_MIN), result remains S32_MIN.
14831 		 */
14832 		if (smax != S32_MIN) {
14833 			smin = S32_MIN;
14834 			smax = S32_MAX;
14835 		}
14836 		goto reset;
14837 	}
14838 
14839 	res1 = smin / src_val;
14840 	res2 = smax / src_val;
14841 	smin = min(res1, res2);
14842 	smax = max(res1, res2);
14843 
14844 reset:
14845 	reg_set_srange32(dst_reg, smin, smax);
14846 	/* Reset other ranges/tnum to unbounded/unknown. */
14847 	reset_reg64_and_tnum(dst_reg);
14848 }
14849 
scalar_min_max_sdiv(struct bpf_reg_state * dst_reg,struct bpf_reg_state * src_reg)14850 static void scalar_min_max_sdiv(struct bpf_reg_state *dst_reg,
14851 				struct bpf_reg_state *src_reg)
14852 {
14853 	s64 smin = reg_smin(dst_reg);
14854 	s64 smax = reg_smax(dst_reg);
14855 	s64 src_val = reg_smin(src_reg); /* non-zero, const divisor */
14856 	s64 res1, res2;
14857 
14858 	/* BPF div specification: S64_MIN / -1 = S64_MIN */
14859 	if (smin == S64_MIN && src_val == -1) {
14860 		/*
14861 		 * If the dividend range contains more than just S64_MIN,
14862 		 * we cannot precisely track the result, so it becomes unbounded.
14863 		 * e.g., [S64_MIN, S64_MIN+10]/(-1),
14864 		 *     = {S64_MIN} U [-(S64_MIN+10), -(S64_MIN+1)]
14865 		 *     = {S64_MIN} U [S64_MAX-9, S64_MAX] = [S64_MIN, S64_MAX]
14866 		 * Otherwise (if dividend is exactly S64_MIN), result remains S64_MIN.
14867 		 */
14868 		if (smax != S64_MIN) {
14869 			smin = S64_MIN;
14870 			smax = S64_MAX;
14871 		}
14872 		goto reset;
14873 	}
14874 
14875 	res1 = div64_s64(smin, src_val);
14876 	res2 = div64_s64(smax, src_val);
14877 	smin = min(res1, res2);
14878 	smax = max(res1, res2);
14879 
14880 reset:
14881 	reg_set_srange64(dst_reg, smin, smax);
14882 	/* Reset other ranges/tnum to unbounded/unknown. */
14883 	reset_reg32_and_tnum(dst_reg);
14884 }
14885 
scalar32_min_max_umod(struct bpf_reg_state * dst_reg,struct bpf_reg_state * src_reg)14886 static void scalar32_min_max_umod(struct bpf_reg_state *dst_reg,
14887 				  struct bpf_reg_state *src_reg)
14888 {
14889 	u32 src_val = reg_u32_min(src_reg); /* non-zero, const divisor */
14890 	u32 res_max = src_val - 1;
14891 
14892 	/*
14893 	 * If dst_umax <= res_max, the result remains unchanged.
14894 	 * e.g., [2, 5] % 10 = [2, 5].
14895 	 */
14896 	if (reg_u32_max(dst_reg) <= res_max)
14897 		return;
14898 
14899 	reg_set_urange32(dst_reg, 0, min(reg_u32_max(dst_reg), res_max));
14900 
14901 	/* Reset other ranges/tnum to unbounded/unknown. */
14902 	reset_reg64_and_tnum(dst_reg);
14903 }
14904 
scalar_min_max_umod(struct bpf_reg_state * dst_reg,struct bpf_reg_state * src_reg)14905 static void scalar_min_max_umod(struct bpf_reg_state *dst_reg,
14906 				struct bpf_reg_state *src_reg)
14907 {
14908 	u64 src_val = reg_umin(src_reg); /* non-zero, const divisor */
14909 	u64 res_max = src_val - 1;
14910 
14911 	/*
14912 	 * If dst_umax <= res_max, the result remains unchanged.
14913 	 * e.g., [2, 5] % 10 = [2, 5].
14914 	 */
14915 	if (reg_umax(dst_reg) <= res_max)
14916 		return;
14917 
14918 	reg_set_urange64(dst_reg, 0, min(reg_umax(dst_reg), res_max));
14919 
14920 	/* Reset other ranges/tnum to unbounded/unknown. */
14921 	reset_reg32_and_tnum(dst_reg);
14922 }
14923 
scalar32_min_max_smod(struct bpf_reg_state * dst_reg,struct bpf_reg_state * src_reg)14924 static void scalar32_min_max_smod(struct bpf_reg_state *dst_reg,
14925 				  struct bpf_reg_state *src_reg)
14926 {
14927 	s32 src_val = reg_s32_min(src_reg); /* non-zero, const divisor */
14928 
14929 	/*
14930 	 * Safe absolute value calculation:
14931 	 * If src_val == S32_MIN (-2147483648), src_abs becomes 2147483648.
14932 	 * Here use unsigned integer to avoid overflow.
14933 	 */
14934 	u32 src_abs = (src_val > 0) ? (u32)src_val : -(u32)src_val;
14935 
14936 	/*
14937 	 * Calculate the maximum possible absolute value of the result.
14938 	 * Even if src_abs is 2147483648 (S32_MIN), subtracting 1 gives
14939 	 * 2147483647 (S32_MAX), which fits perfectly in s32.
14940 	 */
14941 	s32 res_max_abs = src_abs - 1;
14942 
14943 	/*
14944 	 * If the dividend is already within the result range,
14945 	 * the result remains unchanged. e.g., [-2, 5] % 10 = [-2, 5].
14946 	 */
14947 	if (reg_s32_min(dst_reg) >= -res_max_abs && reg_s32_max(dst_reg) <= res_max_abs)
14948 		return;
14949 
14950 	/* General case: result has the same sign as the dividend. */
14951 	if (reg_s32_min(dst_reg) >= 0) {
14952 		reg_set_srange32(dst_reg, 0, min(reg_s32_max(dst_reg), res_max_abs));
14953 	} else if (reg_s32_max(dst_reg) <= 0) {
14954 		reg_set_srange32(dst_reg, max(reg_s32_min(dst_reg), -res_max_abs), 0);
14955 	} else {
14956 		reg_set_srange32(dst_reg, -res_max_abs, res_max_abs);
14957 	}
14958 
14959 	/* Reset other ranges/tnum to unbounded/unknown. */
14960 	reset_reg64_and_tnum(dst_reg);
14961 }
14962 
scalar_min_max_smod(struct bpf_reg_state * dst_reg,struct bpf_reg_state * src_reg)14963 static void scalar_min_max_smod(struct bpf_reg_state *dst_reg,
14964 				struct bpf_reg_state *src_reg)
14965 {
14966 	s64 src_val = reg_smin(src_reg); /* non-zero, const divisor */
14967 
14968 	/*
14969 	 * Safe absolute value calculation:
14970 	 * If src_val == S64_MIN (-2^63), src_abs becomes 2^63.
14971 	 * Here use unsigned integer to avoid overflow.
14972 	 */
14973 	u64 src_abs = (src_val > 0) ? (u64)src_val : -(u64)src_val;
14974 
14975 	/*
14976 	 * Calculate the maximum possible absolute value of the result.
14977 	 * Even if src_abs is 2^63 (S64_MIN), subtracting 1 gives
14978 	 * 2^63 - 1 (S64_MAX), which fits perfectly in s64.
14979 	 */
14980 	s64 res_max_abs = src_abs - 1;
14981 
14982 	/*
14983 	 * If the dividend is already within the result range,
14984 	 * the result remains unchanged. e.g., [-2, 5] % 10 = [-2, 5].
14985 	 */
14986 	if (reg_smin(dst_reg) >= -res_max_abs && reg_smax(dst_reg) <= res_max_abs)
14987 		return;
14988 
14989 	/* General case: result has the same sign as the dividend. */
14990 	if (reg_smin(dst_reg) >= 0) {
14991 		reg_set_srange64(dst_reg, 0, min(reg_smax(dst_reg), res_max_abs));
14992 	} else if (reg_smax(dst_reg) <= 0) {
14993 		reg_set_srange64(dst_reg, max(reg_smin(dst_reg), -res_max_abs), 0);
14994 	} else {
14995 		reg_set_srange64(dst_reg, -res_max_abs, res_max_abs);
14996 	}
14997 
14998 	/* Reset other ranges/tnum to unbounded/unknown. */
14999 	reset_reg32_and_tnum(dst_reg);
15000 }
15001 
scalar32_min_max_and(struct bpf_reg_state * dst_reg,struct bpf_reg_state * src_reg)15002 static void scalar32_min_max_and(struct bpf_reg_state *dst_reg,
15003 				 struct bpf_reg_state *src_reg)
15004 {
15005 	bool src_known = tnum_subreg_is_const(src_reg->var_off);
15006 	bool dst_known = tnum_subreg_is_const(dst_reg->var_off);
15007 	struct tnum var32_off = tnum_subreg(dst_reg->var_off);
15008 	u32 umax_val = reg_u32_max(src_reg);
15009 
15010 	if (src_known && dst_known) {
15011 		__mark_reg32_known(dst_reg, var32_off.value);
15012 		return;
15013 	}
15014 
15015 	/* We get our minimum from the var_off, since that's inherently
15016 	 * bitwise.  Our maximum is the minimum of the operands' maxima.
15017 	 */
15018 	reg_set_urange32(dst_reg,
15019 			 var32_off.value,
15020 			 min(reg_u32_max(dst_reg), umax_val));
15021 }
15022 
scalar_min_max_and(struct bpf_reg_state * dst_reg,struct bpf_reg_state * src_reg)15023 static void scalar_min_max_and(struct bpf_reg_state *dst_reg,
15024 			       struct bpf_reg_state *src_reg)
15025 {
15026 	bool src_known = tnum_is_const(src_reg->var_off);
15027 	bool dst_known = tnum_is_const(dst_reg->var_off);
15028 	u64 umax_val = reg_umax(src_reg);
15029 
15030 	if (src_known && dst_known) {
15031 		__mark_reg_known(dst_reg, dst_reg->var_off.value);
15032 		return;
15033 	}
15034 
15035 	/* We get our minimum from the var_off, since that's inherently
15036 	 * bitwise.  Our maximum is the minimum of the operands' maxima.
15037 	 */
15038 	reg_set_urange64(dst_reg,
15039 			 dst_reg->var_off.value,
15040 			 min(reg_umax(dst_reg), umax_val));
15041 
15042 	/* We may learn something more from the var_off */
15043 	__update_reg_bounds(dst_reg);
15044 }
15045 
scalar32_min_max_or(struct bpf_reg_state * dst_reg,struct bpf_reg_state * src_reg)15046 static void scalar32_min_max_or(struct bpf_reg_state *dst_reg,
15047 				struct bpf_reg_state *src_reg)
15048 {
15049 	bool src_known = tnum_subreg_is_const(src_reg->var_off);
15050 	bool dst_known = tnum_subreg_is_const(dst_reg->var_off);
15051 	struct tnum var32_off = tnum_subreg(dst_reg->var_off);
15052 	u32 umin_val = reg_u32_min(src_reg);
15053 
15054 	if (src_known && dst_known) {
15055 		__mark_reg32_known(dst_reg, var32_off.value);
15056 		return;
15057 	}
15058 
15059 	/* We get our maximum from the var_off, and our minimum is the
15060 	 * maximum of the operands' minima
15061 	 */
15062 	reg_set_urange32(dst_reg,
15063 			 max(reg_u32_min(dst_reg), umin_val),
15064 			 var32_off.value | var32_off.mask);
15065 }
15066 
scalar_min_max_or(struct bpf_reg_state * dst_reg,struct bpf_reg_state * src_reg)15067 static void scalar_min_max_or(struct bpf_reg_state *dst_reg,
15068 			      struct bpf_reg_state *src_reg)
15069 {
15070 	bool src_known = tnum_is_const(src_reg->var_off);
15071 	bool dst_known = tnum_is_const(dst_reg->var_off);
15072 	u64 umin_val = reg_umin(src_reg);
15073 
15074 	if (src_known && dst_known) {
15075 		__mark_reg_known(dst_reg, dst_reg->var_off.value);
15076 		return;
15077 	}
15078 
15079 	/* We get our maximum from the var_off, and our minimum is the
15080 	 * maximum of the operands' minima
15081 	 */
15082 	reg_set_urange64(dst_reg,
15083 			 max(reg_umin(dst_reg), umin_val),
15084 			 dst_reg->var_off.value | dst_reg->var_off.mask);
15085 
15086 	/* We may learn something more from the var_off */
15087 	__update_reg_bounds(dst_reg);
15088 }
15089 
scalar32_min_max_xor(struct bpf_reg_state * dst_reg,struct bpf_reg_state * src_reg)15090 static void scalar32_min_max_xor(struct bpf_reg_state *dst_reg,
15091 				 struct bpf_reg_state *src_reg)
15092 {
15093 	bool src_known = tnum_subreg_is_const(src_reg->var_off);
15094 	bool dst_known = tnum_subreg_is_const(dst_reg->var_off);
15095 	struct tnum var32_off = tnum_subreg(dst_reg->var_off);
15096 
15097 	if (src_known && dst_known) {
15098 		__mark_reg32_known(dst_reg, var32_off.value);
15099 		return;
15100 	}
15101 
15102 	/* We get both minimum and maximum from the var32_off. */
15103 	reg_set_urange32(dst_reg, var32_off.value, var32_off.value | var32_off.mask);
15104 }
15105 
scalar_min_max_xor(struct bpf_reg_state * dst_reg,struct bpf_reg_state * src_reg)15106 static void scalar_min_max_xor(struct bpf_reg_state *dst_reg,
15107 			       struct bpf_reg_state *src_reg)
15108 {
15109 	bool src_known = tnum_is_const(src_reg->var_off);
15110 	bool dst_known = tnum_is_const(dst_reg->var_off);
15111 
15112 	if (src_known && dst_known) {
15113 		/* dst_reg->var_off.value has been updated earlier */
15114 		__mark_reg_known(dst_reg, dst_reg->var_off.value);
15115 		return;
15116 	}
15117 
15118 	/* We get both minimum and maximum from the var_off. */
15119 	reg_set_urange64(dst_reg,
15120 			 dst_reg->var_off.value,
15121 			 dst_reg->var_off.value | dst_reg->var_off.mask);
15122 }
15123 
__scalar32_min_max_lsh(struct bpf_reg_state * dst_reg,u64 umin_val,u64 umax_val)15124 static void __scalar32_min_max_lsh(struct bpf_reg_state *dst_reg,
15125 				   u64 umin_val, u64 umax_val)
15126 {
15127 	/* If we might shift our top bit out, then we know nothing */
15128 	if (umax_val > 31 || reg_u32_max(dst_reg) > 1ULL << (31 - umax_val))
15129 		reg_set_urange32(dst_reg, 0, U32_MAX);
15130 	else
15131 		/* We lose all sign bit information (except what we can pick
15132 		 * up from var_off)
15133 		 */
15134 		reg_set_urange32(dst_reg, reg_u32_min(dst_reg) << umin_val,
15135 				 reg_u32_max(dst_reg) << umax_val);
15136 }
15137 
scalar32_min_max_lsh(struct bpf_reg_state * dst_reg,struct bpf_reg_state * src_reg)15138 static void scalar32_min_max_lsh(struct bpf_reg_state *dst_reg,
15139 				 struct bpf_reg_state *src_reg)
15140 {
15141 	u32 umax_val = reg_u32_max(src_reg);
15142 	u32 umin_val = reg_u32_min(src_reg);
15143 	/* u32 alu operation will zext upper bits */
15144 	struct tnum subreg = tnum_subreg(dst_reg->var_off);
15145 
15146 	__scalar32_min_max_lsh(dst_reg, umin_val, umax_val);
15147 	dst_reg->var_off = tnum_subreg(tnum_lshift(subreg, umin_val));
15148 	/* Not required but being careful mark reg64 bounds as unknown so
15149 	 * that we are forced to pick them up from tnum and zext later and
15150 	 * if some path skips this step we are still safe.
15151 	 */
15152 	__mark_reg64_unbounded(dst_reg);
15153 	__update_reg32_bounds(dst_reg);
15154 }
15155 
__scalar64_min_max_lsh(struct bpf_reg_state * dst_reg,u64 umin_val,u64 umax_val)15156 static void __scalar64_min_max_lsh(struct bpf_reg_state *dst_reg,
15157 				   u64 umin_val, u64 umax_val)
15158 {
15159 	struct cnum64 u, s;
15160 
15161 	/* Special case <<32 because it is a common compiler pattern to sign
15162 	 * extend subreg by doing <<32 s>>32. smin/smax assignments are correct
15163 	 * because s32 bounds don't flip sign when shifting to the left by
15164 	 * 32bits.
15165 	 */
15166 	if (umin_val == 32 && umax_val == 32)
15167 		s = cnum64_from_srange((s64)reg_s32_min(dst_reg) << 32,
15168 				       (s64)reg_s32_max(dst_reg) << 32);
15169 	else
15170 		s = CNUM64_UNBOUNDED;
15171 
15172 	/* If we might shift our top bit out, then we know nothing */
15173 	if (reg_umax(dst_reg) > 1ULL << (63 - umax_val))
15174 		u = CNUM64_UNBOUNDED;
15175 	else
15176 		u = cnum64_from_urange(reg_umin(dst_reg) << umin_val,
15177 				       reg_umax(dst_reg) << umax_val);
15178 
15179 	dst_reg->r64 = cnum64_intersect(u, s);
15180 }
15181 
scalar_min_max_lsh(struct bpf_reg_state * dst_reg,struct bpf_reg_state * src_reg)15182 static void scalar_min_max_lsh(struct bpf_reg_state *dst_reg,
15183 			       struct bpf_reg_state *src_reg)
15184 {
15185 	u64 umax_val = reg_umax(src_reg);
15186 	u64 umin_val = reg_umin(src_reg);
15187 
15188 	/* scalar64 calc uses 32bit unshifted bounds so must be called first */
15189 	__scalar64_min_max_lsh(dst_reg, umin_val, umax_val);
15190 	__scalar32_min_max_lsh(dst_reg, umin_val, umax_val);
15191 
15192 	dst_reg->var_off = tnum_lshift(dst_reg->var_off, umin_val);
15193 	/* We may learn something more from the var_off */
15194 	__update_reg_bounds(dst_reg);
15195 }
15196 
scalar32_min_max_rsh(struct bpf_reg_state * dst_reg,struct bpf_reg_state * src_reg)15197 static void scalar32_min_max_rsh(struct bpf_reg_state *dst_reg,
15198 				 struct bpf_reg_state *src_reg)
15199 {
15200 	struct tnum subreg = tnum_subreg(dst_reg->var_off);
15201 	u32 umax_val = reg_u32_max(src_reg);
15202 	u32 umin_val = reg_u32_min(src_reg);
15203 
15204 	/* BPF_RSH is an unsigned shift.  If the value in dst_reg might
15205 	 * be negative, then either:
15206 	 * 1) src_reg might be zero, so the sign bit of the result is
15207 	 *    unknown, so we lose our signed bounds
15208 	 * 2) it's known negative, thus the unsigned bounds capture the
15209 	 *    signed bounds
15210 	 * 3) the signed bounds cross zero, so they tell us nothing
15211 	 *    about the result
15212 	 * If the value in dst_reg is known nonnegative, then again the
15213 	 * unsigned bounds capture the signed bounds.
15214 	 * Thus, in all cases it suffices to blow away our signed bounds
15215 	 * and rely on inferring new ones from the unsigned bounds and
15216 	 * var_off of the result.
15217 	 */
15218 
15219 	dst_reg->var_off = tnum_rshift(subreg, umin_val);
15220 	reg_set_urange32(dst_reg, reg_u32_min(dst_reg) >> umax_val,
15221 			 reg_u32_max(dst_reg) >> umin_val);
15222 
15223 	__mark_reg64_unbounded(dst_reg);
15224 	__update_reg32_bounds(dst_reg);
15225 }
15226 
scalar_min_max_rsh(struct bpf_reg_state * dst_reg,struct bpf_reg_state * src_reg)15227 static void scalar_min_max_rsh(struct bpf_reg_state *dst_reg,
15228 			       struct bpf_reg_state *src_reg)
15229 {
15230 	u64 umax_val = reg_umax(src_reg);
15231 	u64 umin_val = reg_umin(src_reg);
15232 
15233 	/* BPF_RSH is an unsigned shift.  If the value in dst_reg might
15234 	 * be negative, then either:
15235 	 * 1) src_reg might be zero, so the sign bit of the result is
15236 	 *    unknown, so we lose our signed bounds
15237 	 * 2) it's known negative, thus the unsigned bounds capture the
15238 	 *    signed bounds
15239 	 * 3) the signed bounds cross zero, so they tell us nothing
15240 	 *    about the result
15241 	 * If the value in dst_reg is known nonnegative, then again the
15242 	 * unsigned bounds capture the signed bounds.
15243 	 * Thus, in all cases it suffices to blow away our signed bounds
15244 	 * and rely on inferring new ones from the unsigned bounds and
15245 	 * var_off of the result.
15246 	 */
15247 	dst_reg->var_off = tnum_rshift(dst_reg->var_off, umin_val);
15248 	reg_set_urange64(dst_reg, reg_umin(dst_reg) >> umax_val,
15249 			 reg_umax(dst_reg) >> umin_val);
15250 
15251 	/* Its not easy to operate on alu32 bounds here because it depends
15252 	 * on bits being shifted in. Take easy way out and mark unbounded
15253 	 * so we can recalculate later from tnum.
15254 	 */
15255 	__mark_reg32_unbounded(dst_reg);
15256 	__update_reg_bounds(dst_reg);
15257 }
15258 
scalar32_min_max_arsh(struct bpf_reg_state * dst_reg,struct bpf_reg_state * src_reg)15259 static void scalar32_min_max_arsh(struct bpf_reg_state *dst_reg,
15260 				  struct bpf_reg_state *src_reg)
15261 {
15262 	u64 umin_val = reg_u32_min(src_reg);
15263 
15264 	/* Upon reaching here, src_known is true and
15265 	 * umax_val is equal to umin_val.
15266 	 * Blow away the dst_reg umin_value/umax_value and rely on
15267 	 * dst_reg var_off to refine the result.
15268 	 */
15269 	reg_set_srange32(dst_reg,
15270 			 (u32)(((s32)reg_s32_min(dst_reg)) >> umin_val),
15271 			 (u32)(((s32)reg_s32_max(dst_reg)) >> umin_val));
15272 
15273 	dst_reg->var_off = tnum_arshift(tnum_subreg(dst_reg->var_off), umin_val, 32);
15274 
15275 	__mark_reg64_unbounded(dst_reg);
15276 	__update_reg32_bounds(dst_reg);
15277 }
15278 
scalar_min_max_arsh(struct bpf_reg_state * dst_reg,struct bpf_reg_state * src_reg)15279 static void scalar_min_max_arsh(struct bpf_reg_state *dst_reg,
15280 				struct bpf_reg_state *src_reg)
15281 {
15282 	u64 umin_val = reg_umin(src_reg);
15283 
15284 	/* Upon reaching here, src_known is true and umax_val is equal
15285 	 * to umin_val.
15286 	 */
15287 	reg_set_srange64(dst_reg, reg_smin(dst_reg) >> umin_val,
15288 			 reg_smax(dst_reg) >> umin_val);
15289 
15290 	dst_reg->var_off = tnum_arshift(dst_reg->var_off, umin_val, 64);
15291 
15292 	/* Its not easy to operate on alu32 bounds here because it depends
15293 	 * on bits being shifted in from upper 32-bits. Take easy way out
15294 	 * and mark unbounded so we can recalculate later from tnum.
15295 	 */
15296 	__mark_reg32_unbounded(dst_reg);
15297 	__update_reg_bounds(dst_reg);
15298 }
15299 
scalar_byte_swap(struct bpf_reg_state * dst_reg,struct bpf_insn * insn)15300 static void scalar_byte_swap(struct bpf_reg_state *dst_reg, struct bpf_insn *insn)
15301 {
15302 	/*
15303 	 * Byte swap operation - update var_off using tnum_bswap.
15304 	 * Three cases:
15305 	 * 1. bswap(16|32|64): opcode=0xd7 (BPF_END | BPF_ALU64 | BPF_TO_LE)
15306 	 *    unconditional swap
15307 	 * 2. to_le(16|32|64): opcode=0xd4 (BPF_END | BPF_ALU | BPF_TO_LE)
15308 	 *    swap on big-endian, truncation or no-op on little-endian
15309 	 * 3. to_be(16|32|64): opcode=0xdc (BPF_END | BPF_ALU | BPF_TO_BE)
15310 	 *    swap on little-endian, truncation or no-op on big-endian
15311 	 */
15312 
15313 	bool alu64 = BPF_CLASS(insn->code) == BPF_ALU64;
15314 	bool to_le = BPF_SRC(insn->code) == BPF_TO_LE;
15315 	bool is_big_endian;
15316 #ifdef CONFIG_CPU_BIG_ENDIAN
15317 	is_big_endian = true;
15318 #else
15319 	is_big_endian = false;
15320 #endif
15321 	/* Apply bswap if alu64 or switch between big-endian and little-endian machines */
15322 	bool need_bswap = alu64 || (to_le == is_big_endian);
15323 
15324 	/*
15325 	 * If the register is mutated, manually reset its scalar ID to break
15326 	 * any existing ties and avoid incorrect bounds propagation.
15327 	 */
15328 	if (need_bswap || insn->imm == 16 || insn->imm == 32)
15329 		clear_scalar_id(dst_reg);
15330 
15331 	if (need_bswap) {
15332 		if (insn->imm == 16)
15333 			dst_reg->var_off = tnum_bswap16(dst_reg->var_off);
15334 		else if (insn->imm == 32)
15335 			dst_reg->var_off = tnum_bswap32(dst_reg->var_off);
15336 		else if (insn->imm == 64)
15337 			dst_reg->var_off = tnum_bswap64(dst_reg->var_off);
15338 		/*
15339 		 * Byteswap scrambles the range, so we must reset bounds.
15340 		 * Bounds will be re-derived from the new tnum later.
15341 		 */
15342 		__mark_reg_unbounded(dst_reg);
15343 	}
15344 	/* For bswap16/32, truncate dst register to match the swapped size */
15345 	if (insn->imm == 16 || insn->imm == 32)
15346 		coerce_reg_to_size(dst_reg, insn->imm / 8);
15347 }
15348 
is_safe_to_compute_dst_reg_range(struct bpf_insn * insn,const struct bpf_reg_state * src_reg)15349 static bool is_safe_to_compute_dst_reg_range(struct bpf_insn *insn,
15350 					     const struct bpf_reg_state *src_reg)
15351 {
15352 	bool src_is_const = false;
15353 	u64 insn_bitness = (BPF_CLASS(insn->code) == BPF_ALU64) ? 64 : 32;
15354 
15355 	if (insn_bitness == 32) {
15356 		if (tnum_subreg_is_const(src_reg->var_off)
15357 		    && reg_s32_min(src_reg) == reg_s32_max(src_reg)
15358 		    && reg_u32_min(src_reg) == reg_u32_max(src_reg))
15359 			src_is_const = true;
15360 	} else {
15361 		if (tnum_is_const(src_reg->var_off)
15362 		    && reg_smin(src_reg) == reg_smax(src_reg)
15363 		    && reg_umin(src_reg) == reg_umax(src_reg))
15364 			src_is_const = true;
15365 	}
15366 
15367 	switch (BPF_OP(insn->code)) {
15368 	case BPF_ADD:
15369 	case BPF_SUB:
15370 	case BPF_NEG:
15371 	case BPF_AND:
15372 	case BPF_XOR:
15373 	case BPF_OR:
15374 	case BPF_MUL:
15375 	case BPF_END:
15376 		return true;
15377 
15378 	/*
15379 	 * Division and modulo operators range is only safe to compute when the
15380 	 * divisor is a constant.
15381 	 */
15382 	case BPF_DIV:
15383 	case BPF_MOD:
15384 		return src_is_const;
15385 
15386 	/* Shift operators range is only computable if shift dimension operand
15387 	 * is a constant. Shifts greater than 31 or 63 are undefined. This
15388 	 * includes shifts by a negative number.
15389 	 */
15390 	case BPF_LSH:
15391 	case BPF_RSH:
15392 	case BPF_ARSH:
15393 		return (src_is_const && reg_umax(src_reg) < insn_bitness);
15394 	default:
15395 		return false;
15396 	}
15397 }
15398 
maybe_fork_scalars(struct bpf_verifier_env * env,struct bpf_insn * insn,struct bpf_reg_state * dst_reg)15399 static int maybe_fork_scalars(struct bpf_verifier_env *env, struct bpf_insn *insn,
15400 			      struct bpf_reg_state *dst_reg)
15401 {
15402 	struct bpf_verifier_state *branch;
15403 	struct bpf_reg_state *regs;
15404 	bool alu32;
15405 
15406 	if (reg_smin(dst_reg) == -1 && reg_smax(dst_reg) == 0)
15407 		alu32 = false;
15408 	else if (reg_s32_min(dst_reg) == -1 && reg_s32_max(dst_reg) == 0)
15409 		alu32 = true;
15410 	else
15411 		return 0;
15412 
15413 	branch = push_stack(env, env->insn_idx, env->insn_idx, false);
15414 	if (IS_ERR(branch))
15415 		return PTR_ERR(branch);
15416 
15417 	regs = branch->frame[branch->curframe]->regs;
15418 	if (alu32) {
15419 		__mark_reg32_known(&regs[insn->dst_reg], 0);
15420 		__mark_reg32_known(dst_reg, -1ull);
15421 	} else {
15422 		__mark_reg_known(&regs[insn->dst_reg], 0);
15423 		__mark_reg_known(dst_reg, -1ull);
15424 	}
15425 	return 0;
15426 }
15427 
15428 /* WARNING: This function does calculations on 64-bit values, but the actual
15429  * execution may occur on 32-bit values. Therefore, things like bitshifts
15430  * need extra checks in the 32-bit case.
15431  */
adjust_scalar_min_max_vals(struct bpf_verifier_env * env,struct bpf_insn * insn,struct bpf_reg_state * dst_reg,struct bpf_reg_state src_reg)15432 static int adjust_scalar_min_max_vals(struct bpf_verifier_env *env,
15433 				      struct bpf_insn *insn,
15434 				      struct bpf_reg_state *dst_reg,
15435 				      struct bpf_reg_state src_reg)
15436 {
15437 	u8 opcode = BPF_OP(insn->code);
15438 	s16 off = insn->off;
15439 	bool alu32 = (BPF_CLASS(insn->code) != BPF_ALU64);
15440 	int ret;
15441 
15442 	if (!is_safe_to_compute_dst_reg_range(insn, &src_reg)) {
15443 		__mark_reg_unknown(env, dst_reg);
15444 		return 0;
15445 	}
15446 
15447 	if (sanitize_needed(opcode)) {
15448 		ret = sanitize_val_alu(env, insn);
15449 		if (ret < 0)
15450 			return sanitize_err(env, insn, ret);
15451 	}
15452 
15453 	/* Calculate sign/unsigned bounds and tnum for alu32 and alu64 bit ops.
15454 	 * There are two classes of instructions: The first class we track both
15455 	 * alu32 and alu64 sign/unsigned bounds independently this provides the
15456 	 * greatest amount of precision when alu operations are mixed with jmp32
15457 	 * operations. These operations are BPF_ADD, BPF_SUB, BPF_MUL, BPF_ADD,
15458 	 * and BPF_OR. This is possible because these ops have fairly easy to
15459 	 * understand and calculate behavior in both 32-bit and 64-bit alu ops.
15460 	 * See alu32 verifier tests for examples. The second class of
15461 	 * operations, BPF_LSH, BPF_RSH, and BPF_ARSH, however are not so easy
15462 	 * with regards to tracking sign/unsigned bounds because the bits may
15463 	 * cross subreg boundaries in the alu64 case. When this happens we mark
15464 	 * the reg unbounded in the subreg bound space and use the resulting
15465 	 * tnum to calculate an approximation of the sign/unsigned bounds.
15466 	 */
15467 	switch (opcode) {
15468 	case BPF_ADD:
15469 		scalar32_min_max_add(dst_reg, &src_reg);
15470 		scalar_min_max_add(dst_reg, &src_reg);
15471 		dst_reg->var_off = tnum_add(dst_reg->var_off, src_reg.var_off);
15472 		break;
15473 	case BPF_SUB:
15474 		scalar32_min_max_sub(dst_reg, &src_reg);
15475 		scalar_min_max_sub(dst_reg, &src_reg);
15476 		dst_reg->var_off = tnum_sub(dst_reg->var_off, src_reg.var_off);
15477 		break;
15478 	case BPF_NEG:
15479 		env->fake_reg[0] = *dst_reg;
15480 		__mark_reg_known(dst_reg, 0);
15481 		scalar32_min_max_sub(dst_reg, &env->fake_reg[0]);
15482 		scalar_min_max_sub(dst_reg, &env->fake_reg[0]);
15483 		dst_reg->var_off = tnum_neg(env->fake_reg[0].var_off);
15484 		break;
15485 	case BPF_MUL:
15486 		dst_reg->var_off = tnum_mul(dst_reg->var_off, src_reg.var_off);
15487 		scalar32_min_max_mul(dst_reg, &src_reg);
15488 		scalar_min_max_mul(dst_reg, &src_reg);
15489 		break;
15490 	case BPF_DIV:
15491 		/* BPF div specification: x / 0 = 0 */
15492 		if ((alu32 && reg_u32_min(&src_reg) == 0) || (!alu32 && reg_umin(&src_reg) == 0)) {
15493 			___mark_reg_known(dst_reg, 0);
15494 			break;
15495 		}
15496 		if (alu32)
15497 			if (off == 1)
15498 				scalar32_min_max_sdiv(dst_reg, &src_reg);
15499 			else
15500 				scalar32_min_max_udiv(dst_reg, &src_reg);
15501 		else
15502 			if (off == 1)
15503 				scalar_min_max_sdiv(dst_reg, &src_reg);
15504 			else
15505 				scalar_min_max_udiv(dst_reg, &src_reg);
15506 		break;
15507 	case BPF_MOD:
15508 		/* BPF mod specification: x % 0 = x */
15509 		if ((alu32 && reg_u32_min(&src_reg) == 0) || (!alu32 && reg_umin(&src_reg) == 0))
15510 			break;
15511 		if (alu32)
15512 			if (off == 1)
15513 				scalar32_min_max_smod(dst_reg, &src_reg);
15514 			else
15515 				scalar32_min_max_umod(dst_reg, &src_reg);
15516 		else
15517 			if (off == 1)
15518 				scalar_min_max_smod(dst_reg, &src_reg);
15519 			else
15520 				scalar_min_max_umod(dst_reg, &src_reg);
15521 		break;
15522 	case BPF_AND:
15523 		if (tnum_is_const(src_reg.var_off)) {
15524 			ret = maybe_fork_scalars(env, insn, dst_reg);
15525 			if (ret)
15526 				return ret;
15527 		}
15528 		dst_reg->var_off = tnum_and(dst_reg->var_off, src_reg.var_off);
15529 		scalar32_min_max_and(dst_reg, &src_reg);
15530 		scalar_min_max_and(dst_reg, &src_reg);
15531 		break;
15532 	case BPF_OR:
15533 		if (tnum_is_const(src_reg.var_off)) {
15534 			ret = maybe_fork_scalars(env, insn, dst_reg);
15535 			if (ret)
15536 				return ret;
15537 		}
15538 		dst_reg->var_off = tnum_or(dst_reg->var_off, src_reg.var_off);
15539 		scalar32_min_max_or(dst_reg, &src_reg);
15540 		scalar_min_max_or(dst_reg, &src_reg);
15541 		break;
15542 	case BPF_XOR:
15543 		dst_reg->var_off = tnum_xor(dst_reg->var_off, src_reg.var_off);
15544 		scalar32_min_max_xor(dst_reg, &src_reg);
15545 		scalar_min_max_xor(dst_reg, &src_reg);
15546 		break;
15547 	case BPF_LSH:
15548 		if (alu32)
15549 			scalar32_min_max_lsh(dst_reg, &src_reg);
15550 		else
15551 			scalar_min_max_lsh(dst_reg, &src_reg);
15552 		break;
15553 	case BPF_RSH:
15554 		if (alu32)
15555 			scalar32_min_max_rsh(dst_reg, &src_reg);
15556 		else
15557 			scalar_min_max_rsh(dst_reg, &src_reg);
15558 		break;
15559 	case BPF_ARSH:
15560 		if (alu32)
15561 			scalar32_min_max_arsh(dst_reg, &src_reg);
15562 		else
15563 			scalar_min_max_arsh(dst_reg, &src_reg);
15564 		break;
15565 	case BPF_END:
15566 		scalar_byte_swap(dst_reg, insn);
15567 		break;
15568 	default:
15569 		break;
15570 	}
15571 
15572 	/*
15573 	 * ALU32 ops are zero extended into 64bit register.
15574 	 *
15575 	 * BPF_END is already handled inside the helper (truncation),
15576 	 * so skip zext here to avoid unexpected zero extension.
15577 	 * e.g., le64: opcode=(BPF_END|BPF_ALU|BPF_TO_LE), imm=0x40
15578 	 * This is a 64bit byte swap operation with alu32==true,
15579 	 * but we should not zero extend the result.
15580 	 */
15581 	if (alu32 && opcode != BPF_END)
15582 		zext_32_to_64(dst_reg);
15583 	reg_bounds_sync(dst_reg);
15584 	return 0;
15585 }
15586 
15587 /* Handles ALU ops other than BPF_END, BPF_NEG and BPF_MOV: computes new min/max
15588  * and var_off.
15589  */
adjust_reg_min_max_vals(struct bpf_verifier_env * env,struct bpf_insn * insn)15590 static int adjust_reg_min_max_vals(struct bpf_verifier_env *env,
15591 				   struct bpf_insn *insn)
15592 {
15593 	struct bpf_verifier_state *vstate = env->cur_state;
15594 	struct bpf_func_state *state = vstate->frame[vstate->curframe];
15595 	struct bpf_reg_state *regs = state->regs, *dst_reg, *src_reg;
15596 	struct bpf_reg_state *ptr_reg = NULL, off_reg = {0};
15597 	bool alu32 = (BPF_CLASS(insn->code) != BPF_ALU64);
15598 	u8 opcode = BPF_OP(insn->code);
15599 	int err;
15600 
15601 	dst_reg = &regs[insn->dst_reg];
15602 	if (BPF_SRC(insn->code) == BPF_X)
15603 		src_reg = &regs[insn->src_reg];
15604 	else
15605 		src_reg = NULL;
15606 
15607 	/* Case where at least one operand is an arena. */
15608 	if (dst_reg->type == PTR_TO_ARENA || (src_reg && src_reg->type == PTR_TO_ARENA)) {
15609 		struct bpf_insn_aux_data *aux = cur_aux(env);
15610 
15611 		if (dst_reg->type != PTR_TO_ARENA)
15612 			*dst_reg = *src_reg;
15613 
15614 		if (BPF_CLASS(insn->code) == BPF_ALU64) {
15615 			/*
15616 			 * 32-bit operations zero upper bits automatically.
15617 			 * 64-bit operations need to be converted to 32.
15618 			 */
15619 			aux->needs_zext = true;
15620 			aux->zext_dst = true;
15621 		}
15622 
15623 		/* Any arithmetic operations are allowed on arena pointers */
15624 		return 0;
15625 	}
15626 
15627 	if (dst_reg->type != SCALAR_VALUE)
15628 		ptr_reg = dst_reg;
15629 
15630 	if (BPF_SRC(insn->code) == BPF_X) {
15631 		if (src_reg->type != SCALAR_VALUE) {
15632 			if (dst_reg->type != SCALAR_VALUE) {
15633 				/* Combining two pointers by any ALU op yields
15634 				 * an arbitrary scalar. Disallow all math except
15635 				 * pointer subtraction
15636 				 */
15637 				if (opcode == BPF_SUB && env->allow_ptr_leaks) {
15638 					mark_reg_unknown(env, regs, insn->dst_reg);
15639 					return 0;
15640 				}
15641 				verbose(env, "R%d pointer %s pointer prohibited\n",
15642 					insn->dst_reg,
15643 					bpf_alu_string[opcode >> 4]);
15644 				return -EACCES;
15645 			} else {
15646 				/* scalar += pointer
15647 				 * This is legal, but we have to reverse our
15648 				 * src/dest handling in computing the range
15649 				 */
15650 				err = mark_chain_precision(env, insn->dst_reg);
15651 				if (err)
15652 					return err;
15653 				off_reg = *dst_reg;
15654 				return adjust_ptr_min_max_vals(env, insn, insn->src_reg, src_reg,
15655 							       &off_reg);
15656 			}
15657 		} else if (ptr_reg) {
15658 			/* pointer += scalar */
15659 			err = mark_chain_precision(env, insn->src_reg);
15660 			if (err)
15661 				return err;
15662 			return adjust_ptr_min_max_vals(env, insn, insn->dst_reg, dst_reg, src_reg);
15663 		} else if (dst_reg->precise) {
15664 			/* if dst_reg is precise, src_reg should be precise as well */
15665 			err = mark_chain_precision(env, insn->src_reg);
15666 			if (err)
15667 				return err;
15668 		}
15669 	} else {
15670 		/* Pretend the src is a reg with a known value, since we only
15671 		 * need to be able to read from this state.
15672 		 */
15673 		off_reg.type = SCALAR_VALUE;
15674 		__mark_reg_known(&off_reg, insn->imm);
15675 		src_reg = &off_reg;
15676 		if (ptr_reg) /* pointer += K */
15677 			return adjust_ptr_min_max_vals(env, insn, insn->dst_reg, ptr_reg, src_reg);
15678 	}
15679 
15680 	/* Got here implies adding two SCALAR_VALUEs */
15681 	if (WARN_ON_ONCE(ptr_reg)) {
15682 		print_verifier_state(env, vstate, vstate->curframe, true);
15683 		verbose(env, "verifier internal error: unexpected ptr_reg\n");
15684 		return -EFAULT;
15685 	}
15686 	if (WARN_ON(!src_reg)) {
15687 		print_verifier_state(env, vstate, vstate->curframe, true);
15688 		verbose(env, "verifier internal error: no src_reg\n");
15689 		return -EFAULT;
15690 	}
15691 	/*
15692 	 * For alu32 linked register tracking, we need to check dst_reg's
15693 	 * umax_value before the ALU operation. After adjust_scalar_min_max_vals(),
15694 	 * alu32 ops will have zero-extended the result, making umax_value <= U32_MAX.
15695 	 */
15696 	u64 dst_umax = reg_umax(dst_reg);
15697 
15698 	err = adjust_scalar_min_max_vals(env, insn, dst_reg, *src_reg);
15699 	if (err)
15700 		return err;
15701 	/*
15702 	 * Compilers can generate the code
15703 	 * r1 = r2
15704 	 * r1 += 0x1
15705 	 * if r2 < 1000 goto ...
15706 	 * use r1 in memory access
15707 	 * So remember constant delta between r2 and r1 and update r1 after
15708 	 * 'if' condition.
15709 	 */
15710 	if (env->bpf_capable &&
15711 	    (BPF_OP(insn->code) == BPF_ADD || BPF_OP(insn->code) == BPF_SUB) &&
15712 	    dst_reg->id && is_reg_const(src_reg, alu32) &&
15713 	    !(BPF_SRC(insn->code) == BPF_X && insn->src_reg == insn->dst_reg)) {
15714 		u64 val = reg_const_value(src_reg, alu32);
15715 		s32 off;
15716 
15717 		if (!alu32 && ((s64)val < S32_MIN || (s64)val > S32_MAX))
15718 			goto clear_id;
15719 
15720 		if (alu32 && (dst_umax > U32_MAX))
15721 			goto clear_id;
15722 
15723 		off = (s32)val;
15724 
15725 		if (BPF_OP(insn->code) == BPF_SUB) {
15726 			/* Negating S32_MIN would overflow */
15727 			if (off == S32_MIN)
15728 				goto clear_id;
15729 			off = -off;
15730 		}
15731 
15732 		if (dst_reg->id & BPF_ADD_CONST) {
15733 			/*
15734 			 * If the register already went through rX += val
15735 			 * we cannot accumulate another val into rx->off.
15736 			 */
15737 clear_id:
15738 			clear_scalar_id(dst_reg);
15739 		} else {
15740 			if (alu32)
15741 				dst_reg->id |= BPF_ADD_CONST32;
15742 			else
15743 				dst_reg->id |= BPF_ADD_CONST64;
15744 			dst_reg->delta = off;
15745 		}
15746 	} else {
15747 		/*
15748 		 * Make sure ID is cleared otherwise dst_reg min/max could be
15749 		 * incorrectly propagated into other registers by sync_linked_regs()
15750 		 */
15751 		clear_scalar_id(dst_reg);
15752 	}
15753 	return 0;
15754 }
15755 
15756 /* check validity of 32-bit and 64-bit arithmetic operations */
check_alu_op(struct bpf_verifier_env * env,struct bpf_insn * insn)15757 static int check_alu_op(struct bpf_verifier_env *env, struct bpf_insn *insn)
15758 {
15759 	struct bpf_reg_state *regs = cur_regs(env);
15760 	u8 opcode = BPF_OP(insn->code);
15761 	int err;
15762 
15763 	bpf_diag_mod_begin(env, &regs[insn->dst_reg], NULL, BPF_DIAG_MOD_WRITE);
15764 
15765 	if (opcode == BPF_END || opcode == BPF_NEG) {
15766 		/* check src operand */
15767 		err = check_reg_arg(env, insn->dst_reg, SRC_OP);
15768 		if (err)
15769 			return err;
15770 
15771 		if (is_pointer_value(env, insn->dst_reg)) {
15772 			verbose(env, "R%d pointer arithmetic prohibited\n",
15773 				insn->dst_reg);
15774 			return -EACCES;
15775 		}
15776 
15777 		/* check dest operand */
15778 		if (regs[insn->dst_reg].type == SCALAR_VALUE) {
15779 			err = check_reg_arg(env, insn->dst_reg, DST_OP_NO_MARK);
15780 			err = err ?: adjust_scalar_min_max_vals(env, insn,
15781 							 &regs[insn->dst_reg],
15782 							 regs[insn->dst_reg]);
15783 		} else {
15784 			err = check_reg_arg(env, insn->dst_reg, DST_OP);
15785 		}
15786 		if (err)
15787 			return err;
15788 
15789 	} else if (opcode == BPF_MOV) {
15790 
15791 		if (BPF_SRC(insn->code) == BPF_X) {
15792 			if (insn->off == BPF_ADDR_SPACE_CAST) {
15793 				if (!env->prog->aux->arena) {
15794 					verbose(env, "addr_space_cast insn can only be used in a program that has an associated arena\n");
15795 					return -EINVAL;
15796 				}
15797 			}
15798 
15799 			/* check src operand */
15800 			err = check_reg_arg(env, insn->src_reg, SRC_OP);
15801 			if (err)
15802 				return err;
15803 		}
15804 
15805 		/* check dest operand, mark as required later */
15806 		err = check_reg_arg(env, insn->dst_reg, DST_OP_NO_MARK);
15807 		if (err)
15808 			return err;
15809 
15810 		if (BPF_SRC(insn->code) == BPF_X) {
15811 			struct bpf_reg_state *src_reg = regs + insn->src_reg;
15812 			struct bpf_reg_state *dst_reg = regs + insn->dst_reg;
15813 
15814 			if (BPF_CLASS(insn->code) == BPF_ALU64) {
15815 				if (insn->imm) {
15816 					/* off == BPF_ADDR_SPACE_CAST */
15817 					mark_reg_unknown(env, regs, insn->dst_reg);
15818 					if (insn->imm == 1) /* cast from as(1) to as(0) */
15819 						dst_reg->type = PTR_TO_ARENA;
15820 				} else if (insn->off == 0) {
15821 					/* case: R1 = R2
15822 					 * copy register state to dest reg
15823 					 */
15824 					assign_scalar_id_before_mov(env, src_reg);
15825 					*dst_reg = *src_reg;
15826 				} else {
15827 					/* case: R1 = (s8, s16 s32)R2 */
15828 					if (is_pointer_value(env, insn->src_reg)) {
15829 						verbose(env,
15830 							"R%d sign-extension part of pointer\n",
15831 							insn->src_reg);
15832 						return -EACCES;
15833 					} else if (src_reg->type == SCALAR_VALUE) {
15834 						bool no_sext;
15835 
15836 						no_sext = reg_umax(src_reg) < (1ULL << (insn->off - 1));
15837 						if (no_sext)
15838 							assign_scalar_id_before_mov(env, src_reg);
15839 						*dst_reg = *src_reg;
15840 						if (!no_sext)
15841 							clear_scalar_id(dst_reg);
15842 						coerce_reg_to_size_sx(dst_reg, insn->off >> 3);
15843 					} else {
15844 						mark_reg_unknown(env, regs, insn->dst_reg);
15845 					}
15846 				}
15847 			} else {
15848 				/* R1 = (u32) R2 */
15849 				if (is_pointer_value(env, insn->src_reg)) {
15850 					verbose(env,
15851 						"R%d partial copy of pointer\n",
15852 						insn->src_reg);
15853 					return -EACCES;
15854 				} else if (src_reg->type == SCALAR_VALUE) {
15855 					if (insn->off == 0) {
15856 						bool is_src_reg_u32 = get_reg_width(src_reg) <= 32;
15857 
15858 						if (is_src_reg_u32)
15859 							assign_scalar_id_before_mov(env, src_reg);
15860 						*dst_reg = *src_reg;
15861 						/* Make sure ID is cleared if src_reg is not in u32
15862 						 * range otherwise dst_reg min/max could be incorrectly
15863 						 * propagated into src_reg by sync_linked_regs()
15864 						 */
15865 						if (!is_src_reg_u32)
15866 							clear_scalar_id(dst_reg);
15867 					} else {
15868 						/* case: W1 = (s8, s16)W2 */
15869 						bool no_sext = reg_umax(src_reg) < (1ULL << (insn->off - 1));
15870 
15871 						if (no_sext)
15872 							assign_scalar_id_before_mov(env, src_reg);
15873 						*dst_reg = *src_reg;
15874 						if (!no_sext)
15875 							clear_scalar_id(dst_reg);
15876 						coerce_subreg_to_size_sx(dst_reg, insn->off >> 3);
15877 					}
15878 				} else {
15879 					mark_reg_unknown(env, regs,
15880 							 insn->dst_reg);
15881 				}
15882 				zext_32_to_64(dst_reg);
15883 				reg_bounds_sync(dst_reg);
15884 			}
15885 		} else {
15886 			/* case: R = imm
15887 			 * remember the value we stored into this reg
15888 			 */
15889 			/* clear any state __mark_reg_known doesn't set */
15890 			mark_reg_unknown(env, regs, insn->dst_reg);
15891 			regs[insn->dst_reg].type = SCALAR_VALUE;
15892 			if (BPF_CLASS(insn->code) == BPF_ALU64) {
15893 				__mark_reg_known(regs + insn->dst_reg,
15894 						 insn->imm);
15895 			} else {
15896 				__mark_reg_known(regs + insn->dst_reg,
15897 						 (u32)insn->imm);
15898 			}
15899 		}
15900 
15901 	} else {	/* all other ALU ops: and, sub, xor, add, ... */
15902 
15903 		if (BPF_SRC(insn->code) == BPF_X) {
15904 			/* check src1 operand */
15905 			err = check_reg_arg(env, insn->src_reg, SRC_OP);
15906 			if (err)
15907 				return err;
15908 		}
15909 
15910 		/* check src2 operand */
15911 		err = check_reg_arg(env, insn->dst_reg, SRC_OP);
15912 		if (err)
15913 			return err;
15914 
15915 		if ((opcode == BPF_MOD || opcode == BPF_DIV) &&
15916 		    BPF_SRC(insn->code) == BPF_K && insn->imm == 0) {
15917 			verbose(env, "div by zero\n");
15918 			return -EINVAL;
15919 		}
15920 
15921 		if ((opcode == BPF_LSH || opcode == BPF_RSH ||
15922 		     opcode == BPF_ARSH) && BPF_SRC(insn->code) == BPF_K) {
15923 			int size = BPF_CLASS(insn->code) == BPF_ALU64 ? 64 : 32;
15924 
15925 			if (insn->imm < 0 || insn->imm >= size) {
15926 				verbose(env, "invalid shift %d\n", insn->imm);
15927 				return -EINVAL;
15928 			}
15929 		}
15930 
15931 		/* check dest operand */
15932 		err = check_reg_arg(env, insn->dst_reg, DST_OP_NO_MARK);
15933 		err = err ?: adjust_reg_min_max_vals(env, insn);
15934 		if (err)
15935 			return err;
15936 	}
15937 
15938 	err = reg_bounds_sanity_check(env, &regs[insn->dst_reg], "alu");
15939 	if (err)
15940 		return err;
15941 
15942 	bpf_diag_mod_end(env);
15943 	return 0;
15944 }
15945 
find_good_pkt_pointers(struct bpf_verifier_state * vstate,struct bpf_reg_state * dst_reg,enum bpf_reg_type type,bool range_right_open)15946 static void find_good_pkt_pointers(struct bpf_verifier_state *vstate,
15947 				   struct bpf_reg_state *dst_reg,
15948 				   enum bpf_reg_type type,
15949 				   bool range_right_open)
15950 {
15951 	struct bpf_func_state *state;
15952 	struct bpf_reg_state *reg;
15953 	int new_range;
15954 
15955 	if (reg_umax(dst_reg) == 0 && range_right_open)
15956 		/* This doesn't give us any range */
15957 		return;
15958 
15959 	if (reg_umax(dst_reg) > MAX_PACKET_OFF)
15960 		/* Risk of overflow.  For instance, ptr + (1<<63) may be less
15961 		 * than pkt_end, but that's because it's also less than pkt.
15962 		 */
15963 		return;
15964 
15965 	new_range = reg_umax(dst_reg);
15966 	if (range_right_open)
15967 		new_range++;
15968 
15969 	/* Examples for register markings:
15970 	 *
15971 	 * pkt_data in dst register:
15972 	 *
15973 	 *   r2 = r3;
15974 	 *   r2 += 8;
15975 	 *   if (r2 > pkt_end) goto <handle exception>
15976 	 *   <access okay>
15977 	 *
15978 	 *   r2 = r3;
15979 	 *   r2 += 8;
15980 	 *   if (r2 < pkt_end) goto <access okay>
15981 	 *   <handle exception>
15982 	 *
15983 	 *   Where:
15984 	 *     r2 == dst_reg, pkt_end == src_reg
15985 	 *     r2=pkt(id=n,off=8,r=0)
15986 	 *     r3=pkt(id=n,off=0,r=0)
15987 	 *
15988 	 * pkt_data in src register:
15989 	 *
15990 	 *   r2 = r3;
15991 	 *   r2 += 8;
15992 	 *   if (pkt_end >= r2) goto <access okay>
15993 	 *   <handle exception>
15994 	 *
15995 	 *   r2 = r3;
15996 	 *   r2 += 8;
15997 	 *   if (pkt_end <= r2) goto <handle exception>
15998 	 *   <access okay>
15999 	 *
16000 	 *   Where:
16001 	 *     pkt_end == dst_reg, r2 == src_reg
16002 	 *     r2=pkt(id=n,off=8,r=0)
16003 	 *     r3=pkt(id=n,off=0,r=0)
16004 	 *
16005 	 * Find register r3 and mark its range as r3=pkt(id=n,off=0,r=8)
16006 	 * or r3=pkt(id=n,off=0,r=8-1), so that range of bytes [r3, r3 + 8)
16007 	 * and [r3, r3 + 8-1) respectively is safe to access depending on
16008 	 * the check.
16009 	 */
16010 
16011 	/* If our ids match, then we must have the same max_value.  And we
16012 	 * don't care about the other reg's fixed offset, since if it's too big
16013 	 * the range won't allow anything.
16014 	 * reg_umax(dst_reg) is known < MAX_PACKET_OFF, therefore it fits in a u16.
16015 	 */
16016 	bpf_for_each_reg_in_vstate(vstate, state, reg, ({
16017 		if (reg->type == type && reg->id == dst_reg->id)
16018 			/* keep the maximum range already checked */
16019 			reg->range = max(reg->range, new_range);
16020 	}));
16021 }
16022 
16023 static void regs_refine_cond_op(struct bpf_reg_state *reg1, struct bpf_reg_state *reg2,
16024 				u8 opcode, bool is_jmp32);
16025 static u8 rev_opcode(u8 opcode);
16026 
16027 /*
16028  * Learn more information about live branches by simulating refinement on both branches.
16029  * regs_refine_cond_op() is sound, so producing ill-formed register bounds for the branch means
16030  * that branch is dead.
16031  */
simulate_both_branches_taken(struct bpf_verifier_env * env,u8 opcode,bool is_jmp32)16032 static int simulate_both_branches_taken(struct bpf_verifier_env *env, u8 opcode, bool is_jmp32)
16033 {
16034 	/* Fallthrough (FALSE) branch */
16035 	regs_refine_cond_op(&env->false_reg1, &env->false_reg2, rev_opcode(opcode), is_jmp32);
16036 	reg_bounds_sync(&env->false_reg1);
16037 	reg_bounds_sync(&env->false_reg2);
16038 	/*
16039 	 * If there is a range bounds violation in *any* of the abstract values in either
16040 	 * reg_states in the FALSE branch (i.e. reg1, reg2), the FALSE branch must be dead. Only
16041 	 * TRUE branch will be taken.
16042 	 */
16043 	if (range_bounds_violation(&env->false_reg1) || range_bounds_violation(&env->false_reg2))
16044 		return 1;
16045 
16046 	/* Jump (TRUE) branch */
16047 	regs_refine_cond_op(&env->true_reg1, &env->true_reg2, opcode, is_jmp32);
16048 	reg_bounds_sync(&env->true_reg1);
16049 	reg_bounds_sync(&env->true_reg2);
16050 	/*
16051 	 * If there is a range bounds violation in *any* of the abstract values in either
16052 	 * reg_states in the TRUE branch (i.e. true_reg1, true_reg2), the TRUE branch must be dead.
16053 	 * Only FALSE branch will be taken.
16054 	 */
16055 	if (range_bounds_violation(&env->true_reg1) || range_bounds_violation(&env->true_reg2))
16056 		return 0;
16057 
16058 	/* Both branches are possible, we can't determine which one will be taken. */
16059 	return -1;
16060 }
16061 
16062 /*
16063  * <reg1> <op> <reg2>, currently assuming reg2 is a constant
16064  */
is_scalar_branch_taken(struct bpf_verifier_env * env,struct bpf_reg_state * reg1,struct bpf_reg_state * reg2,u8 opcode,bool is_jmp32)16065 static int is_scalar_branch_taken(struct bpf_verifier_env *env, struct bpf_reg_state *reg1,
16066 				  struct bpf_reg_state *reg2, u8 opcode, bool is_jmp32)
16067 {
16068 	struct tnum t1 = is_jmp32 ? tnum_subreg(reg1->var_off) : reg1->var_off;
16069 	struct tnum t2 = is_jmp32 ? tnum_subreg(reg2->var_off) : reg2->var_off;
16070 	u64 umin1 = is_jmp32 ? (u64)reg_u32_min(reg1) : reg_umin(reg1);
16071 	u64 umax1 = is_jmp32 ? (u64)reg_u32_max(reg1) : reg_umax(reg1);
16072 	s64 smin1 = is_jmp32 ? (s64)reg_s32_min(reg1) : reg_smin(reg1);
16073 	s64 smax1 = is_jmp32 ? (s64)reg_s32_max(reg1) : reg_smax(reg1);
16074 	u64 umin2 = is_jmp32 ? (u64)reg_u32_min(reg2) : reg_umin(reg2);
16075 	u64 umax2 = is_jmp32 ? (u64)reg_u32_max(reg2) : reg_umax(reg2);
16076 	s64 smin2 = is_jmp32 ? (s64)reg_s32_min(reg2) : reg_smin(reg2);
16077 	s64 smax2 = is_jmp32 ? (s64)reg_s32_max(reg2) : reg_smax(reg2);
16078 
16079 	if (reg1 == reg2) {
16080 		switch (opcode) {
16081 		case BPF_JGE:
16082 		case BPF_JLE:
16083 		case BPF_JSGE:
16084 		case BPF_JSLE:
16085 		case BPF_JEQ:
16086 			return 1;
16087 		case BPF_JGT:
16088 		case BPF_JLT:
16089 		case BPF_JSGT:
16090 		case BPF_JSLT:
16091 		case BPF_JNE:
16092 			return 0;
16093 		case BPF_JSET:
16094 			if (tnum_is_const(t1))
16095 				return t1.value != 0;
16096 			else
16097 				return (smin1 <= 0 && smax1 >= 0) ? -1 : 1;
16098 		default:
16099 			return -1;
16100 		}
16101 	}
16102 
16103 	switch (opcode) {
16104 	case BPF_JEQ:
16105 		/* constants, umin/umax and smin/smax checks would be
16106 		 * redundant in this case because they all should match
16107 		 */
16108 		if (tnum_is_const(t1) && tnum_is_const(t2))
16109 			return t1.value == t2.value;
16110 		if (!tnum_overlap(t1, t2))
16111 			return 0;
16112 		/* non-overlapping ranges */
16113 		if (umin1 > umax2 || umax1 < umin2)
16114 			return 0;
16115 		if (smin1 > smax2 || smax1 < smin2)
16116 			return 0;
16117 		if (!is_jmp32) {
16118 			/* if 64-bit ranges are inconclusive, see if we can
16119 			 * utilize 32-bit subrange knowledge to eliminate
16120 			 * branches that can't be taken a priori
16121 			 */
16122 			if (reg_u32_min(reg1) > reg_u32_max(reg2) ||
16123 			    reg_u32_max(reg1) < reg_u32_min(reg2))
16124 				return 0;
16125 			if (reg_s32_min(reg1) > reg_s32_max(reg2) ||
16126 			    reg_s32_max(reg1) < reg_s32_min(reg2))
16127 				return 0;
16128 		}
16129 		break;
16130 	case BPF_JNE:
16131 		/* constants, umin/umax and smin/smax checks would be
16132 		 * redundant in this case because they all should match
16133 		 */
16134 		if (tnum_is_const(t1) && tnum_is_const(t2))
16135 			return t1.value != t2.value;
16136 		if (!tnum_overlap(t1, t2))
16137 			return 1;
16138 		/* non-overlapping ranges */
16139 		if (umin1 > umax2 || umax1 < umin2)
16140 			return 1;
16141 		if (smin1 > smax2 || smax1 < smin2)
16142 			return 1;
16143 		if (!is_jmp32) {
16144 			/* if 64-bit ranges are inconclusive, see if we can
16145 			 * utilize 32-bit subrange knowledge to eliminate
16146 			 * branches that can't be taken a priori
16147 			 */
16148 			if (reg_u32_min(reg1) > reg_u32_max(reg2) ||
16149 			    reg_u32_max(reg1) < reg_u32_min(reg2))
16150 				return 1;
16151 			if (reg_s32_min(reg1) > reg_s32_max(reg2) ||
16152 			    reg_s32_max(reg1) < reg_s32_min(reg2))
16153 				return 1;
16154 		}
16155 		break;
16156 	case BPF_JSET:
16157 		if (!is_reg_const(reg2, is_jmp32)) {
16158 			swap(reg1, reg2);
16159 			swap(t1, t2);
16160 		}
16161 		if (!is_reg_const(reg2, is_jmp32))
16162 			return -1;
16163 		if ((~t1.mask & t1.value) & t2.value)
16164 			return 1;
16165 		if (!((t1.mask | t1.value) & t2.value))
16166 			return 0;
16167 		break;
16168 	case BPF_JGT:
16169 		if (umin1 > umax2)
16170 			return 1;
16171 		else if (umax1 <= umin2)
16172 			return 0;
16173 		break;
16174 	case BPF_JSGT:
16175 		if (smin1 > smax2)
16176 			return 1;
16177 		else if (smax1 <= smin2)
16178 			return 0;
16179 		break;
16180 	case BPF_JLT:
16181 		if (umax1 < umin2)
16182 			return 1;
16183 		else if (umin1 >= umax2)
16184 			return 0;
16185 		break;
16186 	case BPF_JSLT:
16187 		if (smax1 < smin2)
16188 			return 1;
16189 		else if (smin1 >= smax2)
16190 			return 0;
16191 		break;
16192 	case BPF_JGE:
16193 		if (umin1 >= umax2)
16194 			return 1;
16195 		else if (umax1 < umin2)
16196 			return 0;
16197 		break;
16198 	case BPF_JSGE:
16199 		if (smin1 >= smax2)
16200 			return 1;
16201 		else if (smax1 < smin2)
16202 			return 0;
16203 		break;
16204 	case BPF_JLE:
16205 		if (umax1 <= umin2)
16206 			return 1;
16207 		else if (umin1 > umax2)
16208 			return 0;
16209 		break;
16210 	case BPF_JSLE:
16211 		if (smax1 <= smin2)
16212 			return 1;
16213 		else if (smin1 > smax2)
16214 			return 0;
16215 		break;
16216 	}
16217 
16218 	return simulate_both_branches_taken(env, opcode, is_jmp32);
16219 }
16220 
flip_opcode(u32 opcode)16221 static int flip_opcode(u32 opcode)
16222 {
16223 	/* How can we transform "a <op> b" into "b <op> a"? */
16224 	static const u8 opcode_flip[16] = {
16225 		/* these stay the same */
16226 		[BPF_JEQ  >> 4] = BPF_JEQ,
16227 		[BPF_JNE  >> 4] = BPF_JNE,
16228 		[BPF_JSET >> 4] = BPF_JSET,
16229 		/* these swap "lesser" and "greater" (L and G in the opcodes) */
16230 		[BPF_JGE  >> 4] = BPF_JLE,
16231 		[BPF_JGT  >> 4] = BPF_JLT,
16232 		[BPF_JLE  >> 4] = BPF_JGE,
16233 		[BPF_JLT  >> 4] = BPF_JGT,
16234 		[BPF_JSGE >> 4] = BPF_JSLE,
16235 		[BPF_JSGT >> 4] = BPF_JSLT,
16236 		[BPF_JSLE >> 4] = BPF_JSGE,
16237 		[BPF_JSLT >> 4] = BPF_JSGT
16238 	};
16239 	return opcode_flip[opcode >> 4];
16240 }
16241 
is_pkt_ptr_branch_taken(struct bpf_reg_state * dst_reg,struct bpf_reg_state * src_reg,u8 opcode)16242 static int is_pkt_ptr_branch_taken(struct bpf_reg_state *dst_reg,
16243 				   struct bpf_reg_state *src_reg,
16244 				   u8 opcode)
16245 {
16246 	struct bpf_reg_state *pkt;
16247 
16248 	if (src_reg->type == PTR_TO_PACKET_END) {
16249 		pkt = dst_reg;
16250 	} else if (dst_reg->type == PTR_TO_PACKET_END) {
16251 		pkt = src_reg;
16252 		opcode = flip_opcode(opcode);
16253 	} else {
16254 		return -1;
16255 	}
16256 
16257 	if (pkt->range >= 0)
16258 		return -1;
16259 
16260 	switch (opcode) {
16261 	case BPF_JLE:
16262 		/* pkt <= pkt_end */
16263 		fallthrough;
16264 	case BPF_JGT:
16265 		/* pkt > pkt_end */
16266 		if (pkt->range == BEYOND_PKT_END)
16267 			/* pkt has at last one extra byte beyond pkt_end */
16268 			return opcode == BPF_JGT;
16269 		break;
16270 	case BPF_JLT:
16271 		/* pkt < pkt_end */
16272 		fallthrough;
16273 	case BPF_JGE:
16274 		/* pkt >= pkt_end */
16275 		if (pkt->range == BEYOND_PKT_END || pkt->range == AT_PKT_END)
16276 			return opcode == BPF_JGE;
16277 		break;
16278 	}
16279 	return -1;
16280 }
16281 
16282 /* compute branch direction of the expression "if (<reg1> opcode <reg2>) goto target;"
16283  * and return:
16284  *  1 - branch will be taken and "goto target" will be executed
16285  *  0 - branch will not be taken and fall-through to next insn
16286  * -1 - unknown. Example: "if (reg1 < 5)" is unknown when register value
16287  *      range [0,10]
16288  */
is_branch_taken(struct bpf_verifier_env * env,struct bpf_reg_state * reg1,struct bpf_reg_state * reg2,u8 opcode,bool is_jmp32)16289 static int is_branch_taken(struct bpf_verifier_env *env, struct bpf_reg_state *reg1,
16290 			   struct bpf_reg_state *reg2, u8 opcode, bool is_jmp32)
16291 {
16292 	if (reg_is_pkt_pointer_any(reg1) && reg_is_pkt_pointer_any(reg2) && !is_jmp32)
16293 		return is_pkt_ptr_branch_taken(reg1, reg2, opcode);
16294 
16295 	if (__is_pointer_value(false, reg1) || __is_pointer_value(false, reg2)) {
16296 		u64 val;
16297 
16298 		/* arrange that reg2 is a scalar, and reg1 is a pointer */
16299 		if (!is_reg_const(reg2, is_jmp32)) {
16300 			opcode = flip_opcode(opcode);
16301 			swap(reg1, reg2);
16302 		}
16303 		/* and ensure that reg2 is a constant */
16304 		if (!is_reg_const(reg2, is_jmp32))
16305 			return -1;
16306 
16307 		if (!reg_not_null(env, reg1))
16308 			return -1;
16309 
16310 		/* If pointer is valid tests against zero will fail so we can
16311 		 * use this to direct branch taken.
16312 		 */
16313 		val = reg_const_value(reg2, is_jmp32);
16314 		if (val != 0)
16315 			return -1;
16316 
16317 		switch (opcode) {
16318 		case BPF_JEQ:
16319 			return 0;
16320 		case BPF_JNE:
16321 			return 1;
16322 		default:
16323 			return -1;
16324 		}
16325 	}
16326 
16327 	/* now deal with two scalars, but not necessarily constants */
16328 	return is_scalar_branch_taken(env, reg1, reg2, opcode, is_jmp32);
16329 }
16330 
16331 /* Opcode that corresponds to a *false* branch condition.
16332  * E.g., if r1 < r2, then reverse (false) condition is r1 >= r2
16333  */
rev_opcode(u8 opcode)16334 static u8 rev_opcode(u8 opcode)
16335 {
16336 	switch (opcode) {
16337 	case BPF_JEQ:		return BPF_JNE;
16338 	case BPF_JNE:		return BPF_JEQ;
16339 	/* JSET doesn't have it's reverse opcode in BPF, so add
16340 	 * BPF_X flag to denote the reverse of that operation
16341 	 */
16342 	case BPF_JSET:		return BPF_JSET | BPF_X;
16343 	case BPF_JSET | BPF_X:	return BPF_JSET;
16344 	case BPF_JGE:		return BPF_JLT;
16345 	case BPF_JGT:		return BPF_JLE;
16346 	case BPF_JLE:		return BPF_JGT;
16347 	case BPF_JLT:		return BPF_JGE;
16348 	case BPF_JSGE:		return BPF_JSLT;
16349 	case BPF_JSGT:		return BPF_JSLE;
16350 	case BPF_JSLE:		return BPF_JSGT;
16351 	case BPF_JSLT:		return BPF_JSGE;
16352 	default:		return 0;
16353 	}
16354 }
16355 
16356 /* Refine range knowledge for <reg1> <op> <reg>2 conditional operation. */
regs_refine_cond_op(struct bpf_reg_state * reg1,struct bpf_reg_state * reg2,u8 opcode,bool is_jmp32)16357 static void regs_refine_cond_op(struct bpf_reg_state *reg1, struct bpf_reg_state *reg2,
16358 				u8 opcode, bool is_jmp32)
16359 {
16360 	struct tnum t;
16361 	u64 val;
16362 
16363 	/* In case of GE/GT/SGE/JST, reuse LE/LT/SLE/SLT logic from below */
16364 	switch (opcode) {
16365 	case BPF_JGE:
16366 	case BPF_JGT:
16367 	case BPF_JSGE:
16368 	case BPF_JSGT:
16369 		opcode = flip_opcode(opcode);
16370 		swap(reg1, reg2);
16371 		break;
16372 	default:
16373 		break;
16374 	}
16375 
16376 	switch (opcode) {
16377 	case BPF_JEQ:
16378 		if (is_jmp32) {
16379 			reg1->r32 = cnum32_intersect(reg1->r32, reg2->r32);
16380 			reg2->r32 = reg1->r32;
16381 
16382 			t = tnum_intersect(tnum_subreg(reg1->var_off), tnum_subreg(reg2->var_off));
16383 			reg1->var_off = tnum_with_subreg(reg1->var_off, t);
16384 			reg2->var_off = tnum_with_subreg(reg2->var_off, t);
16385 		} else {
16386 			reg1->r64 = cnum64_intersect(reg1->r64, reg2->r64);
16387 			reg2->r64 = reg1->r64;
16388 
16389 			reg1->var_off = tnum_intersect(reg1->var_off, reg2->var_off);
16390 			reg2->var_off = reg1->var_off;
16391 		}
16392 		break;
16393 	case BPF_JNE:
16394 		if (!is_reg_const(reg2, is_jmp32))
16395 			swap(reg1, reg2);
16396 		if (!is_reg_const(reg2, is_jmp32))
16397 			break;
16398 
16399 		/* try to recompute the bound of reg1 if reg2 is a const and
16400 		 * is exactly the edge of reg1.
16401 		 */
16402 		val = reg_const_value(reg2, is_jmp32);
16403 		if (is_jmp32) {
16404 			/* Complement of the range [val, val] as cnum32. */
16405 			cnum32_intersect_with(&reg1->r32, (struct cnum32){ val + 1, U32_MAX - 1 });
16406 		} else {
16407 			/* Complement of the range [val, val] as cnum64. */
16408 			cnum64_intersect_with(&reg1->r64, (struct cnum64){ val + 1, U64_MAX - 1 });
16409 		}
16410 		break;
16411 	case BPF_JSET:
16412 		if (!is_reg_const(reg2, is_jmp32))
16413 			swap(reg1, reg2);
16414 		if (!is_reg_const(reg2, is_jmp32))
16415 			break;
16416 		val = reg_const_value(reg2, is_jmp32);
16417 		/* BPF_JSET (i.e., TRUE branch, *not* BPF_JSET | BPF_X)
16418 		 * requires single bit to learn something useful. E.g., if we
16419 		 * know that `r1 & 0x3` is true, then which bits (0, 1, or both)
16420 		 * are actually set? We can learn something definite only if
16421 		 * it's a single-bit value to begin with.
16422 		 *
16423 		 * BPF_JSET | BPF_X (i.e., negation of BPF_JSET) doesn't have
16424 		 * this restriction. I.e., !(r1 & 0x3) means neither bit 0 nor
16425 		 * bit 1 is set, which we can readily use in adjustments.
16426 		 */
16427 		if (!is_power_of_2(val))
16428 			break;
16429 		if (is_jmp32) {
16430 			t = tnum_or(tnum_subreg(reg1->var_off), tnum_const(val));
16431 			reg1->var_off = tnum_with_subreg(reg1->var_off, t);
16432 		} else {
16433 			reg1->var_off = tnum_or(reg1->var_off, tnum_const(val));
16434 		}
16435 		break;
16436 	case BPF_JSET | BPF_X: /* reverse of BPF_JSET, see rev_opcode() */
16437 		if (!is_reg_const(reg2, is_jmp32))
16438 			swap(reg1, reg2);
16439 		if (!is_reg_const(reg2, is_jmp32))
16440 			break;
16441 		val = reg_const_value(reg2, is_jmp32);
16442 		/* Forget the ranges before narrowing tnums, to avoid invariant
16443 		 * violations if we're on a dead branch.
16444 		 */
16445 		__mark_reg_unbounded(reg1);
16446 		if (is_jmp32) {
16447 			t = tnum_and(tnum_subreg(reg1->var_off), tnum_const(~val));
16448 			reg1->var_off = tnum_with_subreg(reg1->var_off, t);
16449 		} else {
16450 			reg1->var_off = tnum_and(reg1->var_off, tnum_const(~val));
16451 		}
16452 		break;
16453 	case BPF_JLE:
16454 		if (is_jmp32) {
16455 			cnum32_intersect_with_urange(&reg1->r32, 0, reg_u32_max(reg2));
16456 			cnum32_intersect_with_urange(&reg2->r32, reg_u32_min(reg1), U32_MAX);
16457 		} else {
16458 			cnum64_intersect_with_urange(&reg1->r64, 0, reg_umax(reg2));
16459 			cnum64_intersect_with_urange(&reg2->r64, reg_umin(reg1), U64_MAX);
16460 		}
16461 		break;
16462 	case BPF_JLT:
16463 		if (is_jmp32) {
16464 			cnum32_intersect_with_urange(&reg1->r32, 0, reg_u32_max(reg2) - 1);
16465 			cnum32_intersect_with_urange(&reg2->r32, reg_u32_min(reg1) + 1, U32_MAX);
16466 		} else {
16467 			cnum64_intersect_with_urange(&reg1->r64, 0, reg_umax(reg2) - 1);
16468 			cnum64_intersect_with_urange(&reg2->r64, reg_umin(reg1) + 1, U64_MAX);
16469 		}
16470 		break;
16471 	case BPF_JSLE:
16472 		if (is_jmp32) {
16473 			cnum32_intersect_with_srange(&reg1->r32, S32_MIN, reg_s32_max(reg2));
16474 			cnum32_intersect_with_srange(&reg2->r32, reg_s32_min(reg1), S32_MAX);
16475 		} else {
16476 			cnum64_intersect_with_srange(&reg1->r64, S64_MIN, reg_smax(reg2));
16477 			cnum64_intersect_with_srange(&reg2->r64, reg_smin(reg1), S64_MAX);
16478 		}
16479 		break;
16480 	case BPF_JSLT:
16481 		if (is_jmp32) {
16482 			cnum32_intersect_with_srange(&reg1->r32, S32_MIN, reg_s32_max(reg2) - 1);
16483 			cnum32_intersect_with_srange(&reg2->r32, reg_s32_min(reg1) + 1, S32_MAX);
16484 		} else {
16485 			cnum64_intersect_with_srange(&reg1->r64, S64_MIN, reg_smax(reg2) - 1);
16486 			cnum64_intersect_with_srange(&reg2->r64, reg_smin(reg1) + 1, S64_MAX);
16487 		}
16488 		break;
16489 	default:
16490 		return;
16491 	}
16492 }
16493 
16494 /* Check for invariant violations on the registers for both branches of a condition */
regs_bounds_sanity_check_branches(struct bpf_verifier_env * env)16495 static int regs_bounds_sanity_check_branches(struct bpf_verifier_env *env)
16496 {
16497 	int err;
16498 
16499 	err = reg_bounds_sanity_check(env, &env->true_reg1, "true_reg1");
16500 	err = err ?: reg_bounds_sanity_check(env, &env->true_reg2, "true_reg2");
16501 	err = err ?: reg_bounds_sanity_check(env, &env->false_reg1, "false_reg1");
16502 	err = err ?: reg_bounds_sanity_check(env, &env->false_reg2, "false_reg2");
16503 	return err;
16504 }
16505 
mark_ptr_or_null_reg(struct bpf_func_state * state,struct bpf_reg_state * reg,u32 id,bool is_null)16506 static void mark_ptr_or_null_reg(struct bpf_func_state *state,
16507 				 struct bpf_reg_state *reg, u32 id,
16508 				 bool is_null)
16509 {
16510 	if (type_may_be_null(reg->type) && reg->id == id &&
16511 	    (is_rcu_reg(reg) || !WARN_ON_ONCE(!reg->id))) {
16512 		/* Old offset should have been known-zero, because we don't
16513 		 * allow pointer arithmetic on pointers that might be NULL.
16514 		 * If we see this happening, don't convert the register.
16515 		 *
16516 		 * But in some cases, some helpers that return local kptrs
16517 		 * advance offset for the returned pointer. In those cases,
16518 		 * it is fine to expect to see reg->var_off.
16519 		 */
16520 		if (!(type_is_ptr_alloc_obj(reg->type) || type_is_non_owning_ref(reg->type)) &&
16521 		    WARN_ON_ONCE(!tnum_equals_const(reg->var_off, 0)))
16522 			return;
16523 		if (is_null) {
16524 			/* We don't need id from this point
16525 			 * onwards anymore, thus we should better reset it,
16526 			 * so that state pruning has chances to take effect.
16527 			 */
16528 			__mark_reg_known_zero(reg);
16529 			reg->type = SCALAR_VALUE;
16530 
16531 			return;
16532 		}
16533 
16534 		mark_ptr_not_null_reg(reg);
16535 
16536 		/*
16537 		 * reg->id is preserved for object relationship tracking
16538 		 * and spin_lock lock state tracking
16539 		 */
16540 	}
16541 }
16542 
16543 /* The logic is similar to find_good_pkt_pointers(), both could eventually
16544  * be folded together at some point.
16545  */
mark_ptr_or_null_regs(struct bpf_verifier_state * vstate,u32 regno,bool is_null)16546 static void mark_ptr_or_null_regs(struct bpf_verifier_state *vstate, u32 regno,
16547 				  bool is_null)
16548 {
16549 	struct bpf_func_state *state = vstate->frame[vstate->curframe];
16550 	struct bpf_reg_state *regs = state->regs, *reg;
16551 	u32 id = regs[regno].id;
16552 
16553 	if (is_null && find_reference_state(vstate, id))
16554 		/* regs[regno] is in the " == NULL" branch.
16555 		 * No one could have freed the reference state before
16556 		 * doing the NULL check.
16557 		 */
16558 		WARN_ON_ONCE(__release_reference_nomark(vstate, id));
16559 
16560 	bpf_for_each_reg_in_vstate(vstate, state, reg, ({
16561 		mark_ptr_or_null_reg(state, reg, id, is_null);
16562 	}));
16563 }
16564 
try_match_pkt_pointers(const struct bpf_insn * insn,struct bpf_reg_state * dst_reg,struct bpf_reg_state * src_reg,struct bpf_verifier_state * this_branch,struct bpf_verifier_state * other_branch)16565 static bool try_match_pkt_pointers(const struct bpf_insn *insn,
16566 				   struct bpf_reg_state *dst_reg,
16567 				   struct bpf_reg_state *src_reg,
16568 				   struct bpf_verifier_state *this_branch,
16569 				   struct bpf_verifier_state *other_branch)
16570 {
16571 	if (BPF_SRC(insn->code) != BPF_X)
16572 		return false;
16573 
16574 	/* Pointers are always 64-bit. */
16575 	if (BPF_CLASS(insn->code) == BPF_JMP32)
16576 		return false;
16577 
16578 	switch (BPF_OP(insn->code)) {
16579 	case BPF_JGT:
16580 		if ((dst_reg->type == PTR_TO_PACKET &&
16581 		     src_reg->type == PTR_TO_PACKET_END) ||
16582 		    (dst_reg->type == PTR_TO_PACKET_META &&
16583 		     reg_is_init_pkt_pointer(src_reg, PTR_TO_PACKET))) {
16584 			/* pkt_data' > pkt_end, pkt_meta' > pkt_data */
16585 			find_good_pkt_pointers(this_branch, dst_reg,
16586 					       dst_reg->type, false);
16587 			mark_pkt_end(other_branch, insn->dst_reg, true);
16588 		} else if ((dst_reg->type == PTR_TO_PACKET_END &&
16589 			    src_reg->type == PTR_TO_PACKET) ||
16590 			   (reg_is_init_pkt_pointer(dst_reg, PTR_TO_PACKET) &&
16591 			    src_reg->type == PTR_TO_PACKET_META)) {
16592 			/* pkt_end > pkt_data', pkt_data > pkt_meta' */
16593 			find_good_pkt_pointers(other_branch, src_reg,
16594 					       src_reg->type, true);
16595 			mark_pkt_end(this_branch, insn->src_reg, false);
16596 		} else {
16597 			return false;
16598 		}
16599 		break;
16600 	case BPF_JLT:
16601 		if ((dst_reg->type == PTR_TO_PACKET &&
16602 		     src_reg->type == PTR_TO_PACKET_END) ||
16603 		    (dst_reg->type == PTR_TO_PACKET_META &&
16604 		     reg_is_init_pkt_pointer(src_reg, PTR_TO_PACKET))) {
16605 			/* pkt_data' < pkt_end, pkt_meta' < pkt_data */
16606 			find_good_pkt_pointers(other_branch, dst_reg,
16607 					       dst_reg->type, true);
16608 			mark_pkt_end(this_branch, insn->dst_reg, false);
16609 		} else if ((dst_reg->type == PTR_TO_PACKET_END &&
16610 			    src_reg->type == PTR_TO_PACKET) ||
16611 			   (reg_is_init_pkt_pointer(dst_reg, PTR_TO_PACKET) &&
16612 			    src_reg->type == PTR_TO_PACKET_META)) {
16613 			/* pkt_end < pkt_data', pkt_data > pkt_meta' */
16614 			find_good_pkt_pointers(this_branch, src_reg,
16615 					       src_reg->type, false);
16616 			mark_pkt_end(other_branch, insn->src_reg, true);
16617 		} else {
16618 			return false;
16619 		}
16620 		break;
16621 	case BPF_JGE:
16622 		if ((dst_reg->type == PTR_TO_PACKET &&
16623 		     src_reg->type == PTR_TO_PACKET_END) ||
16624 		    (dst_reg->type == PTR_TO_PACKET_META &&
16625 		     reg_is_init_pkt_pointer(src_reg, PTR_TO_PACKET))) {
16626 			/* pkt_data' >= pkt_end, pkt_meta' >= pkt_data */
16627 			find_good_pkt_pointers(this_branch, dst_reg,
16628 					       dst_reg->type, true);
16629 			mark_pkt_end(other_branch, insn->dst_reg, false);
16630 		} else if ((dst_reg->type == PTR_TO_PACKET_END &&
16631 			    src_reg->type == PTR_TO_PACKET) ||
16632 			   (reg_is_init_pkt_pointer(dst_reg, PTR_TO_PACKET) &&
16633 			    src_reg->type == PTR_TO_PACKET_META)) {
16634 			/* pkt_end >= pkt_data', pkt_data >= pkt_meta' */
16635 			find_good_pkt_pointers(other_branch, src_reg,
16636 					       src_reg->type, false);
16637 			mark_pkt_end(this_branch, insn->src_reg, true);
16638 		} else {
16639 			return false;
16640 		}
16641 		break;
16642 	case BPF_JLE:
16643 		if ((dst_reg->type == PTR_TO_PACKET &&
16644 		     src_reg->type == PTR_TO_PACKET_END) ||
16645 		    (dst_reg->type == PTR_TO_PACKET_META &&
16646 		     reg_is_init_pkt_pointer(src_reg, PTR_TO_PACKET))) {
16647 			/* pkt_data' <= pkt_end, pkt_meta' <= pkt_data */
16648 			find_good_pkt_pointers(other_branch, dst_reg,
16649 					       dst_reg->type, false);
16650 			mark_pkt_end(this_branch, insn->dst_reg, true);
16651 		} else if ((dst_reg->type == PTR_TO_PACKET_END &&
16652 			    src_reg->type == PTR_TO_PACKET) ||
16653 			   (reg_is_init_pkt_pointer(dst_reg, PTR_TO_PACKET) &&
16654 			    src_reg->type == PTR_TO_PACKET_META)) {
16655 			/* pkt_end <= pkt_data', pkt_data <= pkt_meta' */
16656 			find_good_pkt_pointers(this_branch, src_reg,
16657 					       src_reg->type, true);
16658 			mark_pkt_end(other_branch, insn->src_reg, false);
16659 		} else {
16660 			return false;
16661 		}
16662 		break;
16663 	default:
16664 		return false;
16665 	}
16666 
16667 	return true;
16668 }
16669 
__collect_linked_regs(struct linked_regs * reg_set,struct bpf_reg_state * reg,u32 id,u32 frameno,u32 spi_or_reg,bool is_reg)16670 static void __collect_linked_regs(struct linked_regs *reg_set, struct bpf_reg_state *reg,
16671 				  u32 id, u32 frameno, u32 spi_or_reg, bool is_reg)
16672 {
16673 	struct linked_reg *e;
16674 
16675 	if (reg->type != SCALAR_VALUE || (reg->id & ~BPF_ADD_CONST) != id)
16676 		return;
16677 
16678 	e = linked_regs_push(reg_set);
16679 	if (e) {
16680 		e->frameno = frameno;
16681 		e->is_reg = is_reg;
16682 		e->regno = spi_or_reg;
16683 	} else {
16684 		clear_scalar_id(reg);
16685 	}
16686 }
16687 
16688 /* For all R being scalar registers or spilled scalar registers
16689  * in verifier state, save R in linked_regs if R->id == id.
16690  * If there are too many Rs sharing same id, reset id for leftover Rs.
16691  */
collect_linked_regs(struct bpf_verifier_env * env,struct bpf_verifier_state * vstate,u32 id,struct linked_regs * linked_regs)16692 static void collect_linked_regs(struct bpf_verifier_env *env,
16693 				struct bpf_verifier_state *vstate,
16694 				u32 id,
16695 				struct linked_regs *linked_regs)
16696 {
16697 	struct bpf_insn_aux_data *aux = env->insn_aux_data;
16698 	struct bpf_func_state *func;
16699 	struct bpf_reg_state *reg;
16700 	u16 live_regs;
16701 	int i, j;
16702 
16703 	id = id & ~BPF_ADD_CONST;
16704 	for (i = vstate->curframe; i >= 0; i--) {
16705 		live_regs = aux[bpf_frame_insn_idx(vstate, i)].live_regs_before;
16706 		func = vstate->frame[i];
16707 		for (j = 0; j < BPF_REG_FP; j++) {
16708 			if (!(live_regs & BIT(j)))
16709 				continue;
16710 			reg = &func->regs[j];
16711 			__collect_linked_regs(linked_regs, reg, id, i, j, true);
16712 		}
16713 		for (j = 0; j < func->allocated_stack / BPF_REG_SIZE; j++) {
16714 			if (!bpf_is_spilled_reg(&func->stack[j]))
16715 				continue;
16716 			reg = &func->stack[j].spilled_ptr;
16717 			__collect_linked_regs(linked_regs, reg, id, i, j, false);
16718 		}
16719 	}
16720 }
16721 
16722 /* For all R in linked_regs, copy known_reg range into R
16723  * if R->id == known_reg->id.
16724  */
sync_linked_regs(struct bpf_verifier_env * env,struct bpf_verifier_state * vstate,struct bpf_reg_state * known_reg,struct linked_regs * linked_regs)16725 static void sync_linked_regs(struct bpf_verifier_env *env, struct bpf_verifier_state *vstate,
16726 			     struct bpf_reg_state *known_reg, struct linked_regs *linked_regs)
16727 {
16728 	struct bpf_reg_state fake_reg;
16729 	struct bpf_reg_state *reg;
16730 	struct linked_reg *e;
16731 	int i;
16732 
16733 	for (i = 0; i < linked_regs->cnt; ++i) {
16734 		e = &linked_regs->entries[i];
16735 		reg = e->is_reg ? &vstate->frame[e->frameno]->regs[e->regno]
16736 				: &vstate->frame[e->frameno]->stack[e->spi].spilled_ptr;
16737 		if (reg->type != SCALAR_VALUE || reg == known_reg)
16738 			continue;
16739 		if ((reg->id & ~BPF_ADD_CONST) != (known_reg->id & ~BPF_ADD_CONST))
16740 			continue;
16741 		/*
16742 		 * Skip mixed 32/64-bit links: the delta relationship doesn't
16743 		 * hold across different ALU widths.
16744 		 */
16745 		if (((reg->id ^ known_reg->id) & BPF_ADD_CONST) == BPF_ADD_CONST)
16746 			continue;
16747 		if ((!(reg->id & BPF_ADD_CONST) && !(known_reg->id & BPF_ADD_CONST)) ||
16748 		    reg->delta == known_reg->delta) {
16749 			*reg = *known_reg;
16750 		} else {
16751 			s32 saved_off = reg->delta;
16752 			u32 saved_id = reg->id;
16753 
16754 			fake_reg.type = SCALAR_VALUE;
16755 			__mark_reg_known(&fake_reg, (s64)reg->delta - (s64)known_reg->delta);
16756 
16757 			/* reg = known_reg; reg += delta */
16758 			*reg = *known_reg;
16759 			/*
16760 			 * Must preserve off and id, otherwise another sync_linked_regs()
16761 			 * will be incorrect.
16762 			 */
16763 			reg->delta = saved_off;
16764 			reg->id = saved_id;
16765 
16766 			scalar32_min_max_add(reg, &fake_reg);
16767 			scalar_min_max_add(reg, &fake_reg);
16768 			reg->var_off = tnum_add(reg->var_off, fake_reg.var_off);
16769 			if ((reg->id | known_reg->id) & BPF_ADD_CONST32)
16770 				zext_32_to_64(reg);
16771 			reg_bounds_sync(reg);
16772 		}
16773 		if (e->is_reg)
16774 			mark_reg_scratched(env, e->regno);
16775 		else
16776 			mark_stack_slot_scratched(env, e->spi);
16777 	}
16778 }
16779 
check_cond_jmp_op(struct bpf_verifier_env * env,struct bpf_insn * insn,int * insn_idx)16780 static int check_cond_jmp_op(struct bpf_verifier_env *env,
16781 			     struct bpf_insn *insn, int *insn_idx)
16782 {
16783 	struct bpf_verifier_state *this_branch = env->cur_state;
16784 	struct bpf_verifier_state *other_branch;
16785 	struct bpf_reg_state *regs = this_branch->frame[this_branch->curframe]->regs;
16786 	struct bpf_reg_state *dst_reg, *other_branch_regs, *src_reg = NULL;
16787 	struct bpf_reg_state *eq_branch_regs;
16788 	struct linked_regs linked_regs = {};
16789 	u8 opcode = BPF_OP(insn->code);
16790 	int insn_flags = 0;
16791 	bool is_jmp32;
16792 	int pred = -1;
16793 	int err;
16794 
16795 	/* Only conditional jumps are expected to reach here. */
16796 	if (opcode == BPF_JA || opcode > BPF_JCOND) {
16797 		verbose(env, "invalid BPF_JMP/JMP32 opcode %x\n", opcode);
16798 		return -EINVAL;
16799 	}
16800 
16801 	if (opcode == BPF_JCOND) {
16802 		struct bpf_verifier_state *cur_st = env->cur_state, *queued_st, *prev_st;
16803 		int idx = *insn_idx;
16804 
16805 		prev_st = find_prev_entry(env, cur_st->parent, idx);
16806 
16807 		/* branch out 'fallthrough' insn as a new state to explore */
16808 		queued_st = push_stack(env, idx + 1, idx, false);
16809 		if (IS_ERR(queued_st))
16810 			return PTR_ERR(queued_st);
16811 
16812 		queued_st->may_goto_depth++;
16813 		if (prev_st)
16814 			widen_imprecise_scalars(env, prev_st, queued_st);
16815 		*insn_idx += insn->off;
16816 		return 0;
16817 	}
16818 
16819 	/* check src2 operand */
16820 	err = check_reg_arg(env, insn->dst_reg, SRC_OP);
16821 	if (err)
16822 		return err;
16823 
16824 	dst_reg = &regs[insn->dst_reg];
16825 	if (BPF_SRC(insn->code) == BPF_X) {
16826 		/* check src1 operand */
16827 		err = check_reg_arg(env, insn->src_reg, SRC_OP);
16828 		if (err)
16829 			return err;
16830 
16831 		src_reg = &regs[insn->src_reg];
16832 		if (!(reg_is_pkt_pointer_any(dst_reg) && reg_is_pkt_pointer_any(src_reg)) &&
16833 		    is_pointer_value(env, insn->src_reg)) {
16834 			verbose(env, "R%d pointer comparison prohibited\n",
16835 				insn->src_reg);
16836 			return -EACCES;
16837 		}
16838 
16839 		if (src_reg->type == PTR_TO_STACK)
16840 			insn_flags |= INSN_F_SRC_REG_STACK;
16841 		if (dst_reg->type == PTR_TO_STACK)
16842 			insn_flags |= INSN_F_DST_REG_STACK;
16843 	} else {
16844 		src_reg = &env->fake_reg[0];
16845 		memset(src_reg, 0, sizeof(*src_reg));
16846 		src_reg->type = SCALAR_VALUE;
16847 		__mark_reg_known(src_reg, insn->imm);
16848 
16849 		if (dst_reg->type == PTR_TO_STACK)
16850 			insn_flags |= INSN_F_DST_REG_STACK;
16851 	}
16852 
16853 	if (insn_flags) {
16854 		err = bpf_push_jmp_history(env, this_branch, insn_flags, 0, 0, 0);
16855 		if (err)
16856 			return err;
16857 	}
16858 
16859 	is_jmp32 = BPF_CLASS(insn->code) == BPF_JMP32;
16860 	env->false_reg1 = *dst_reg;
16861 	env->false_reg2 = *src_reg;
16862 	env->true_reg1 = *dst_reg;
16863 	env->true_reg2 = *src_reg;
16864 	pred = is_branch_taken(env, dst_reg, src_reg, opcode, is_jmp32);
16865 	if (pred >= 0) {
16866 		/* If we get here with a dst_reg pointer type it is because
16867 		 * above is_branch_taken() special cased the 0 comparison.
16868 		 */
16869 		if (!__is_pointer_value(false, dst_reg))
16870 			err = mark_chain_precision(env, insn->dst_reg);
16871 		if (BPF_SRC(insn->code) == BPF_X && !err &&
16872 		    !__is_pointer_value(false, src_reg))
16873 			err = mark_chain_precision(env, insn->src_reg);
16874 		if (err)
16875 			return err;
16876 	}
16877 
16878 	if (pred == 1) {
16879 		/* Only follow the goto, ignore fall-through. If needed, push
16880 		 * the fall-through branch for simulation under speculative
16881 		 * execution.
16882 		 */
16883 		if (!env->bypass_spec_v1) {
16884 			err = sanitize_speculative_path(env, insn, *insn_idx + 1, *insn_idx);
16885 			if (err < 0)
16886 				return err;
16887 		}
16888 		if (env->log.level & BPF_LOG_LEVEL)
16889 			print_insn_state(env, this_branch, this_branch->curframe);
16890 		*insn_idx += insn->off;
16891 		return 0;
16892 	} else if (pred == 0) {
16893 		/* Only follow the fall-through branch, since that's where the
16894 		 * program will go. If needed, push the goto branch for
16895 		 * simulation under speculative execution.
16896 		 */
16897 		if (!env->bypass_spec_v1) {
16898 			err = sanitize_speculative_path(env, insn, *insn_idx + insn->off + 1,
16899 							*insn_idx);
16900 			if (err < 0)
16901 				return err;
16902 		}
16903 		if (env->log.level & BPF_LOG_LEVEL)
16904 			print_insn_state(env, this_branch, this_branch->curframe);
16905 		return 0;
16906 	}
16907 
16908 	/* Push scalar registers sharing same ID to jump history,
16909 	 * do this before creating 'other_branch', so that both
16910 	 * 'this_branch' and 'other_branch' share this history
16911 	 * if parent state is created.
16912 	 */
16913 	if (BPF_SRC(insn->code) == BPF_X && src_reg->type == SCALAR_VALUE && src_reg->id)
16914 		collect_linked_regs(env, this_branch, src_reg->id, &linked_regs);
16915 	if (dst_reg->type == SCALAR_VALUE && dst_reg->id)
16916 		collect_linked_regs(env, this_branch, dst_reg->id, &linked_regs);
16917 	if (linked_regs.cnt > 1) {
16918 		err = bpf_push_jmp_history(env, this_branch, 0, 0, 0, linked_regs_pack(&linked_regs));
16919 		if (err)
16920 			return err;
16921 	}
16922 
16923 	other_branch = push_stack(env, *insn_idx + insn->off + 1, *insn_idx, false);
16924 	if (IS_ERR(other_branch))
16925 		return PTR_ERR(other_branch);
16926 	other_branch_regs = other_branch->frame[other_branch->curframe]->regs;
16927 
16928 	err = regs_bounds_sanity_check_branches(env);
16929 	if (err)
16930 		return err;
16931 
16932 	*dst_reg = env->false_reg1;
16933 	*src_reg = env->false_reg2;
16934 	other_branch_regs[insn->dst_reg] = env->true_reg1;
16935 	if (BPF_SRC(insn->code) == BPF_X)
16936 		other_branch_regs[insn->src_reg] = env->true_reg2;
16937 
16938 	if (BPF_SRC(insn->code) == BPF_X &&
16939 	    src_reg->type == SCALAR_VALUE && src_reg->id &&
16940 	    !WARN_ON_ONCE(src_reg->id != other_branch_regs[insn->src_reg].id)) {
16941 		sync_linked_regs(env, this_branch, src_reg, &linked_regs);
16942 		sync_linked_regs(env, other_branch, &other_branch_regs[insn->src_reg],
16943 				 &linked_regs);
16944 	}
16945 	if (dst_reg->type == SCALAR_VALUE && dst_reg->id &&
16946 	    !WARN_ON_ONCE(dst_reg->id != other_branch_regs[insn->dst_reg].id)) {
16947 		sync_linked_regs(env, this_branch, dst_reg, &linked_regs);
16948 		sync_linked_regs(env, other_branch, &other_branch_regs[insn->dst_reg],
16949 				 &linked_regs);
16950 	}
16951 
16952 	/* if one pointer register is compared to another pointer
16953 	 * register check if PTR_MAYBE_NULL could be lifted.
16954 	 * E.g. register A - maybe null
16955 	 *      register B - not null
16956 	 * for JNE A, B, ... - A is not null in the false branch;
16957 	 * for JEQ A, B, ... - A is not null in the true branch.
16958 	 *
16959 	 * Since PTR_TO_BTF_ID points to a kernel struct that does
16960 	 * not need to be null checked by the BPF program, i.e.,
16961 	 * could be null even without PTR_MAYBE_NULL marking, so
16962 	 * only propagate nullness when neither reg is that type.
16963 	 */
16964 	if (!is_jmp32 && BPF_SRC(insn->code) == BPF_X &&
16965 	    __is_pointer_value(false, src_reg) && __is_pointer_value(false, dst_reg) &&
16966 	    type_may_be_null(src_reg->type) != type_may_be_null(dst_reg->type) &&
16967 	    base_type(src_reg->type) != PTR_TO_BTF_ID &&
16968 	    base_type(dst_reg->type) != PTR_TO_BTF_ID) {
16969 		eq_branch_regs = NULL;
16970 		switch (opcode) {
16971 		case BPF_JEQ:
16972 			eq_branch_regs = other_branch_regs;
16973 			break;
16974 		case BPF_JNE:
16975 			eq_branch_regs = regs;
16976 			break;
16977 		default:
16978 			/* do nothing */
16979 			break;
16980 		}
16981 		if (eq_branch_regs) {
16982 			if (type_may_be_null(src_reg->type))
16983 				mark_ptr_not_null_reg(&eq_branch_regs[insn->src_reg]);
16984 			else
16985 				mark_ptr_not_null_reg(&eq_branch_regs[insn->dst_reg]);
16986 		}
16987 	}
16988 
16989 	/* detect if R == 0 where R is returned from bpf_map_lookup_elem().
16990 	 * Also does the same detection for a register whose the value is
16991 	 * known to be 0.
16992 	 * NOTE: these optimizations below are related with pointer comparison
16993 	 *       which will never be JMP32.
16994 	 */
16995 	if (!is_jmp32 && (opcode == BPF_JEQ || opcode == BPF_JNE) &&
16996 	    type_may_be_null(dst_reg->type) &&
16997 	    ((BPF_SRC(insn->code) == BPF_K && insn->imm == 0) ||
16998 	     (BPF_SRC(insn->code) == BPF_X && bpf_register_is_null(src_reg)))) {
16999 		/* Mark all identical registers in each branch as either
17000 		 * safe or unknown depending R == 0 or R != 0 conditional.
17001 		 */
17002 		mark_ptr_or_null_regs(this_branch, insn->dst_reg,
17003 				      opcode == BPF_JNE);
17004 		mark_ptr_or_null_regs(other_branch, insn->dst_reg,
17005 				      opcode == BPF_JEQ);
17006 	} else if (!try_match_pkt_pointers(insn, dst_reg, &regs[insn->src_reg],
17007 					   this_branch, other_branch) &&
17008 		   is_pointer_value(env, insn->dst_reg)) {
17009 		verbose(env, "R%d pointer comparison prohibited\n",
17010 			insn->dst_reg);
17011 		return -EACCES;
17012 	}
17013 	if (env->log.level & BPF_LOG_LEVEL)
17014 		print_insn_state(env, this_branch, this_branch->curframe);
17015 	return 0;
17016 }
17017 
17018 /* verify BPF_LD_IMM64 instruction */
check_ld_imm(struct bpf_verifier_env * env,struct bpf_insn * insn)17019 static int check_ld_imm(struct bpf_verifier_env *env, struct bpf_insn *insn)
17020 {
17021 	struct bpf_insn_aux_data *aux = cur_aux(env);
17022 	struct bpf_reg_state *regs = cur_regs(env);
17023 	struct bpf_reg_state *dst_reg;
17024 	struct bpf_map *map;
17025 	int err;
17026 
17027 	if (BPF_SIZE(insn->code) != BPF_DW) {
17028 		verbose(env, "invalid BPF_LD_IMM insn\n");
17029 		return -EINVAL;
17030 	}
17031 
17032 	err = check_reg_arg(env, insn->dst_reg, DST_OP);
17033 	if (err)
17034 		return err;
17035 
17036 	dst_reg = &regs[insn->dst_reg];
17037 	bpf_diag_mod_begin(env, dst_reg, NULL, BPF_DIAG_MOD_WRITE);
17038 	if (insn->src_reg == 0) {
17039 		u64 imm = ((u64)(insn + 1)->imm << 32) | (u32)insn->imm;
17040 
17041 		dst_reg->type = SCALAR_VALUE;
17042 		__mark_reg_known(&regs[insn->dst_reg], imm);
17043 		bpf_diag_mod_end(env);
17044 		return 0;
17045 	}
17046 
17047 	/* All special src_reg cases are listed below. From this point onwards
17048 	 * we either succeed and assign a corresponding dst_reg->type after
17049 	 * zeroing the offset, or fail and reject the program.
17050 	 */
17051 	mark_reg_known_zero(env, regs, insn->dst_reg);
17052 
17053 	if (insn->src_reg == BPF_PSEUDO_BTF_ID) {
17054 		dst_reg->type = aux->btf_var.reg_type;
17055 		switch (base_type(dst_reg->type)) {
17056 		case PTR_TO_MEM:
17057 			dst_reg->mem_size = aux->btf_var.mem_size;
17058 			break;
17059 		case PTR_TO_BTF_ID:
17060 			dst_reg->btf = aux->btf_var.btf;
17061 			dst_reg->btf_id = aux->btf_var.btf_id;
17062 			break;
17063 		default:
17064 			verifier_bug(env, "pseudo btf id: unexpected dst reg type");
17065 			return -EFAULT;
17066 		}
17067 		bpf_diag_mod_end(env);
17068 		return 0;
17069 	}
17070 
17071 	if (insn->src_reg == BPF_PSEUDO_FUNC) {
17072 		struct bpf_prog_aux *aux = env->prog->aux;
17073 		u32 subprogno = bpf_find_subprog(env,
17074 						 env->insn_idx + insn->imm + 1);
17075 
17076 		if (!aux->func_info) {
17077 			verbose(env, "missing btf func_info\n");
17078 			return -EINVAL;
17079 		}
17080 		if (aux->func_info_aux[subprogno].linkage != BTF_FUNC_STATIC) {
17081 			verbose(env, "callback function not static\n");
17082 			return -EINVAL;
17083 		}
17084 
17085 		dst_reg->type = PTR_TO_FUNC;
17086 		dst_reg->subprogno = subprogno;
17087 		bpf_diag_mod_end(env);
17088 		return 0;
17089 	}
17090 
17091 	map = env->used_maps[aux->map_index];
17092 
17093 	if (insn->src_reg == BPF_PSEUDO_MAP_VALUE ||
17094 	    insn->src_reg == BPF_PSEUDO_MAP_IDX_VALUE) {
17095 		if (map->map_type == BPF_MAP_TYPE_ARENA) {
17096 			__mark_reg_unknown(env, dst_reg);
17097 			dst_reg->map_ptr = map;
17098 			bpf_diag_mod_end(env);
17099 			return 0;
17100 		}
17101 		__mark_reg_known(dst_reg, aux->map_off);
17102 		dst_reg->type = PTR_TO_MAP_VALUE;
17103 		dst_reg->map_ptr = map;
17104 		WARN_ON_ONCE(map->map_type != BPF_MAP_TYPE_INSN_ARRAY &&
17105 			     map->max_entries != 1);
17106 		/* We want reg->id to be same (0) as map_value is not distinct */
17107 	} else if (insn->src_reg == BPF_PSEUDO_MAP_FD ||
17108 		   insn->src_reg == BPF_PSEUDO_MAP_IDX) {
17109 		dst_reg->type = CONST_PTR_TO_MAP;
17110 		dst_reg->map_ptr = map;
17111 	} else {
17112 		verifier_bug(env, "unexpected src reg value for ldimm64");
17113 		return -EFAULT;
17114 	}
17115 
17116 	bpf_diag_mod_end(env);
17117 	return 0;
17118 }
17119 
may_access_skb(enum bpf_prog_type type)17120 static bool may_access_skb(enum bpf_prog_type type)
17121 {
17122 	switch (type) {
17123 	case BPF_PROG_TYPE_SOCKET_FILTER:
17124 	case BPF_PROG_TYPE_SCHED_CLS:
17125 	case BPF_PROG_TYPE_SCHED_ACT:
17126 		return true;
17127 	default:
17128 		return false;
17129 	}
17130 }
17131 
17132 /* verify safety of LD_ABS|LD_IND instructions:
17133  * - they can only appear in the programs where ctx == skb
17134  * - since they are wrappers of function calls, they scratch R1-R5 registers,
17135  *   preserve R6-R9, and store return value into R0
17136  *
17137  * Implicit input:
17138  *   ctx == skb == R6 == CTX
17139  *
17140  * Explicit input:
17141  *   SRC == any register
17142  *   IMM == 32-bit immediate
17143  *
17144  * Output:
17145  *   R0 - 8/16/32-bit skb data converted to cpu endianness
17146  */
check_ld_abs(struct bpf_verifier_env * env,struct bpf_insn * insn)17147 static int check_ld_abs(struct bpf_verifier_env *env, struct bpf_insn *insn)
17148 {
17149 	struct bpf_reg_state *regs = cur_regs(env);
17150 	static const int ctx_reg = BPF_REG_6;
17151 	u8 mode = BPF_MODE(insn->code);
17152 	int i, err;
17153 
17154 	if (!may_access_skb(resolve_prog_type(env->prog))) {
17155 		verbose(env, "BPF_LD_[ABS|IND] instructions not allowed for this program type\n");
17156 		return -EINVAL;
17157 	}
17158 
17159 	if (!env->ops->gen_ld_abs) {
17160 		verifier_bug(env, "gen_ld_abs is null");
17161 		return -EFAULT;
17162 	}
17163 
17164 	/* check whether implicit source operand (register R6) is readable */
17165 	err = check_reg_arg(env, ctx_reg, SRC_OP);
17166 	if (err)
17167 		return err;
17168 
17169 	/* Disallow usage of BPF_LD_[ABS|IND] with reference tracking, as
17170 	 * gen_ld_abs() may terminate the program at runtime, leading to
17171 	 * reference leak.
17172 	 */
17173 	err = check_resource_leak(env, false, true, "BPF_LD_[ABS|IND]");
17174 	if (err)
17175 		return err;
17176 
17177 	if (regs[ctx_reg].type != PTR_TO_CTX) {
17178 		verbose(env,
17179 			"at the time of BPF_LD_ABS|IND R6 != pointer to skb\n");
17180 		return -EINVAL;
17181 	}
17182 
17183 	if (mode == BPF_IND) {
17184 		/* check explicit source operand */
17185 		err = check_reg_arg(env, insn->src_reg, SRC_OP);
17186 		if (err)
17187 			return err;
17188 	}
17189 
17190 	err = check_ptr_off_reg(env, &regs[ctx_reg], ctx_reg);
17191 	if (err < 0)
17192 		return err;
17193 
17194 	/* reset caller saved regs to unreadable */
17195 	bpf_diag_record_caller_saved(env, regs);
17196 	bpf_diag_mod_begin(env, &regs[BPF_REG_0], NULL, BPF_DIAG_MOD_WRITE);
17197 	for (i = 0; i < CALLER_SAVED_REGS; i++) {
17198 		bpf_mark_reg_not_init(env, &regs[caller_saved[i]]);
17199 		check_reg_arg(env, caller_saved[i], DST_OP_NO_MARK);
17200 	}
17201 
17202 	/* mark destination R0 register as readable, since it contains
17203 	 * the value fetched from the packet.
17204 	 * Already marked as written above.
17205 	 */
17206 	mark_reg_unknown(env, regs, BPF_REG_0);
17207 	bpf_diag_mod_end(env);
17208 	/*
17209 	 * See bpf_gen_ld_abs() which emits a hidden BPF_EXIT with r0=0
17210 	 * which must be explored by the verifier when in a subprog.
17211 	 */
17212 	if (env->cur_state->curframe) {
17213 		struct bpf_verifier_state *branch;
17214 
17215 		mark_reg_scratched(env, BPF_REG_0);
17216 		branch = push_stack(env, env->insn_idx + 1, env->insn_idx, false);
17217 		if (IS_ERR(branch))
17218 			return PTR_ERR(branch);
17219 		mark_reg_known_zero(env, regs, BPF_REG_0);
17220 		err = prepare_func_exit(env, &env->insn_idx);
17221 		if (err)
17222 			return err;
17223 		env->insn_idx--;
17224 	}
17225 	return 0;
17226 }
17227 
return_retval_range(struct bpf_verifier_env * env,struct bpf_retval_range * range)17228 static bool return_retval_range(struct bpf_verifier_env *env, struct bpf_retval_range *range)
17229 {
17230 	enum bpf_prog_type prog_type = resolve_prog_type(env->prog);
17231 
17232 	/* Default return value range. */
17233 	*range = retval_range(0, 1);
17234 
17235 	switch (prog_type) {
17236 	case BPF_PROG_TYPE_CGROUP_SOCK_ADDR:
17237 		switch (env->prog->expected_attach_type) {
17238 		case BPF_CGROUP_UDP4_RECVMSG:
17239 		case BPF_CGROUP_UDP6_RECVMSG:
17240 		case BPF_CGROUP_UNIX_RECVMSG:
17241 		case BPF_CGROUP_INET4_GETPEERNAME:
17242 		case BPF_CGROUP_INET6_GETPEERNAME:
17243 		case BPF_CGROUP_UNIX_GETPEERNAME:
17244 		case BPF_CGROUP_INET4_GETSOCKNAME:
17245 		case BPF_CGROUP_INET6_GETSOCKNAME:
17246 		case BPF_CGROUP_UNIX_GETSOCKNAME:
17247 			*range = retval_range(1, 1);
17248 			break;
17249 		case BPF_CGROUP_INET4_BIND:
17250 		case BPF_CGROUP_INET6_BIND:
17251 			*range = retval_range(0, 3);
17252 			break;
17253 		default:
17254 			break;
17255 		}
17256 		break;
17257 	case BPF_PROG_TYPE_CGROUP_SKB:
17258 		if (env->prog->expected_attach_type == BPF_CGROUP_INET_EGRESS)
17259 			*range = retval_range(0, 3);
17260 		break;
17261 	case BPF_PROG_TYPE_CGROUP_SOCK:
17262 	case BPF_PROG_TYPE_SOCK_OPS:
17263 	case BPF_PROG_TYPE_CGROUP_DEVICE:
17264 	case BPF_PROG_TYPE_CGROUP_SYSCTL:
17265 	case BPF_PROG_TYPE_CGROUP_SOCKOPT:
17266 		break;
17267 	case BPF_PROG_TYPE_RAW_TRACEPOINT:
17268 		if (!env->prog->aux->attach_btf_id)
17269 			return false;
17270 		*range = retval_range(0, 0);
17271 		break;
17272 	case BPF_PROG_TYPE_TRACING:
17273 		switch (env->prog->expected_attach_type) {
17274 		case BPF_TRACE_FENTRY:
17275 		case BPF_TRACE_FEXIT:
17276 		case BPF_TRACE_FSESSION:
17277 		case BPF_TRACE_FENTRY_MULTI:
17278 		case BPF_TRACE_FEXIT_MULTI:
17279 		case BPF_TRACE_FSESSION_MULTI:
17280 			*range = retval_range(0, 0);
17281 			break;
17282 		case BPF_TRACE_RAW_TP:
17283 		case BPF_MODIFY_RETURN:
17284 			return false;
17285 		case BPF_TRACE_ITER:
17286 		default:
17287 			break;
17288 		}
17289 		break;
17290 	case BPF_PROG_TYPE_KPROBE:
17291 		switch (env->prog->expected_attach_type) {
17292 		case BPF_TRACE_KPROBE_SESSION:
17293 		case BPF_TRACE_UPROBE_SESSION:
17294 			break;
17295 		default:
17296 			return false;
17297 		}
17298 		break;
17299 	case BPF_PROG_TYPE_SK_LOOKUP:
17300 		*range = retval_range(SK_DROP, SK_PASS);
17301 		break;
17302 
17303 	case BPF_PROG_TYPE_LSM:
17304 		if (env->prog->expected_attach_type != BPF_LSM_CGROUP) {
17305 			/* no range found, any return value is allowed */
17306 			if (!get_func_retval_range(env->prog, range))
17307 				return false;
17308 			/* no restricted range, any return value is allowed */
17309 			if (range->minval == S32_MIN && range->maxval == S32_MAX)
17310 				return false;
17311 			range->return_32bit = true;
17312 		} else if (!env->prog->aux->attach_func_proto->type) {
17313 			/* Make sure programs that attach to void
17314 			 * hooks don't try to modify return value.
17315 			 */
17316 			*range = retval_range(1, 1);
17317 		}
17318 		break;
17319 
17320 	case BPF_PROG_TYPE_NETFILTER:
17321 		*range = retval_range(NF_DROP, NF_ACCEPT);
17322 		break;
17323 	case BPF_PROG_TYPE_STRUCT_OPS:
17324 		*range = retval_range(0, 0);
17325 		break;
17326 	case BPF_PROG_TYPE_EXT:
17327 		/* freplace program can return anything as its return value
17328 		 * depends on the to-be-replaced kernel func or bpf program.
17329 		 */
17330 	default:
17331 		return false;
17332 	}
17333 
17334 	/* Continue calculating. */
17335 
17336 	return true;
17337 }
17338 
program_returns_void(struct bpf_verifier_env * env)17339 static bool program_returns_void(struct bpf_verifier_env *env)
17340 {
17341 	const struct bpf_prog *prog = env->prog;
17342 	enum bpf_prog_type prog_type = prog->type;
17343 
17344 	switch (prog_type) {
17345 	case BPF_PROG_TYPE_LSM:
17346 		/* See return_retval_range, for BPF_LSM_CGROUP can be 0 or 0-1 depending on hook. */
17347 		if (prog->expected_attach_type != BPF_LSM_CGROUP &&
17348 		    !prog->aux->attach_func_proto->type)
17349 			return true;
17350 		break;
17351 	case BPF_PROG_TYPE_STRUCT_OPS:
17352 		if (!prog->aux->attach_func_proto->type)
17353 			return true;
17354 		break;
17355 	case BPF_PROG_TYPE_EXT:
17356 		/*
17357 		 * If the actual program is an extension, let it
17358 		 * return void - attaching will succeed only if the
17359 		 * program being replaced also returns void, and since
17360 		 * it has passed verification its actual type doesn't matter.
17361 		 */
17362 		if (subprog_returns_void(env, 0))
17363 			return true;
17364 		break;
17365 	default:
17366 		break;
17367 	}
17368 	return false;
17369 }
17370 
check_return_code(struct bpf_verifier_env * env,int regno,const char * reg_name)17371 static int check_return_code(struct bpf_verifier_env *env, int regno, const char *reg_name)
17372 {
17373 	const char *exit_ctx = "At program exit";
17374 	struct tnum enforce_attach_type_range = tnum_unknown;
17375 	const struct bpf_prog *prog = env->prog;
17376 	struct bpf_reg_state *reg = reg_state(env, regno);
17377 	struct bpf_retval_range range = retval_range(0, 1);
17378 	enum bpf_prog_type prog_type = resolve_prog_type(env->prog);
17379 	struct bpf_func_state *frame = env->cur_state->frame[0];
17380 	const struct btf_type *reg_type, *ret_type = NULL;
17381 	int err;
17382 
17383 	/* LSM and struct_ops func-ptr's return type could be "void" */
17384 	if (!frame->in_async_callback_fn && program_returns_void(env))
17385 		return 0;
17386 
17387 	if (prog_type == BPF_PROG_TYPE_STRUCT_OPS) {
17388 		/* Allow a struct_ops program to return a referenced kptr if it
17389 		 * matches the operator's return type and is in its unmodified
17390 		 * form. A scalar zero (i.e., a null pointer) is also allowed.
17391 		 */
17392 		reg_type = reg->btf ? btf_type_by_id(reg->btf, reg->btf_id) : NULL;
17393 		ret_type = btf_type_resolve_ptr(prog->aux->attach_btf,
17394 						prog->aux->attach_func_proto->type,
17395 						NULL);
17396 		if (ret_type && ret_type == reg_type && reg_is_referenced(env, reg))
17397 			return __check_ptr_off_reg(env, reg, argno_from_reg(regno), false);
17398 	}
17399 
17400 	/* eBPF calling convention is such that R0 is used
17401 	 * to return the value from eBPF program.
17402 	 * Make sure that it's readable at this time
17403 	 * of bpf_exit, which means that program wrote
17404 	 * something into it earlier
17405 	 */
17406 	err = check_reg_arg(env, regno, SRC_OP);
17407 	if (err)
17408 		return err;
17409 
17410 	if (is_pointer_value(env, regno)) {
17411 		verbose(env, "R%d leaks addr as return value\n", regno);
17412 		return -EACCES;
17413 	}
17414 
17415 	if (frame->in_async_callback_fn) {
17416 		exit_ctx = "At async callback return";
17417 		range = frame->callback_ret_range;
17418 		goto enforce_retval;
17419 	}
17420 
17421 	if (prog_type == BPF_PROG_TYPE_STRUCT_OPS && !ret_type)
17422 		return 0;
17423 
17424 	if (prog_type == BPF_PROG_TYPE_CGROUP_SKB && (env->prog->expected_attach_type == BPF_CGROUP_INET_EGRESS))
17425 		enforce_attach_type_range = tnum_range(2, 3);
17426 
17427 	if (!return_retval_range(env, &range))
17428 		return 0;
17429 
17430 enforce_retval:
17431 	if (reg->type != SCALAR_VALUE) {
17432 		verbose(env, "%s the register R%d is not a known value (%s)\n",
17433 			exit_ctx, regno, reg_type_str(env, reg->type));
17434 		return -EINVAL;
17435 	}
17436 
17437 	err = mark_chain_precision(env, regno);
17438 	if (err)
17439 		return err;
17440 
17441 	if (!retval_range_within(range, reg)) {
17442 		verbose_invalid_scalar(env, reg, range, exit_ctx, reg_name);
17443 		if (prog->expected_attach_type == BPF_LSM_CGROUP &&
17444 		    prog_type == BPF_PROG_TYPE_LSM &&
17445 		    !prog->aux->attach_func_proto->type)
17446 			verbose(env, "Note, BPF_LSM_CGROUP that attach to void LSM hooks can't modify return value!\n");
17447 		return -EINVAL;
17448 	}
17449 
17450 	if (!tnum_is_unknown(enforce_attach_type_range) &&
17451 	    tnum_in(enforce_attach_type_range, reg->var_off))
17452 		env->prog->enforce_expected_attach_type = 1;
17453 	return 0;
17454 }
17455 
check_global_subprog_return_code(struct bpf_verifier_env * env)17456 static int check_global_subprog_return_code(struct bpf_verifier_env *env)
17457 {
17458 	struct bpf_reg_state *reg = reg_state(env, BPF_REG_0);
17459 	struct bpf_func_state *cur_frame = cur_func(env);
17460 	int err;
17461 
17462 	if (subprog_returns_void(env, cur_frame->subprogno))
17463 		return 0;
17464 
17465 	err = check_reg_arg(env, BPF_REG_0, SRC_OP);
17466 	if (err)
17467 		return err;
17468 
17469 	/* Pointers to arena are safe to pass between subprograms. */
17470 	if (is_arena_reg(env, BPF_REG_0))
17471 		return 0;
17472 
17473 	if (is_pointer_value(env, BPF_REG_0)) {
17474 		verbose(env, "R%d leaks addr as return value\n", BPF_REG_0);
17475 		return -EACCES;
17476 	}
17477 
17478 	if (reg->type != SCALAR_VALUE) {
17479 		verbose(env, "At subprogram exit the register R0 is not a scalar value (%s)\n",
17480 			reg_type_str(env, reg->type));
17481 		return -EINVAL;
17482 	}
17483 
17484 	return 0;
17485 }
17486 
17487 /* Bitmask with 1s for all caller saved registers */
17488 #define ALL_CALLER_SAVED_REGS ((1u << CALLER_SAVED_REGS) - 1)
17489 
17490 /* True if do_misc_fixups() replaces calls to helper number 'imm',
17491  * replacement patch is presumed to follow bpf_fastcall contract
17492  * (see mark_fastcall_pattern_for_call() below).
17493  */
bpf_verifier_inlines_helper_call(struct bpf_verifier_env * env,s32 imm)17494 bool bpf_verifier_inlines_helper_call(struct bpf_verifier_env *env, s32 imm)
17495 {
17496 	switch (imm) {
17497 #ifdef CONFIG_X86_64
17498 	case BPF_FUNC_get_smp_processor_id:
17499 #ifdef CONFIG_SMP
17500 	case BPF_FUNC_get_current_task_btf:
17501 	case BPF_FUNC_get_current_task:
17502 #endif
17503 		return env->prog->jit_requested && bpf_jit_supports_percpu_insn();
17504 #endif
17505 	default:
17506 		return false;
17507 	}
17508 }
17509 
17510 /* If @call is a kfunc or helper call, fills @cs and returns true,
17511  * otherwise returns false.
17512  */
bpf_get_call_summary(struct bpf_verifier_env * env,struct bpf_insn * call,struct bpf_call_summary * cs)17513 bool bpf_get_call_summary(struct bpf_verifier_env *env, struct bpf_insn *call,
17514 			  struct bpf_call_summary *cs)
17515 {
17516 	struct bpf_call_arg_meta meta;
17517 	const struct bpf_func_proto *fn;
17518 	int i;
17519 
17520 	if (bpf_helper_call(call)) {
17521 		if (bpf_get_helper_proto(env, call->imm, &fn) < 0)
17522 			/* error would be reported later */
17523 			return false;
17524 		cs->fastcall = fn->allow_fastcall &&
17525 			       (bpf_verifier_inlines_helper_call(env, call->imm) ||
17526 				bpf_jit_inlines_helper_call(call->imm));
17527 		cs->is_void = fn->ret_type == RET_VOID;
17528 		cs->num_params = 0;
17529 		for (i = 0; i < ARRAY_SIZE(fn->arg_type); ++i) {
17530 			if (fn->arg_type[i] == ARG_DONTCARE)
17531 				break;
17532 			cs->num_params++;
17533 		}
17534 		return true;
17535 	}
17536 
17537 	if (bpf_pseudo_kfunc_call(call)) {
17538 		int err;
17539 
17540 		err = bpf_fetch_kfunc_arg_meta(env, call->imm, call->off, &meta);
17541 		if (err < 0)
17542 			/* error would be reported later */
17543 			return false;
17544 		cs->num_params = btf_type_vlen(meta.func_proto);
17545 		cs->fastcall = meta.kfunc_flags & KF_FASTCALL;
17546 		cs->is_void = btf_type_is_void(btf_type_by_id(meta.btf, meta.func_proto->type));
17547 		return true;
17548 	}
17549 
17550 	return false;
17551 }
17552 
17553 /* LLVM define a bpf_fastcall function attribute.
17554  * This attribute means that function scratches only some of
17555  * the caller saved registers defined by ABI.
17556  * For BPF the set of such registers could be defined as follows:
17557  * - R0 is scratched only if function is non-void;
17558  * - R1-R5 are scratched only if corresponding parameter type is defined
17559  *   in the function prototype.
17560  *
17561  * The contract between kernel and clang allows to simultaneously use
17562  * such functions and maintain backwards compatibility with old
17563  * kernels that don't understand bpf_fastcall calls:
17564  *
17565  * - for bpf_fastcall calls clang allocates registers as-if relevant r0-r5
17566  *   registers are not scratched by the call;
17567  *
17568  * - as a post-processing step, clang visits each bpf_fastcall call and adds
17569  *   spill/fill for every live r0-r5;
17570  *
17571  * - stack offsets used for the spill/fill are allocated as lowest
17572  *   stack offsets in whole function and are not used for any other
17573  *   purposes;
17574  *
17575  * - when kernel loads a program, it looks for such patterns
17576  *   (bpf_fastcall function surrounded by spills/fills) and checks if
17577  *   spill/fill stack offsets are used exclusively in fastcall patterns;
17578  *
17579  * - if so, and if verifier or current JIT inlines the call to the
17580  *   bpf_fastcall function (e.g. a helper call), kernel removes unnecessary
17581  *   spill/fill pairs;
17582  *
17583  * - when old kernel loads a program, presence of spill/fill pairs
17584  *   keeps BPF program valid, albeit slightly less efficient.
17585  *
17586  * For example:
17587  *
17588  *   r1 = 1;
17589  *   r2 = 2;
17590  *   *(u64 *)(r10 - 8)  = r1;            r1 = 1;
17591  *   *(u64 *)(r10 - 16) = r2;            r2 = 2;
17592  *   call %[to_be_inlined]         -->   call %[to_be_inlined]
17593  *   r2 = *(u64 *)(r10 - 16);            r0 = r1;
17594  *   r1 = *(u64 *)(r10 - 8);             r0 += r2;
17595  *   r0 = r1;                            exit;
17596  *   r0 += r2;
17597  *   exit;
17598  *
17599  * The purpose of mark_fastcall_pattern_for_call is to:
17600  * - look for such patterns;
17601  * - mark spill and fill instructions in env->insn_aux_data[*].fastcall_pattern;
17602  * - mark set env->insn_aux_data[*].fastcall_spills_num for call instruction;
17603  * - update env->subprog_info[*]->fastcall_stack_off to find an offset
17604  *   at which bpf_fastcall spill/fill stack slots start;
17605  * - update env->subprog_info[*]->keep_fastcall_stack.
17606  *
17607  * The .fastcall_pattern and .fastcall_stack_off are used by
17608  * check_fastcall_stack_contract() to check if every stack access to
17609  * fastcall spill/fill stack slot originates from spill/fill
17610  * instructions, members of fastcall patterns.
17611  *
17612  * If such condition holds true for a subprogram, fastcall patterns could
17613  * be rewritten by remove_fastcall_spills_fills().
17614  * Otherwise bpf_fastcall patterns are not changed in the subprogram
17615  * (code, presumably, generated by an older clang version).
17616  *
17617  * For example, it is *not* safe to remove spill/fill below:
17618  *
17619  *   r1 = 1;
17620  *   *(u64 *)(r10 - 8)  = r1;            r1 = 1;
17621  *   call %[to_be_inlined]         -->   call %[to_be_inlined]
17622  *   r1 = *(u64 *)(r10 - 8);             r0 = *(u64 *)(r10 - 8);  <---- wrong !!!
17623  *   r0 = *(u64 *)(r10 - 8);             r0 += r1;
17624  *   r0 += r1;                           exit;
17625  *   exit;
17626  */
mark_fastcall_pattern_for_call(struct bpf_verifier_env * env,struct bpf_subprog_info * subprog,int insn_idx,s16 lowest_off)17627 static void mark_fastcall_pattern_for_call(struct bpf_verifier_env *env,
17628 					   struct bpf_subprog_info *subprog,
17629 					   int insn_idx, s16 lowest_off)
17630 {
17631 	struct bpf_insn *insns = env->prog->insnsi, *stx, *ldx;
17632 	struct bpf_insn *call = &env->prog->insnsi[insn_idx];
17633 	u32 clobbered_regs_mask;
17634 	struct bpf_call_summary cs;
17635 	u32 expected_regs_mask;
17636 	s16 off;
17637 	int i;
17638 
17639 	if (!bpf_get_call_summary(env, call, &cs))
17640 		return;
17641 
17642 	/* A bitmask specifying which caller saved registers are clobbered
17643 	 * by a call to a helper/kfunc *as if* this helper/kfunc follows
17644 	 * bpf_fastcall contract:
17645 	 * - includes R0 if function is non-void;
17646 	 * - includes R1-R5 if corresponding parameter has is described
17647 	 *   in the function prototype.
17648 	 */
17649 	clobbered_regs_mask = GENMASK(cs.num_params, cs.is_void ? 1 : 0);
17650 	/* e.g. if helper call clobbers r{0,1}, expect r{2,3,4,5} in the pattern */
17651 	expected_regs_mask = ~clobbered_regs_mask & ALL_CALLER_SAVED_REGS;
17652 
17653 	/* match pairs of form:
17654 	 *
17655 	 * *(u64 *)(r10 - Y) = rX   (where Y % 8 == 0)
17656 	 * ...
17657 	 * call %[to_be_inlined]
17658 	 * ...
17659 	 * rX = *(u64 *)(r10 - Y)
17660 	 */
17661 	for (i = 1, off = lowest_off; i <= ARRAY_SIZE(caller_saved); ++i, off += BPF_REG_SIZE) {
17662 		if (insn_idx - i < 0 || insn_idx + i >= env->prog->len)
17663 			break;
17664 		stx = &insns[insn_idx - i];
17665 		ldx = &insns[insn_idx + i];
17666 		/* must be a stack spill/fill pair */
17667 		if (stx->code != (BPF_STX | BPF_MEM | BPF_DW) ||
17668 		    ldx->code != (BPF_LDX | BPF_MEM | BPF_DW) ||
17669 		    stx->dst_reg != BPF_REG_10 ||
17670 		    ldx->src_reg != BPF_REG_10)
17671 			break;
17672 		/* must be a spill/fill for the same reg */
17673 		if (stx->src_reg != ldx->dst_reg)
17674 			break;
17675 		/* must be one of the previously unseen registers */
17676 		if ((BIT(stx->src_reg) & expected_regs_mask) == 0)
17677 			break;
17678 		/* must be a spill/fill for the same expected offset,
17679 		 * no need to check offset alignment, BPF_DW stack access
17680 		 * is always 8-byte aligned.
17681 		 */
17682 		if (stx->off != off || ldx->off != off)
17683 			break;
17684 		expected_regs_mask &= ~BIT(stx->src_reg);
17685 		env->insn_aux_data[insn_idx - i].fastcall_pattern = 1;
17686 		env->insn_aux_data[insn_idx + i].fastcall_pattern = 1;
17687 	}
17688 	if (i == 1)
17689 		return;
17690 
17691 	/* Conditionally set 'fastcall_spills_num' to allow forward
17692 	 * compatibility when more helper functions are marked as
17693 	 * bpf_fastcall at compile time than current kernel supports, e.g:
17694 	 *
17695 	 *   1: *(u64 *)(r10 - 8) = r1
17696 	 *   2: call A                  ;; assume A is bpf_fastcall for current kernel
17697 	 *   3: r1 = *(u64 *)(r10 - 8)
17698 	 *   4: *(u64 *)(r10 - 8) = r1
17699 	 *   5: call B                  ;; assume B is not bpf_fastcall for current kernel
17700 	 *   6: r1 = *(u64 *)(r10 - 8)
17701 	 *
17702 	 * There is no need to block bpf_fastcall rewrite for such program.
17703 	 * Set 'fastcall_pattern' for both calls to keep check_fastcall_stack_contract() happy,
17704 	 * don't set 'fastcall_spills_num' for call B so that remove_fastcall_spills_fills()
17705 	 * does not remove spill/fill pair {4,6}.
17706 	 */
17707 	if (cs.fastcall)
17708 		env->insn_aux_data[insn_idx].fastcall_spills_num = i - 1;
17709 	else
17710 		subprog->keep_fastcall_stack = 1;
17711 	subprog->fastcall_stack_off = min(subprog->fastcall_stack_off, off);
17712 }
17713 
mark_fastcall_patterns(struct bpf_verifier_env * env)17714 static int mark_fastcall_patterns(struct bpf_verifier_env *env)
17715 {
17716 	struct bpf_subprog_info *subprog = env->subprog_info;
17717 	struct bpf_insn *insn;
17718 	s16 lowest_off;
17719 	int s, i;
17720 
17721 	for (s = 0; s < env->subprog_cnt; ++s, ++subprog) {
17722 		/* find lowest stack spill offset used in this subprog */
17723 		lowest_off = 0;
17724 		for (i = subprog->start; i < (subprog + 1)->start; ++i) {
17725 			insn = env->prog->insnsi + i;
17726 			if (insn->code != (BPF_STX | BPF_MEM | BPF_DW) ||
17727 			    insn->dst_reg != BPF_REG_10)
17728 				continue;
17729 			lowest_off = min(lowest_off, insn->off);
17730 		}
17731 		/* use this offset to find fastcall patterns */
17732 		for (i = subprog->start; i < (subprog + 1)->start; ++i) {
17733 			insn = env->prog->insnsi + i;
17734 			if (insn->code != (BPF_JMP | BPF_CALL))
17735 				continue;
17736 			mark_fastcall_pattern_for_call(env, subprog, i, lowest_off);
17737 		}
17738 	}
17739 	return 0;
17740 }
17741 
adjust_btf_func(struct bpf_verifier_env * env)17742 static void adjust_btf_func(struct bpf_verifier_env *env)
17743 {
17744 	struct bpf_prog_aux *aux = env->prog->aux;
17745 	int i;
17746 
17747 	if (!aux->func_info)
17748 		return;
17749 
17750 	/* func_info is not available for hidden subprogs */
17751 	for (i = 0; i < env->subprog_cnt - env->hidden_subprog_cnt; i++)
17752 		aux->func_info[i].insn_off = env->subprog_info[i].start;
17753 }
17754 
17755 /* Find id in idset and increment its count, or add new entry */
idset_cnt_inc(struct bpf_idset * idset,u32 id)17756 static void idset_cnt_inc(struct bpf_idset *idset, u32 id)
17757 {
17758 	u32 i;
17759 
17760 	for (i = 0; i < idset->num_ids; i++) {
17761 		if (idset->entries[i].id == id) {
17762 			idset->entries[i].cnt++;
17763 			return;
17764 		}
17765 	}
17766 	/* New id */
17767 	if (idset->num_ids < BPF_ID_MAP_SIZE) {
17768 		idset->entries[idset->num_ids].id = id;
17769 		idset->entries[idset->num_ids].cnt = 1;
17770 		idset->num_ids++;
17771 	}
17772 }
17773 
17774 /* Find id in idset and return its count, or 0 if not found */
idset_cnt_get(struct bpf_idset * idset,u32 id)17775 static u32 idset_cnt_get(struct bpf_idset *idset, u32 id)
17776 {
17777 	u32 i;
17778 
17779 	for (i = 0; i < idset->num_ids; i++) {
17780 		if (idset->entries[i].id == id)
17781 			return idset->entries[i].cnt;
17782 	}
17783 	return 0;
17784 }
17785 
17786 /*
17787  * Clear singular scalar ids in a state.
17788  * A register with a non-zero id is called singular if no other register shares
17789  * the same base id. Such registers can be treated as independent (id=0).
17790  */
bpf_clear_singular_ids(struct bpf_verifier_env * env,struct bpf_verifier_state * st)17791 void bpf_clear_singular_ids(struct bpf_verifier_env *env,
17792 			    struct bpf_verifier_state *st)
17793 {
17794 	struct bpf_idset *idset = &env->idset_scratch;
17795 	struct bpf_func_state *func;
17796 	struct bpf_reg_state *reg;
17797 
17798 	idset->num_ids = 0;
17799 
17800 	bpf_for_each_reg_in_vstate(st, func, reg, ({
17801 		if (reg->type != SCALAR_VALUE)
17802 			continue;
17803 		if (!reg->id)
17804 			continue;
17805 		idset_cnt_inc(idset, reg->id & ~BPF_ADD_CONST);
17806 	}));
17807 
17808 	bpf_for_each_reg_in_vstate(st, func, reg, ({
17809 		if (reg->type != SCALAR_VALUE)
17810 			continue;
17811 		if (!reg->id)
17812 			continue;
17813 		if (idset_cnt_get(idset, reg->id & ~BPF_ADD_CONST) == 1)
17814 			clear_scalar_id(reg);
17815 	}));
17816 }
17817 
17818 /* Return true if it's OK to have the same insn return a different type. */
reg_type_mismatch_ok(enum bpf_reg_type type)17819 static bool reg_type_mismatch_ok(enum bpf_reg_type type)
17820 {
17821 	switch (base_type(type)) {
17822 	case PTR_TO_CTX:
17823 	case PTR_TO_SOCKET:
17824 	case PTR_TO_SOCK_COMMON:
17825 	case PTR_TO_TCP_SOCK:
17826 	case PTR_TO_XDP_SOCK:
17827 	case PTR_TO_BTF_ID:
17828 	case PTR_TO_ARENA:
17829 		return false;
17830 	case PTR_TO_MEM:
17831 		return !bpf_may_fault_on_deref(type);
17832 	default:
17833 		return true;
17834 	}
17835 }
17836 
17837 /* If an instruction was previously used with particular pointer types, then we
17838  * need to be careful to avoid cases such as the below, where it may be ok
17839  * for one branch accessing the pointer, but not ok for the other branch:
17840  *
17841  * R1 = sock_ptr
17842  * goto X;
17843  * ...
17844  * R1 = some_other_valid_ptr;
17845  * goto X;
17846  * ...
17847  * R2 = *(u32 *)(R1 + 0);
17848  */
reg_type_mismatch(enum bpf_reg_type src,enum bpf_reg_type prev)17849 static bool reg_type_mismatch(enum bpf_reg_type src, enum bpf_reg_type prev)
17850 {
17851 	return src != prev && (!reg_type_mismatch_ok(src) ||
17852 			       !reg_type_mismatch_ok(prev));
17853 }
17854 
is_ptr_to_mem(enum bpf_reg_type type)17855 static bool is_ptr_to_mem(enum bpf_reg_type type)
17856 {
17857 	return base_type(type) == PTR_TO_MEM;
17858 }
17859 
merge_ptr_types(enum bpf_reg_type type_a,enum bpf_reg_type type_b)17860 static enum bpf_reg_type merge_ptr_types(enum bpf_reg_type type_a,
17861 					 enum bpf_reg_type type_b)
17862 {
17863 	bool to_mem = is_ptr_to_mem(type_a) || is_ptr_to_mem(type_b);
17864 	enum bpf_reg_type type_merged = to_mem ? PTR_TO_MEM : PTR_TO_BTF_ID;
17865 
17866 	if (bpf_may_fault_on_deref(type_a) || bpf_may_fault_on_deref(type_b))
17867 		type_merged |= to_mem ? MEM_RDONLY | PTR_UNTRUSTED :
17868 					PTR_UNTRUSTED;
17869 	else
17870 		type_merged |= ((type_a | type_b) & MEM_RDONLY);
17871 	return type_merged;
17872 }
17873 
save_aux_ptr_type(struct bpf_verifier_env * env,enum bpf_reg_type type,bool allow_trust_mismatch)17874 static int save_aux_ptr_type(struct bpf_verifier_env *env, enum bpf_reg_type type,
17875 			     bool allow_trust_mismatch)
17876 {
17877 	enum bpf_reg_type *prev_type = &env->insn_aux_data[env->insn_idx].ptr_type;
17878 
17879 	if (*prev_type == NOT_INIT) {
17880 		/* Saw a valid insn
17881 		 * dst_reg = *(u32 *)(src_reg + off)
17882 		 * save type to validate intersecting paths
17883 		 */
17884 		*prev_type = type;
17885 	} else if (reg_type_mismatch(type, *prev_type)) {
17886 		/* Abuser program is trying to use the same insn
17887 		 * dst_reg = *(u32*) (src_reg + off)
17888 		 * with different pointer types:
17889 		 * src_reg == ctx in one branch and
17890 		 * src_reg == stack|map in some other branch.
17891 		 * Reject it.
17892 		 */
17893 		if (allow_trust_mismatch &&
17894 		    bpf_is_ptr_to_mem_or_btf_id(type) &&
17895 		    bpf_is_ptr_to_mem_or_btf_id(*prev_type)) {
17896 			/*
17897 			 * Have to support a use case when one path through
17898 			 * the program yields a TRUSTED pointer while another
17899 			 * is UNTRUSTED. Merge them into a type which keeps
17900 			 * the BPF_PROBE_MEM/BPF_PROBE_MEMSX rewrite when
17901 			 * either side needs it.
17902 			 */
17903 			*prev_type = merge_ptr_types(type, *prev_type);
17904 		} else {
17905 			verbose(env, "same insn cannot be used with different pointers\n");
17906 			return -EINVAL;
17907 		}
17908 	}
17909 
17910 	return 0;
17911 }
17912 
17913 enum {
17914 	PROCESS_BPF_EXIT = 1,
17915 	INSN_IDX_UPDATED = 2,
17916 };
17917 
process_bpf_exit_full(struct bpf_verifier_env * env,bool * do_print_state,bool exception_exit)17918 static int process_bpf_exit_full(struct bpf_verifier_env *env,
17919 				 bool *do_print_state,
17920 				 bool exception_exit)
17921 {
17922 	struct bpf_func_state *cur_frame = cur_func(env);
17923 
17924 	/* We must do check_reference_leak here before
17925 	 * prepare_func_exit to handle the case when
17926 	 * state->curframe > 0, it may be a callback function,
17927 	 * for which reference_state must match caller reference
17928 	 * state when it exits.
17929 	 */
17930 	int err = check_resource_leak(env, exception_exit,
17931 				      exception_exit || !env->cur_state->curframe,
17932 				      exception_exit ? "bpf_throw" :
17933 				      "BPF_EXIT instruction in main prog");
17934 	if (err)
17935 		return err;
17936 
17937 	/* The side effect of the prepare_func_exit which is
17938 	 * being skipped is that it frees bpf_func_state.
17939 	 * Typically, process_bpf_exit will only be hit with
17940 	 * outermost exit. copy_verifier_state in pop_stack will
17941 	 * handle freeing of any extra bpf_func_state left over
17942 	 * from not processing all nested function exits. We
17943 	 * also skip return code checks as they are not needed
17944 	 * for exceptional exits.
17945 	 */
17946 	if (exception_exit)
17947 		return PROCESS_BPF_EXIT;
17948 
17949 	if (env->cur_state->curframe) {
17950 		/* exit from nested function */
17951 		err = prepare_func_exit(env, &env->insn_idx);
17952 		if (err)
17953 			return err;
17954 		*do_print_state = true;
17955 		return INSN_IDX_UPDATED;
17956 	}
17957 
17958 	/*
17959 	 * Return from a regular global subprogram differs from return
17960 	 * from the main program or async/exception callback.
17961 	 * Main program exit implies return code restrictions
17962 	 * that depend on program type.
17963 	 * Exit from exception callback is equivalent to main program exit.
17964 	 * Exit from async callback implies return code restrictions
17965 	 * that depend on async scheduling mechanism.
17966 	 */
17967 	if (cur_frame->subprogno &&
17968 	    !cur_frame->in_async_callback_fn &&
17969 	    !cur_frame->in_exception_callback_fn)
17970 		err = check_global_subprog_return_code(env);
17971 	else
17972 		err = check_return_code(env, BPF_REG_0, "R0");
17973 	if (err)
17974 		return err;
17975 	return PROCESS_BPF_EXIT;
17976 }
17977 
indirect_jump_min_max_index(struct bpf_verifier_env * env,int regno,struct bpf_map * map,u32 * pmin_index,u32 * pmax_index)17978 static int indirect_jump_min_max_index(struct bpf_verifier_env *env,
17979 				       int regno,
17980 				       struct bpf_map *map,
17981 				       u32 *pmin_index, u32 *pmax_index)
17982 {
17983 	struct bpf_reg_state *reg = reg_state(env, regno);
17984 	u64 min_index = reg_umin(reg);
17985 	u64 max_index = reg_umax(reg);
17986 	const u32 size = 8;
17987 
17988 	if (min_index > (u64) U32_MAX * size) {
17989 		verbose(env, "the sum of R%u umin_value %llu is too big\n", regno, reg_umin(reg));
17990 		return -ERANGE;
17991 	}
17992 	if (max_index > (u64) U32_MAX * size) {
17993 		verbose(env, "the sum of R%u umax_value %llu is too big\n", regno, reg_umax(reg));
17994 		return -ERANGE;
17995 	}
17996 
17997 	min_index /= size;
17998 	max_index /= size;
17999 
18000 	if (max_index >= map->max_entries) {
18001 		verbose(env, "R%u points to outside of jump table: [%llu,%llu] max_entries %u\n",
18002 			     regno, min_index, max_index, map->max_entries);
18003 		return -EINVAL;
18004 	}
18005 
18006 	*pmin_index = min_index;
18007 	*pmax_index = max_index;
18008 	return 0;
18009 }
18010 
18011 /* gotox *dst_reg */
check_indirect_jump(struct bpf_verifier_env * env,struct bpf_insn * insn)18012 static int check_indirect_jump(struct bpf_verifier_env *env, struct bpf_insn *insn)
18013 {
18014 	struct bpf_verifier_state *other_branch;
18015 	struct bpf_reg_state *dst_reg;
18016 	struct bpf_map *map;
18017 	u32 min_index, max_index;
18018 	int err = 0;
18019 	int n;
18020 	int i;
18021 
18022 	dst_reg = reg_state(env, insn->dst_reg);
18023 	if (dst_reg->type != PTR_TO_INSN) {
18024 		verbose(env, "R%d has type %s, expected PTR_TO_INSN\n",
18025 			     insn->dst_reg, reg_type_str(env, dst_reg->type));
18026 		return -EINVAL;
18027 	}
18028 
18029 	map = dst_reg->map_ptr;
18030 	if (verifier_bug_if(!map, env, "R%d has an empty map pointer", insn->dst_reg))
18031 		return -EFAULT;
18032 
18033 	if (verifier_bug_if(map->map_type != BPF_MAP_TYPE_INSN_ARRAY, env,
18034 			    "R%d has incorrect map type %d", insn->dst_reg, map->map_type))
18035 		return -EFAULT;
18036 
18037 	err = indirect_jump_min_max_index(env, insn->dst_reg, map, &min_index, &max_index);
18038 	if (err)
18039 		return err;
18040 
18041 	/* Ensure that the buffer is large enough */
18042 	if (!env->gotox_tmp_buf || env->gotox_tmp_buf->cnt < max_index - min_index + 1) {
18043 		env->gotox_tmp_buf = bpf_iarray_realloc(env->gotox_tmp_buf,
18044 						        max_index - min_index + 1);
18045 		if (!env->gotox_tmp_buf)
18046 			return -ENOMEM;
18047 	}
18048 
18049 	n = bpf_copy_insn_array_uniq(map, min_index, max_index, env->gotox_tmp_buf->items);
18050 	if (n < 0)
18051 		return n;
18052 	if (n == 0) {
18053 		verbose(env, "register R%d doesn't point to any offset in map id=%d\n",
18054 			     insn->dst_reg, map->id);
18055 		return -EINVAL;
18056 	}
18057 
18058 	for (i = 0; i < n - 1; i++) {
18059 		mark_indirect_target(env, env->gotox_tmp_buf->items[i]);
18060 		other_branch = push_stack(env, env->gotox_tmp_buf->items[i],
18061 					  env->insn_idx, env->cur_state->speculative);
18062 		if (IS_ERR(other_branch))
18063 			return PTR_ERR(other_branch);
18064 	}
18065 	env->insn_idx = env->gotox_tmp_buf->items[n-1];
18066 	mark_indirect_target(env, env->insn_idx);
18067 	return INSN_IDX_UPDATED;
18068 }
18069 
do_check_insn(struct bpf_verifier_env * env,bool * do_print_state)18070 static int do_check_insn(struct bpf_verifier_env *env, bool *do_print_state)
18071 {
18072 	int err;
18073 	struct bpf_insn *insn = &env->prog->insnsi[env->insn_idx];
18074 	u8 class = BPF_CLASS(insn->code);
18075 
18076 	switch (class) {
18077 	case BPF_ALU:
18078 	case BPF_ALU64:
18079 		return check_alu_op(env, insn);
18080 
18081 	case BPF_LDX:
18082 		return check_load_mem(env, insn, false,
18083 				      BPF_MODE(insn->code) == BPF_MEMSX,
18084 				      true, "ldx");
18085 
18086 	case BPF_STX:
18087 		if (BPF_MODE(insn->code) == BPF_ATOMIC)
18088 			return check_atomic(env, insn);
18089 		return check_store_reg(env, insn, false);
18090 
18091 	case BPF_ST: {
18092 		/* Handle stack arg write (store immediate) */
18093 		if (is_stack_arg_st(insn)) {
18094 			struct bpf_verifier_state *vstate = env->cur_state;
18095 			struct bpf_func_state *state = vstate->frame[vstate->curframe];
18096 
18097 			return check_stack_arg_write(env, state, insn->off, NULL);
18098 		}
18099 
18100 		enum bpf_reg_type dst_reg_type;
18101 
18102 		err = check_reg_arg(env, insn->dst_reg, SRC_OP);
18103 		if (err)
18104 			return err;
18105 
18106 		dst_reg_type = cur_regs(env)[insn->dst_reg].type;
18107 
18108 		err = check_mem_access(env, env->insn_idx, cur_regs(env) + insn->dst_reg, argno_from_reg(insn->dst_reg),
18109 				       insn->off, BPF_SIZE(insn->code),
18110 				       BPF_WRITE, -1, false, false);
18111 		if (err)
18112 			return err;
18113 
18114 		return save_aux_ptr_type(env, dst_reg_type, false);
18115 	}
18116 	case BPF_JMP:
18117 	case BPF_JMP32: {
18118 		u8 opcode = BPF_OP(insn->code);
18119 
18120 		env->jmps_processed++;
18121 		if (opcode == BPF_CALL) {
18122 			if (env->cur_state->active_locks) {
18123 				if ((insn->src_reg == BPF_REG_0 &&
18124 				     insn->imm != BPF_FUNC_spin_unlock &&
18125 				     insn->imm != BPF_FUNC_kptr_xchg) ||
18126 				    (insn->src_reg == BPF_PSEUDO_KFUNC_CALL &&
18127 				     !kfunc_spin_allowed(env, insn->imm, insn->off))) {
18128 					verbose(env,
18129 						"function calls are not allowed while holding a lock\n");
18130 					bpf_diag_ctx_active(
18131 						env, env->insn_idx,
18132 						"function call", BPF_DIAG_CONTEXT_LOCK,
18133 						"Release the BPF spin lock before making this call, or move the call outside the locked region.");
18134 					return -EINVAL;
18135 				}
18136 			}
18137 			mark_reg_scratched(env, BPF_REG_0);
18138 			if (bpf_in_stack_arg_cnt(&env->subprog_info[cur_func(env)->subprogno]))
18139 				cur_func(env)->no_stack_arg_load = true;
18140 			if (insn->src_reg == BPF_PSEUDO_CALL)
18141 				return check_func_call(env, insn, &env->insn_idx);
18142 			if (insn->src_reg == BPF_PSEUDO_KFUNC_CALL)
18143 				return check_kfunc_call(env, insn, &env->insn_idx);
18144 			return check_helper_call(env, insn, &env->insn_idx);
18145 		} else if (opcode == BPF_JA) {
18146 			if (BPF_SRC(insn->code) == BPF_X)
18147 				return check_indirect_jump(env, insn);
18148 
18149 			if (class == BPF_JMP)
18150 				env->insn_idx += insn->off + 1;
18151 			else
18152 				env->insn_idx += insn->imm + 1;
18153 			return INSN_IDX_UPDATED;
18154 		} else if (opcode == BPF_EXIT) {
18155 			return process_bpf_exit_full(env, do_print_state, false);
18156 		}
18157 		return check_cond_jmp_op(env, insn, &env->insn_idx);
18158 	}
18159 	case BPF_LD: {
18160 		u8 mode = BPF_MODE(insn->code);
18161 
18162 		if (mode == BPF_ABS || mode == BPF_IND)
18163 			return check_ld_abs(env, insn);
18164 
18165 		if (mode == BPF_IMM) {
18166 			err = check_ld_imm(env, insn);
18167 			if (err)
18168 				return err;
18169 
18170 			env->insn_idx++;
18171 			sanitize_mark_insn_seen(env);
18172 		}
18173 		return 0;
18174 	}
18175 	}
18176 	/* all class values are handled above. silence compiler warning */
18177 	return -EFAULT;
18178 }
18179 
do_check(struct bpf_verifier_env * env)18180 static int do_check(struct bpf_verifier_env *env)
18181 {
18182 	bool pop_log = !(env->log.level & BPF_LOG_LEVEL2);
18183 	struct bpf_verifier_state *state = env->cur_state;
18184 	struct bpf_insn *insns = env->prog->insnsi;
18185 	int insn_cnt = env->prog->len;
18186 	bool do_print_state = false;
18187 	int prev_insn_idx = -1;
18188 
18189 	for (;;) {
18190 		struct bpf_insn *insn;
18191 		struct bpf_insn_aux_data *insn_aux;
18192 		int err;
18193 
18194 		/* reset current history entry on each new instruction */
18195 		env->cur_hist_ent = NULL;
18196 
18197 		env->prev_insn_idx = prev_insn_idx;
18198 		if (env->insn_idx >= insn_cnt) {
18199 			verbose(env, "invalid insn idx %d insn_cnt %d\n",
18200 				env->insn_idx, insn_cnt);
18201 			return -EFAULT;
18202 		}
18203 
18204 		insn = &insns[env->insn_idx];
18205 		insn_aux = &env->insn_aux_data[env->insn_idx];
18206 
18207 		account_processed_insn(env);
18208 
18209 		if (env->insn_processed > BPF_COMPLEXITY_LIMIT_INSNS) {
18210 			verbose(env,
18211 				"BPF program is too large. Processed %d insn\n",
18212 				env->insn_processed);
18213 			return -E2BIG;
18214 		}
18215 
18216 		state->last_insn_idx = env->prev_insn_idx;
18217 		state->insn_idx = env->insn_idx;
18218 		/*
18219 		 * Record the incoming edge so active and queued paths use the same
18220 		 * branch-recording path. A zero-offset conditional has identical
18221 		 * successors, so its outcome cannot be reconstructed from the edge.
18222 		 */
18223 		if (!state->speculative && prev_insn_idx >= 0 && prev_insn_idx < insn_cnt) {
18224 			struct bpf_insn *prev_insn = &insns[prev_insn_idx];
18225 			int fallthrough_idx = prev_insn_idx + 1;
18226 			int branch_idx = prev_insn_idx + bpf_jmp_offset(prev_insn) + 1;
18227 			u8 class = BPF_CLASS(prev_insn->code);
18228 			u8 opcode = BPF_OP(prev_insn->code);
18229 
18230 			if ((class == BPF_JMP || class == BPF_JMP32) &&
18231 			    opcode != BPF_JA && opcode != BPF_CALL && opcode != BPF_EXIT &&
18232 			    opcode <= BPF_JCOND && branch_idx != fallthrough_idx) {
18233 				if (env->insn_idx == branch_idx)
18234 					bpf_diag_record_branch(env, prev_insn_idx, true);
18235 				else if (env->insn_idx == fallthrough_idx)
18236 					bpf_diag_record_branch(env, prev_insn_idx, false);
18237 			}
18238 		}
18239 
18240 		if (bpf_is_prune_point(env, env->insn_idx)) {
18241 			err = bpf_is_state_visited(env, env->insn_idx);
18242 			if (err < 0)
18243 				return err;
18244 			if (err == 1) {
18245 				/* found equivalent state, can prune the search */
18246 				if (env->log.level & BPF_LOG_LEVEL) {
18247 					if (do_print_state)
18248 						verbose(env, "\nfrom %d to %d%s: safe\n",
18249 							env->prev_insn_idx, env->insn_idx,
18250 							env->cur_state->speculative ?
18251 							" (speculative execution)" : "");
18252 					else
18253 						verbose(env, "%d: safe\n", env->insn_idx);
18254 				}
18255 				goto process_bpf_exit;
18256 			}
18257 		}
18258 
18259 		if (bpf_is_jmp_point(env, env->insn_idx)) {
18260 			err = bpf_push_jmp_history(env, state, 0, 0, 0, 0);
18261 			if (err)
18262 				return err;
18263 		}
18264 
18265 		if (signal_pending(current))
18266 			return -EAGAIN;
18267 
18268 		if (need_resched())
18269 			cond_resched();
18270 
18271 		if (env->log.level & BPF_LOG_LEVEL2 && do_print_state) {
18272 			verbose(env, "\nfrom %d to %d%s:",
18273 				env->prev_insn_idx, env->insn_idx,
18274 				env->cur_state->speculative ?
18275 				" (speculative execution)" : "");
18276 			print_verifier_state(env, state, state->curframe, true);
18277 			do_print_state = false;
18278 		}
18279 
18280 		if (env->log.level & BPF_LOG_LEVEL) {
18281 			if (verifier_state_scratched(env))
18282 				print_insn_state(env, state, state->curframe);
18283 
18284 			verbose_linfo(env, env->insn_idx, "; ");
18285 			env->prev_log_pos = env->log.end_pos;
18286 			verbose(env, "%d: ", env->insn_idx);
18287 			bpf_verbose_insn(env, insn);
18288 			verbose(env, "\n");
18289 			env->prev_insn_print_pos = env->log.end_pos - env->prev_log_pos;
18290 			env->prev_log_pos = env->log.end_pos;
18291 		}
18292 
18293 		if (bpf_prog_is_offloaded(env->prog->aux)) {
18294 			err = bpf_prog_offload_verify_insn(env, env->insn_idx,
18295 							   env->prev_insn_idx);
18296 			if (err)
18297 				return err;
18298 		}
18299 
18300 		sanitize_mark_insn_seen(env);
18301 		prev_insn_idx = env->insn_idx;
18302 
18303 		/* Sanity check: precomputed constants must match verifier state */
18304 		if (!state->speculative && insn_aux->const_reg_mask) {
18305 			struct bpf_reg_state *regs = cur_regs(env);
18306 			u16 mask = insn_aux->const_reg_mask;
18307 
18308 			for (int r = 0; r < ARRAY_SIZE(insn_aux->const_reg_vals); r++) {
18309 				u32 cval = insn_aux->const_reg_vals[r];
18310 
18311 				if (!(mask & BIT(r)))
18312 					continue;
18313 				if (regs[r].type != SCALAR_VALUE)
18314 					continue;
18315 				if (!tnum_is_const(regs[r].var_off))
18316 					continue;
18317 				if (verifier_bug_if((u32)regs[r].var_off.value != cval,
18318 						    env, "const R%d: %u != %llu",
18319 						    r, cval, regs[r].var_off.value))
18320 					return -EFAULT;
18321 			}
18322 		}
18323 
18324 		/* Reduce verification complexity by stopping speculative path
18325 		 * verification when a nospec is encountered.
18326 		 */
18327 		if (state->speculative && insn_aux->nospec)
18328 			goto process_bpf_exit;
18329 
18330 		err = do_check_insn(env, &do_print_state);
18331 		if (error_recoverable_with_nospec(err) && state->speculative) {
18332 			/* Prevent this speculative path from ever reaching the
18333 			 * insn that would have been unsafe to execute.
18334 			 */
18335 			insn_aux->nospec = true;
18336 			/* If it was an ADD/SUB insn, potentially remove any
18337 			 * markings for alu sanitization.
18338 			 */
18339 			insn_aux->alu_state = 0;
18340 			goto process_bpf_exit;
18341 		} else if (err < 0) {
18342 			return err;
18343 		} else if (err == PROCESS_BPF_EXIT) {
18344 			goto process_bpf_exit;
18345 		} else if (err == INSN_IDX_UPDATED) {
18346 		} else if (err == 0) {
18347 			env->insn_idx++;
18348 		}
18349 
18350 		if (state->speculative && insn_aux->nospec_result) {
18351 			/* If we are on a path that performed a jump-op, this
18352 			 * may skip a nospec patched-in after the jump. This can
18353 			 * currently never happen because nospec_result is only
18354 			 * used for the write-ops
18355 			 * `*(size*)(dst_reg+off)=src_reg|imm32` and helper
18356 			 * calls. These must never skip the following insn
18357 			 * (i.e., bpf_insn_successors()'s opcode_info.can_jump
18358 			 * is false). Still, add a warning to document this in
18359 			 * case nospec_result is used elsewhere in the future.
18360 			 *
18361 			 * All non-branch instructions have a single
18362 			 * fall-through edge. For these, nospec_result should
18363 			 * already work.
18364 			 */
18365 			if (verifier_bug_if((BPF_CLASS(insn->code) == BPF_JMP ||
18366 					     BPF_CLASS(insn->code) == BPF_JMP32) &&
18367 					    BPF_OP(insn->code) != BPF_CALL, env,
18368 					    "speculation barrier after jump instruction may not have the desired effect"))
18369 				return -EFAULT;
18370 process_bpf_exit:
18371 			account_current_path(env);
18372 			mark_verifier_state_scratched(env);
18373 			err = bpf_update_branch_counts(env, env->cur_state);
18374 			if (err)
18375 				return err;
18376 			err = pop_stack(env, &prev_insn_idx, &env->insn_idx,
18377 					pop_log);
18378 			if (err < 0) {
18379 				if (err != -ENOENT)
18380 					return err;
18381 				break;
18382 			} else {
18383 				do_print_state = true;
18384 				continue;
18385 			}
18386 		}
18387 	}
18388 
18389 	return 0;
18390 }
18391 
find_btf_percpu_datasec(struct btf * btf)18392 static int find_btf_percpu_datasec(struct btf *btf)
18393 {
18394 	const struct btf_type *t;
18395 	const char *tname;
18396 	int i, n;
18397 
18398 	/*
18399 	 * Both vmlinux and module each have their own ".data..percpu"
18400 	 * DATASECs in BTF. So for module's case, we need to skip vmlinux BTF
18401 	 * types to look at only module's own BTF types.
18402 	 */
18403 	n = btf_nr_types(btf);
18404 	for (i = btf_named_start_id(btf, true); i < n; i++) {
18405 		t = btf_type_by_id(btf, i);
18406 		if (BTF_INFO_KIND(t->info) != BTF_KIND_DATASEC)
18407 			continue;
18408 
18409 		tname = btf_name_by_offset(btf, t->name_off);
18410 		if (!strcmp(tname, ".data..percpu"))
18411 			return i;
18412 	}
18413 
18414 	return -ENOENT;
18415 }
18416 
18417 /*
18418  * Add btf to the env->used_btfs array. If needed, refcount the
18419  * corresponding kernel module. To simplify caller's logic
18420  * in case of error or if btf was added before the function
18421  * decreases the btf refcount.
18422  */
__add_used_btf(struct bpf_verifier_env * env,struct btf * btf)18423 static int __add_used_btf(struct bpf_verifier_env *env, struct btf *btf)
18424 {
18425 	struct btf_mod_pair *btf_mod;
18426 	int ret = 0;
18427 	int i;
18428 
18429 	/* check whether we recorded this BTF (and maybe module) already */
18430 	for (i = 0; i < env->used_btf_cnt; i++)
18431 		if (env->used_btfs[i].btf == btf)
18432 			goto ret_put;
18433 
18434 	if (env->signature) {
18435 		verbose(env, "signed program cannot bind any BTF\n");
18436 		ret = -EACCES;
18437 		goto ret_put;
18438 	}
18439 	if (env->used_btf_cnt >= MAX_USED_BTFS) {
18440 		verbose(env, "The total number of btfs per program has reached the limit of %u\n",
18441 			MAX_USED_BTFS);
18442 		ret = -E2BIG;
18443 		goto ret_put;
18444 	}
18445 
18446 	btf_mod = &env->used_btfs[env->used_btf_cnt];
18447 	btf_mod->btf = btf;
18448 	btf_mod->module = NULL;
18449 
18450 	/* if we reference variables from kernel module, bump its refcount */
18451 	if (btf_is_module(btf)) {
18452 		btf_mod->module = btf_try_get_module(btf);
18453 		if (!btf_mod->module) {
18454 			ret = -ENXIO;
18455 			goto ret_put;
18456 		}
18457 	}
18458 
18459 	env->used_btf_cnt++;
18460 	return 0;
18461 
18462 ret_put:
18463 	/* Either error or this BTF was already added */
18464 	btf_put(btf);
18465 	return ret;
18466 }
18467 
18468 /* replace pseudo btf_id with kernel symbol address */
__check_pseudo_btf_id(struct bpf_verifier_env * env,struct bpf_insn * insn,struct bpf_insn_aux_data * aux,struct btf * btf)18469 static int __check_pseudo_btf_id(struct bpf_verifier_env *env,
18470 				 struct bpf_insn *insn,
18471 				 struct bpf_insn_aux_data *aux,
18472 				 struct btf *btf)
18473 {
18474 	const struct btf_var_secinfo *vsi;
18475 	const struct btf_type *datasec;
18476 	const struct btf_type *t;
18477 	const char *sym_name;
18478 	bool percpu = false;
18479 	u32 type, id = insn->imm;
18480 	s32 datasec_id;
18481 	u64 addr;
18482 	int i;
18483 
18484 	t = btf_type_by_id(btf, id);
18485 	if (!t) {
18486 		verbose(env, "ldimm64 insn specifies invalid btf_id %d.\n", id);
18487 		return -ENOENT;
18488 	}
18489 
18490 	if (!btf_type_is_var(t) && !btf_type_is_func(t)) {
18491 		verbose(env, "pseudo btf_id %d in ldimm64 isn't KIND_VAR or KIND_FUNC\n", id);
18492 		return -EINVAL;
18493 	}
18494 
18495 	sym_name = btf_name_by_offset(btf, t->name_off);
18496 	addr = kallsyms_lookup_name(sym_name);
18497 	if (!addr) {
18498 		verbose(env, "ldimm64 failed to find the address for kernel symbol '%s'.\n",
18499 			sym_name);
18500 		return -ENOENT;
18501 	}
18502 	insn[0].imm = (u32)addr;
18503 	insn[1].imm = addr >> 32;
18504 
18505 	if (btf_type_is_func(t)) {
18506 		aux->btf_var.reg_type = PTR_TO_MEM | MEM_RDONLY;
18507 		aux->btf_var.mem_size = 0;
18508 		return 0;
18509 	}
18510 
18511 	datasec_id = find_btf_percpu_datasec(btf);
18512 	if (datasec_id > 0) {
18513 		datasec = btf_type_by_id(btf, datasec_id);
18514 		for_each_vsi(i, datasec, vsi) {
18515 			if (vsi->type == id) {
18516 				percpu = true;
18517 				break;
18518 			}
18519 		}
18520 	}
18521 
18522 	type = t->type;
18523 	t = btf_type_skip_modifiers(btf, type, NULL);
18524 	if (percpu) {
18525 		aux->btf_var.reg_type = PTR_TO_BTF_ID | MEM_PERCPU;
18526 		aux->btf_var.btf = btf;
18527 		aux->btf_var.btf_id = type;
18528 	} else if (!btf_type_is_struct(t)) {
18529 		const struct btf_type *ret;
18530 		const char *tname;
18531 		u32 tsize;
18532 
18533 		/* resolve the type size of ksym. */
18534 		ret = btf_resolve_size(btf, t, &tsize);
18535 		if (IS_ERR(ret)) {
18536 			tname = btf_name_by_offset(btf, t->name_off);
18537 			verbose(env, "ldimm64 unable to resolve the size of type '%s': %ld\n",
18538 				tname, PTR_ERR(ret));
18539 			return -EINVAL;
18540 		}
18541 		aux->btf_var.reg_type = PTR_TO_MEM | MEM_RDONLY;
18542 		aux->btf_var.mem_size = tsize;
18543 	} else {
18544 		aux->btf_var.reg_type = PTR_TO_BTF_ID;
18545 		aux->btf_var.btf = btf;
18546 		aux->btf_var.btf_id = type;
18547 	}
18548 
18549 	return 0;
18550 }
18551 
check_pseudo_btf_id(struct bpf_verifier_env * env,struct bpf_insn * insn,struct bpf_insn_aux_data * aux)18552 static int check_pseudo_btf_id(struct bpf_verifier_env *env,
18553 			       struct bpf_insn *insn,
18554 			       struct bpf_insn_aux_data *aux)
18555 {
18556 	struct btf *btf;
18557 	int btf_fd;
18558 	int err;
18559 
18560 	btf_fd = insn[1].imm;
18561 	if (btf_fd) {
18562 		btf = btf_get_by_fd(btf_fd);
18563 		if (IS_ERR(btf)) {
18564 			verbose(env, "invalid module BTF object FD specified.\n");
18565 			return -EINVAL;
18566 		}
18567 	} else {
18568 		if (!btf_vmlinux) {
18569 			verbose(env, "kernel is missing BTF, make sure CONFIG_DEBUG_INFO_BTF=y is specified in Kconfig.\n");
18570 			return -EINVAL;
18571 		}
18572 		btf_get(btf_vmlinux);
18573 		btf = btf_vmlinux;
18574 	}
18575 
18576 	err = __check_pseudo_btf_id(env, insn, aux, btf);
18577 	if (err) {
18578 		btf_put(btf);
18579 		return err;
18580 	}
18581 
18582 	return __add_used_btf(env, btf);
18583 }
18584 
is_tracing_prog_type(enum bpf_prog_type type)18585 static bool is_tracing_prog_type(enum bpf_prog_type type)
18586 {
18587 	switch (type) {
18588 	case BPF_PROG_TYPE_KPROBE:
18589 	case BPF_PROG_TYPE_TRACEPOINT:
18590 	case BPF_PROG_TYPE_PERF_EVENT:
18591 	case BPF_PROG_TYPE_RAW_TRACEPOINT:
18592 	case BPF_PROG_TYPE_RAW_TRACEPOINT_WRITABLE:
18593 		return true;
18594 	default:
18595 		return false;
18596 	}
18597 }
18598 
bpf_map_is_cgroup_storage(struct bpf_map * map)18599 static bool bpf_map_is_cgroup_storage(struct bpf_map *map)
18600 {
18601 	return (map->map_type == BPF_MAP_TYPE_CGROUP_STORAGE ||
18602 		map->map_type == BPF_MAP_TYPE_PERCPU_CGROUP_STORAGE);
18603 }
18604 
check_map_prog_compatibility(struct bpf_verifier_env * env,struct bpf_map * map,struct bpf_prog * prog)18605 static int check_map_prog_compatibility(struct bpf_verifier_env *env,
18606 					struct bpf_map *map,
18607 					struct bpf_prog *prog)
18608 
18609 {
18610 	enum bpf_prog_type prog_type = resolve_prog_type(prog);
18611 
18612 	if (map->excl_prog_sha &&
18613 	    memcmp(map->excl_prog_sha, prog->digest, SHA256_DIGEST_SIZE)) {
18614 		verbose(env, "program's hash doesn't match map's excl_prog_hash\n");
18615 		return -EACCES;
18616 	}
18617 
18618 	if (btf_record_has_field(map->record, BPF_LIST_HEAD) ||
18619 	    btf_record_has_field(map->record, BPF_RB_ROOT)) {
18620 		if (is_tracing_prog_type(prog_type)) {
18621 			verbose(env, "tracing progs cannot use bpf_{list_head,rb_root} yet\n");
18622 			return -EINVAL;
18623 		}
18624 	}
18625 
18626 	if (btf_record_has_field(map->record, BPF_SPIN_LOCK | BPF_RES_SPIN_LOCK)) {
18627 		if (prog_type == BPF_PROG_TYPE_SOCKET_FILTER) {
18628 			verbose(env, "socket filter progs cannot use bpf_spin_lock yet\n");
18629 			return -EINVAL;
18630 		}
18631 	}
18632 
18633 	if (btf_record_has_field(map->record, BPF_SPIN_LOCK)) {
18634 		if (is_tracing_prog_type(prog_type)) {
18635 			verbose(env, "tracing progs cannot use bpf_spin_lock yet\n");
18636 			return -EINVAL;
18637 		}
18638 	}
18639 
18640 	if ((bpf_prog_is_offloaded(prog->aux) || bpf_map_is_offloaded(map)) &&
18641 	    !bpf_offload_prog_map_match(prog, map)) {
18642 		verbose(env, "offload device mismatch between prog and map\n");
18643 		return -EINVAL;
18644 	}
18645 
18646 	if (map->map_type == BPF_MAP_TYPE_STRUCT_OPS) {
18647 		verbose(env, "bpf_struct_ops map cannot be used in prog\n");
18648 		return -EINVAL;
18649 	}
18650 
18651 	if (prog->sleepable)
18652 		switch (map->map_type) {
18653 		case BPF_MAP_TYPE_HASH:
18654 		case BPF_MAP_TYPE_RHASH:
18655 		case BPF_MAP_TYPE_LRU_HASH:
18656 		case BPF_MAP_TYPE_ARRAY:
18657 		case BPF_MAP_TYPE_PERCPU_HASH:
18658 		case BPF_MAP_TYPE_PERCPU_ARRAY:
18659 		case BPF_MAP_TYPE_LRU_PERCPU_HASH:
18660 		case BPF_MAP_TYPE_LPM_TRIE:
18661 		case BPF_MAP_TYPE_ARRAY_OF_MAPS:
18662 		case BPF_MAP_TYPE_HASH_OF_MAPS:
18663 		case BPF_MAP_TYPE_RINGBUF:
18664 		case BPF_MAP_TYPE_USER_RINGBUF:
18665 		case BPF_MAP_TYPE_INODE_STORAGE:
18666 		case BPF_MAP_TYPE_SK_STORAGE:
18667 		case BPF_MAP_TYPE_TASK_STORAGE:
18668 		case BPF_MAP_TYPE_CGRP_STORAGE:
18669 		case BPF_MAP_TYPE_QUEUE:
18670 		case BPF_MAP_TYPE_STACK:
18671 		case BPF_MAP_TYPE_ARENA:
18672 		case BPF_MAP_TYPE_INSN_ARRAY:
18673 		case BPF_MAP_TYPE_PROG_ARRAY:
18674 			break;
18675 		default:
18676 			verbose(env,
18677 				"Sleepable programs can only use array, hash, ringbuf and local storage maps\n");
18678 			return -EINVAL;
18679 		}
18680 
18681 	if (bpf_map_is_cgroup_storage(map) &&
18682 	    bpf_cgroup_storage_assign(env->prog->aux, map)) {
18683 		verbose(env, "only one cgroup storage of each type is allowed\n");
18684 		return -EBUSY;
18685 	}
18686 
18687 	if (map->map_type == BPF_MAP_TYPE_ARENA) {
18688 		if (env->prog->aux->arena) {
18689 			verbose(env, "Only one arena per program\n");
18690 			return -EBUSY;
18691 		}
18692 		if (!env->allow_ptr_leaks || !env->bpf_capable) {
18693 			verbose(env, "CAP_BPF and CAP_PERFMON are required to use arena\n");
18694 			return -EPERM;
18695 		}
18696 		if (!env->prog->jit_requested) {
18697 			verbose(env, "JIT is required to use arena\n");
18698 			return -EOPNOTSUPP;
18699 		}
18700 		if (!bpf_jit_supports_arena()) {
18701 			verbose(env, "JIT doesn't support arena\n");
18702 			return -EOPNOTSUPP;
18703 		}
18704 		env->prog->aux->arena = (void *)map;
18705 		env->prog->jit_required = true;
18706 		if (!bpf_arena_get_user_vm_start(env->prog->aux->arena)) {
18707 			verbose(env, "arena's user address must be set via map_extra or mmap()\n");
18708 			return -EINVAL;
18709 		}
18710 	}
18711 
18712 	return 0;
18713 }
18714 
__add_used_map(struct bpf_verifier_env * env,struct bpf_map * map)18715 static int __add_used_map(struct bpf_verifier_env *env, struct bpf_map *map)
18716 {
18717 	int i, err;
18718 
18719 	/* check whether we recorded this map already */
18720 	for (i = 0; i < env->used_map_cnt; i++)
18721 		if (env->used_maps[i] == map)
18722 			return i;
18723 
18724 	if (env->signature &&
18725 	    env->prog->aux->sig.verdict == BPF_SIG_VERIFIED) {
18726 		verbose(env, "signed program cannot bind map '%s' not covered by the signature\n",
18727 			map->name);
18728 		return -EACCES;
18729 	}
18730 	if (env->used_map_cnt >= MAX_USED_MAPS) {
18731 		verbose(env, "The total number of maps per program has reached the limit of %u\n",
18732 			MAX_USED_MAPS);
18733 		return -E2BIG;
18734 	}
18735 
18736 	err = check_map_prog_compatibility(env, map, env->prog);
18737 	if (err)
18738 		return err;
18739 
18740 	if (env->prog->sleepable)
18741 		atomic64_inc(&map->sleepable_refcnt);
18742 
18743 	/* hold the map. If the program is rejected by verifier,
18744 	 * the map will be released by release_maps() or it
18745 	 * will be used by the valid program until it's unloaded
18746 	 * and all maps are released in bpf_free_used_maps()
18747 	 */
18748 	bpf_map_inc(map);
18749 
18750 	env->used_maps[env->used_map_cnt++] = map;
18751 
18752 	if (map->map_type == BPF_MAP_TYPE_INSN_ARRAY) {
18753 		err = bpf_insn_array_init(map, env->prog);
18754 		if (err) {
18755 			verbose(env, "Failed to properly initialize insn array\n");
18756 			return err;
18757 		}
18758 		env->insn_array_maps[env->insn_array_map_cnt++] = map;
18759 		env->prog->jit_required = true;
18760 	}
18761 
18762 	return env->used_map_cnt - 1;
18763 }
18764 
18765 /* Add map behind fd to used maps list, if it's not already there, and return
18766  * its index.
18767  * Returns <0 on error, or >= 0 index, on success.
18768  */
add_used_map(struct bpf_verifier_env * env,int fd)18769 static int add_used_map(struct bpf_verifier_env *env, int fd)
18770 {
18771 	struct bpf_map *map;
18772 	CLASS(fd, f)(fd);
18773 
18774 	map = __bpf_map_get(f);
18775 	if (IS_ERR(map)) {
18776 		verbose(env, "fd %d is not pointing to valid bpf_map\n", fd);
18777 		return PTR_ERR(map);
18778 	}
18779 
18780 	return __add_used_map(env, map);
18781 }
18782 
fd_array_get_map_idx_continuous(struct bpf_verifier_env * env,u32 idx)18783 static int fd_array_get_map_idx_continuous(struct bpf_verifier_env *env, u32 idx)
18784 {
18785 	struct bpf_map *map;
18786 
18787 	if (idx >= env->fd_array_cnt) {
18788 		verbose(env, "fd_idx %u out of bounds, fd_array_cnt %u\n",
18789 			idx, env->fd_array_cnt);
18790 		return -EINVAL;
18791 	}
18792 	map = fd_slot_map(env->fd_array[idx]);
18793 	if (!map) {
18794 		verbose(env, "fd_idx %u is not a map\n", idx);
18795 		return -EINVAL;
18796 	}
18797 	return __add_used_map(env, map);
18798 }
18799 
fd_array_get_map_idx_sparse(struct bpf_verifier_env * env,u32 idx)18800 static int fd_array_get_map_idx_sparse(struct bpf_verifier_env *env, u32 idx)
18801 {
18802 	int fd;
18803 
18804 	if (copy_from_bpfptr_offset(&fd, env->fd_array_raw,
18805 				    (size_t)idx * sizeof(fd), sizeof(fd)))
18806 		return -EFAULT;
18807 	return add_used_map(env, fd);
18808 }
18809 
fd_array_get_map_idx(struct bpf_verifier_env * env,u32 idx)18810 static int fd_array_get_map_idx(struct bpf_verifier_env *env, u32 idx)
18811 {
18812 	if (env->fd_array)
18813 		return fd_array_get_map_idx_continuous(env, idx);
18814 	if (env->signature) {
18815 		verbose(env, "signed program must bind maps via a continuous fd_array (fd_array_cnt)\n");
18816 		return -EACCES;
18817 	}
18818 	if (!bpfptr_is_null(env->fd_array_raw))
18819 		return fd_array_get_map_idx_sparse(env, idx);
18820 
18821 	verbose(env, "fd_idx without fd_array is invalid\n");
18822 	return -EPROTO;
18823 }
18824 
check_alu_fields(struct bpf_verifier_env * env,struct bpf_insn * insn)18825 static int check_alu_fields(struct bpf_verifier_env *env, struct bpf_insn *insn)
18826 {
18827 	u8 class = BPF_CLASS(insn->code);
18828 	u8 opcode = BPF_OP(insn->code);
18829 
18830 	switch (opcode) {
18831 	case BPF_NEG:
18832 		if (BPF_SRC(insn->code) != BPF_K || insn->src_reg != BPF_REG_0 ||
18833 		    insn->off != 0 || insn->imm != 0) {
18834 			verbose(env, "BPF_NEG uses reserved fields\n");
18835 			return -EINVAL;
18836 		}
18837 		return 0;
18838 	case BPF_END:
18839 		if (insn->src_reg != BPF_REG_0 || insn->off != 0 ||
18840 		    (insn->imm != 16 && insn->imm != 32 && insn->imm != 64) ||
18841 		    (class == BPF_ALU64 && BPF_SRC(insn->code) != BPF_TO_LE)) {
18842 			verbose(env, "BPF_END uses reserved fields\n");
18843 			return -EINVAL;
18844 		}
18845 		return 0;
18846 	case BPF_MOV:
18847 		if (BPF_SRC(insn->code) == BPF_X) {
18848 			if (class == BPF_ALU) {
18849 				if ((insn->off != 0 && insn->off != 8 && insn->off != 16) ||
18850 				    insn->imm) {
18851 					verbose(env, "BPF_MOV uses reserved fields\n");
18852 					return -EINVAL;
18853 				}
18854 			} else if (insn->off == BPF_ADDR_SPACE_CAST) {
18855 				if (insn->imm != 1 && insn->imm != 1u << 16) {
18856 					verbose(env, "addr_space_cast insn can only convert between address space 1 and 0\n");
18857 					return -EINVAL;
18858 				}
18859 			} else if ((insn->off != 0 && insn->off != 8 &&
18860 				    insn->off != 16 && insn->off != 32) || insn->imm) {
18861 				verbose(env, "BPF_MOV uses reserved fields\n");
18862 				return -EINVAL;
18863 			}
18864 		} else if (insn->src_reg != BPF_REG_0 || insn->off != 0) {
18865 			verbose(env, "BPF_MOV uses reserved fields\n");
18866 			return -EINVAL;
18867 		}
18868 		return 0;
18869 	case BPF_ADD:
18870 	case BPF_SUB:
18871 	case BPF_AND:
18872 	case BPF_OR:
18873 	case BPF_XOR:
18874 	case BPF_LSH:
18875 	case BPF_RSH:
18876 	case BPF_ARSH:
18877 	case BPF_MUL:
18878 	case BPF_DIV:
18879 	case BPF_MOD:
18880 		if (BPF_SRC(insn->code) == BPF_X) {
18881 			if (insn->imm != 0 || (insn->off != 0 && insn->off != 1) ||
18882 			    (insn->off == 1 && opcode != BPF_MOD && opcode != BPF_DIV)) {
18883 				verbose(env, "BPF_ALU uses reserved fields\n");
18884 				return -EINVAL;
18885 			}
18886 		} else if (insn->src_reg != BPF_REG_0 ||
18887 			   (insn->off != 0 && insn->off != 1) ||
18888 			   (insn->off == 1 && opcode != BPF_MOD && opcode != BPF_DIV)) {
18889 			verbose(env, "BPF_ALU uses reserved fields\n");
18890 			return -EINVAL;
18891 		}
18892 		return 0;
18893 	default:
18894 		verbose(env, "invalid BPF_ALU opcode %x\n", opcode);
18895 		return -EINVAL;
18896 	}
18897 }
18898 
check_jmp_fields(struct bpf_verifier_env * env,struct bpf_insn * insn)18899 static int check_jmp_fields(struct bpf_verifier_env *env, struct bpf_insn *insn)
18900 {
18901 	u8 class = BPF_CLASS(insn->code);
18902 	u8 opcode = BPF_OP(insn->code);
18903 
18904 	switch (opcode) {
18905 	case BPF_CALL:
18906 		if (BPF_SRC(insn->code) != BPF_K ||
18907 		    (insn->src_reg != BPF_PSEUDO_KFUNC_CALL && insn->off != 0) ||
18908 		    (insn->src_reg != BPF_REG_0 && insn->src_reg != BPF_PSEUDO_CALL &&
18909 		     insn->src_reg != BPF_PSEUDO_KFUNC_CALL) ||
18910 		    insn->dst_reg != BPF_REG_0 || class == BPF_JMP32) {
18911 			verbose(env, "BPF_CALL uses reserved fields\n");
18912 			return -EINVAL;
18913 		}
18914 		return 0;
18915 	case BPF_JA:
18916 		if (BPF_SRC(insn->code) == BPF_X) {
18917 			if (insn->src_reg != BPF_REG_0 || insn->imm != 0 || insn->off != 0) {
18918 				verbose(env, "BPF_JA|BPF_X uses reserved fields\n");
18919 				return -EINVAL;
18920 			}
18921 		} else if (insn->src_reg != BPF_REG_0 || insn->dst_reg != BPF_REG_0 ||
18922 			   (class == BPF_JMP && insn->imm != 0) ||
18923 			   (class == BPF_JMP32 && insn->off != 0)) {
18924 			verbose(env, "BPF_JA uses reserved fields\n");
18925 			return -EINVAL;
18926 		}
18927 		return 0;
18928 	case BPF_EXIT:
18929 		if (BPF_SRC(insn->code) != BPF_K || insn->imm != 0 ||
18930 		    insn->src_reg != BPF_REG_0 || insn->dst_reg != BPF_REG_0 ||
18931 		    class == BPF_JMP32) {
18932 			verbose(env, "BPF_EXIT uses reserved fields\n");
18933 			return -EINVAL;
18934 		}
18935 		return 0;
18936 	case BPF_JCOND:
18937 		if (insn->code != (BPF_JMP | BPF_JCOND) || insn->src_reg != BPF_MAY_GOTO ||
18938 		    insn->dst_reg || insn->imm) {
18939 			verbose(env, "invalid may_goto imm %d\n", insn->imm);
18940 			return -EINVAL;
18941 		}
18942 		return 0;
18943 	default:
18944 		if (BPF_SRC(insn->code) == BPF_X) {
18945 			if (insn->imm != 0) {
18946 				verbose(env, "BPF_JMP/JMP32 uses reserved fields\n");
18947 				return -EINVAL;
18948 			}
18949 		} else if (insn->src_reg != BPF_REG_0) {
18950 			verbose(env, "BPF_JMP/JMP32 uses reserved fields\n");
18951 			return -EINVAL;
18952 		}
18953 		return 0;
18954 	}
18955 }
18956 
check_insn_fields(struct bpf_verifier_env * env,struct bpf_insn * insn)18957 static int check_insn_fields(struct bpf_verifier_env *env, struct bpf_insn *insn)
18958 {
18959 	switch (BPF_CLASS(insn->code)) {
18960 	case BPF_ALU:
18961 	case BPF_ALU64:
18962 		return check_alu_fields(env, insn);
18963 	case BPF_LDX:
18964 		if ((BPF_MODE(insn->code) != BPF_MEM && BPF_MODE(insn->code) != BPF_MEMSX) ||
18965 		    insn->imm != 0) {
18966 			verbose(env, "BPF_LDX uses reserved fields\n");
18967 			return -EINVAL;
18968 		}
18969 		return 0;
18970 	case BPF_STX:
18971 		if (BPF_MODE(insn->code) == BPF_ATOMIC)
18972 			return 0;
18973 		if (BPF_MODE(insn->code) != BPF_MEM || insn->imm != 0) {
18974 			verbose(env, "BPF_STX uses reserved fields\n");
18975 			return -EINVAL;
18976 		}
18977 		return 0;
18978 	case BPF_ST:
18979 		if (BPF_MODE(insn->code) != BPF_MEM || insn->src_reg != BPF_REG_0) {
18980 			verbose(env, "BPF_ST uses reserved fields\n");
18981 			return -EINVAL;
18982 		}
18983 		return 0;
18984 	case BPF_JMP:
18985 	case BPF_JMP32:
18986 		return check_jmp_fields(env, insn);
18987 	case BPF_LD: {
18988 		u8 mode = BPF_MODE(insn->code);
18989 
18990 		if (mode == BPF_ABS || mode == BPF_IND) {
18991 			if (insn->dst_reg != BPF_REG_0 || insn->off != 0 ||
18992 			    BPF_SIZE(insn->code) == BPF_DW ||
18993 			    (mode == BPF_ABS && insn->src_reg != BPF_REG_0)) {
18994 				verbose(env, "BPF_LD_[ABS|IND] uses reserved fields\n");
18995 				return -EINVAL;
18996 			}
18997 		} else if (mode != BPF_IMM) {
18998 			verbose(env, "invalid BPF_LD mode\n");
18999 			return -EINVAL;
19000 		}
19001 		return 0;
19002 	}
19003 	default:
19004 		verbose(env, "unknown insn class %d\n", BPF_CLASS(insn->code));
19005 		return -EINVAL;
19006 	}
19007 }
19008 
19009 /*
19010  * Check that insns are sane and rewrite pseudo imm in ld_imm64 instructions:
19011  *
19012  * 1. if it accesses map FD, replace it with actual map pointer.
19013  * 2. if it accesses btf_id of a VAR, replace it with pointer to the var.
19014  *
19015  * NOTE: btf_vmlinux is required for converting pseudo btf_id.
19016  */
check_and_resolve_insns(struct bpf_verifier_env * env)19017 static int check_and_resolve_insns(struct bpf_verifier_env *env)
19018 {
19019 	struct bpf_insn *insn = env->prog->insnsi;
19020 	int insn_cnt = env->prog->len;
19021 	int i, err;
19022 
19023 	err = bpf_prog_calc_tag(env->prog);
19024 	if (err)
19025 		return err;
19026 
19027 	for (i = 0; i < insn_cnt; i++, insn++) {
19028 		if (insn->dst_reg >= MAX_BPF_REG &&
19029 		    !is_stack_arg_st(insn) && !is_stack_arg_stx(insn)) {
19030 			verbose(env, "R%d is invalid\n", insn->dst_reg);
19031 			return -EINVAL;
19032 		}
19033 		if (insn->src_reg >= MAX_BPF_REG && !is_stack_arg_ldx(insn)) {
19034 			verbose(env, "R%d is invalid\n", insn->src_reg);
19035 			return -EINVAL;
19036 		}
19037 		if (insn[0].code == (BPF_LD | BPF_IMM | BPF_DW)) {
19038 			struct bpf_insn_aux_data *aux;
19039 			struct bpf_map *map;
19040 			int map_idx;
19041 			u64 addr;
19042 
19043 			if (i == insn_cnt - 1 || insn[1].code != 0 ||
19044 			    insn[1].dst_reg != 0 || insn[1].src_reg != 0 ||
19045 			    insn[1].off != 0) {
19046 				verbose(env, "invalid bpf_ld_imm64 insn\n");
19047 				return -EINVAL;
19048 			}
19049 
19050 			if (insn[0].off != 0) {
19051 				verbose(env, "BPF_LD_IMM64 uses reserved fields\n");
19052 				return -EINVAL;
19053 			}
19054 
19055 			if (insn[0].src_reg == 0)
19056 				/* valid generic load 64-bit imm */
19057 				goto next_insn;
19058 
19059 			if (insn[0].src_reg == BPF_PSEUDO_BTF_ID) {
19060 				aux = &env->insn_aux_data[i];
19061 				err = check_pseudo_btf_id(env, insn, aux);
19062 				if (err)
19063 					return err;
19064 				goto next_insn;
19065 			}
19066 
19067 			if (insn[0].src_reg == BPF_PSEUDO_FUNC) {
19068 				aux = &env->insn_aux_data[i];
19069 				aux->ptr_type = PTR_TO_FUNC;
19070 				goto next_insn;
19071 			}
19072 
19073 			/* In final convert_pseudo_ld_imm64() step, this is
19074 			 * converted into regular 64-bit imm load insn.
19075 			 */
19076 			switch (insn[0].src_reg) {
19077 			case BPF_PSEUDO_MAP_VALUE:
19078 			case BPF_PSEUDO_MAP_IDX_VALUE:
19079 				break;
19080 			case BPF_PSEUDO_MAP_FD:
19081 			case BPF_PSEUDO_MAP_IDX:
19082 				if (insn[1].imm == 0)
19083 					break;
19084 				fallthrough;
19085 			default:
19086 				verbose(env, "unrecognized bpf_ld_imm64 insn\n");
19087 				return -EINVAL;
19088 			}
19089 
19090 			switch (insn[0].src_reg) {
19091 			case BPF_PSEUDO_MAP_IDX_VALUE:
19092 			case BPF_PSEUDO_MAP_IDX:
19093 				map_idx = fd_array_get_map_idx(env, insn[0].imm);
19094 				break;
19095 			default:
19096 				if (env->signature) {
19097 					verbose(env, "signed program cannot reference a map by fd, only via fd_array index\n");
19098 					return -EINVAL;
19099 				}
19100 				map_idx = add_used_map(env, insn[0].imm);
19101 				break;
19102 			}
19103 
19104 			if (map_idx < 0)
19105 				return map_idx;
19106 			map = env->used_maps[map_idx];
19107 
19108 			aux = &env->insn_aux_data[i];
19109 			aux->map_index = map_idx;
19110 
19111 			if (insn[0].src_reg == BPF_PSEUDO_MAP_FD ||
19112 			    insn[0].src_reg == BPF_PSEUDO_MAP_IDX) {
19113 				addr = (unsigned long)map;
19114 			} else {
19115 				u32 off = insn[1].imm;
19116 
19117 				if (!map->ops->map_direct_value_addr) {
19118 					verbose(env, "no direct value access support for this map type\n");
19119 					return -EINVAL;
19120 				}
19121 
19122 				err = map->ops->map_direct_value_addr(map, &addr, off);
19123 				if (err) {
19124 					verbose(env, "invalid access to map value pointer, value_size=%u off=%u\n",
19125 						map->value_size, off);
19126 					return err;
19127 				}
19128 
19129 				aux->map_off = off;
19130 				addr += off;
19131 			}
19132 
19133 			insn[0].imm = (u32)addr;
19134 			insn[1].imm = addr >> 32;
19135 
19136 next_insn:
19137 			insn++;
19138 			i++;
19139 			continue;
19140 		}
19141 
19142 		/* Basic sanity check before we invest more work here. */
19143 		if (!bpf_opcode_in_insntable(insn->code)) {
19144 			verbose(env, "unknown opcode %02x\n", insn->code);
19145 			return -EINVAL;
19146 		}
19147 
19148 		err = check_insn_fields(env, insn);
19149 		if (err)
19150 			return err;
19151 	}
19152 
19153 	/* now all pseudo BPF_LD_IMM64 instructions load valid
19154 	 * 'struct bpf_map *' into a register instead of user map_fd.
19155 	 * These pointers will be used later by verifier to validate map access.
19156 	 */
19157 	return 0;
19158 }
19159 
19160 /* drop refcnt of maps used by the rejected program */
release_maps(struct bpf_verifier_env * env)19161 static void release_maps(struct bpf_verifier_env *env)
19162 {
19163 	__bpf_free_used_maps(env->prog->aux, env->used_maps,
19164 			     env->used_map_cnt);
19165 }
19166 
19167 /* drop refcnt of maps used by the rejected program */
release_btfs(struct bpf_verifier_env * env)19168 static void release_btfs(struct bpf_verifier_env *env)
19169 {
19170 	__bpf_free_used_btfs(env->used_btfs, env->used_btf_cnt);
19171 }
19172 
19173 /* convert pseudo BPF_LD_IMM64 into generic BPF_LD_IMM64 */
convert_pseudo_ld_imm64(struct bpf_verifier_env * env)19174 static void convert_pseudo_ld_imm64(struct bpf_verifier_env *env)
19175 {
19176 	struct bpf_insn *insn = env->prog->insnsi;
19177 	int insn_cnt = env->prog->len;
19178 	int i;
19179 
19180 	for (i = 0; i < insn_cnt; i++, insn++) {
19181 		if (insn->code != (BPF_LD | BPF_IMM | BPF_DW))
19182 			continue;
19183 		if (insn->src_reg == BPF_PSEUDO_FUNC)
19184 			continue;
19185 		insn->src_reg = 0;
19186 	}
19187 }
19188 
release_insn_arrays(struct bpf_verifier_env * env)19189 static void release_insn_arrays(struct bpf_verifier_env *env)
19190 {
19191 	int i;
19192 
19193 	for (i = 0; i < env->insn_array_map_cnt; i++)
19194 		bpf_insn_array_release(env->insn_array_maps[i]);
19195 }
19196 
19197 /* The verifier does more data flow analysis than llvm and will not
19198  * explore branches that are dead at run time. Malicious programs can
19199  * have dead code too. Therefore replace all dead at-run-time code
19200  * with 'ja -1'.
19201  *
19202  * Just nops are not optimal, e.g. if they would sit at the end of the
19203  * program and through another bug we would manage to jump there, then
19204  * we'd execute beyond program memory otherwise. Returning exception
19205  * code also wouldn't work since we can have subprogs where the dead
19206  * code could be located.
19207  */
sanitize_dead_code(struct bpf_verifier_env * env)19208 static void sanitize_dead_code(struct bpf_verifier_env *env)
19209 {
19210 	struct bpf_insn_aux_data *aux_data = env->insn_aux_data;
19211 	struct bpf_insn trap = BPF_JMP_IMM(BPF_JA, 0, 0, -1);
19212 	struct bpf_insn *insn = env->prog->insnsi;
19213 	const int insn_cnt = env->prog->len;
19214 	int i;
19215 
19216 	for (i = 0; i < insn_cnt; i++) {
19217 		if (aux_data[i].seen)
19218 			continue;
19219 		memcpy(insn + i, &trap, sizeof(trap));
19220 		aux_data[i].zext_dst = false;
19221 	}
19222 }
19223 
free_states(struct bpf_verifier_env * env)19224 static void free_states(struct bpf_verifier_env *env)
19225 {
19226 	struct bpf_verifier_state_list *sl;
19227 	struct list_head *head, *pos, *tmp;
19228 	struct bpf_scc_info *info;
19229 	int i, j;
19230 
19231 	bpf_free_verifier_state(env->cur_state, true);
19232 	env->cur_state = NULL;
19233 	while (!pop_stack(env, NULL, NULL, false));
19234 
19235 	list_for_each_safe(pos, tmp, &env->free_list) {
19236 		sl = container_of(pos, struct bpf_verifier_state_list, node);
19237 		bpf_free_verifier_state(&sl->state, false);
19238 		kfree(sl);
19239 	}
19240 	INIT_LIST_HEAD(&env->free_list);
19241 
19242 	for (i = 0; i < env->scc_cnt; ++i) {
19243 		info = env->scc_info[i];
19244 		if (!info)
19245 			continue;
19246 		for (j = 0; j < info->num_visits; j++)
19247 			bpf_free_backedges(&info->visits[j]);
19248 		kvfree(info);
19249 		env->scc_info[i] = NULL;
19250 	}
19251 
19252 	if (!env->explored_states)
19253 		return;
19254 
19255 	for (i = 0; i < state_htab_size(env); i++) {
19256 		head = &env->explored_states[i];
19257 
19258 		list_for_each_safe(pos, tmp, head) {
19259 			sl = container_of(pos, struct bpf_verifier_state_list, node);
19260 			bpf_free_verifier_state(&sl->state, false);
19261 			kfree(sl);
19262 		}
19263 		INIT_LIST_HEAD(&env->explored_states[i]);
19264 	}
19265 }
19266 
do_check_common(struct bpf_verifier_env * env,int subprog)19267 static int do_check_common(struct bpf_verifier_env *env, int subprog)
19268 {
19269 	bool pop_log = !(env->log.level & BPF_LOG_LEVEL2);
19270 	struct bpf_subprog_info *sub = subprog_info(env, subprog);
19271 	struct bpf_prog_aux *aux = env->prog->aux;
19272 	struct bpf_verifier_state *state;
19273 	struct bpf_reg_state *regs;
19274 	u32 insn_processed = env->insn_processed;
19275 	int ret, i;
19276 
19277 	env->prev_linfo = NULL;
19278 	env->pass_cnt++;
19279 
19280 	state = kzalloc_obj(struct bpf_verifier_state, GFP_KERNEL_ACCOUNT);
19281 	if (!state)
19282 		return -ENOMEM;
19283 	state->curframe = 0;
19284 	state->speculative = false;
19285 	state->branches = 1;
19286 	state->in_sleepable = env->prog->sleepable;
19287 	state->frame[0] = kzalloc_obj(struct bpf_func_state, GFP_KERNEL_ACCOUNT);
19288 	if (!state->frame[0]) {
19289 		kfree(state);
19290 		return -ENOMEM;
19291 	}
19292 	env->cur_state = state;
19293 	init_func_state(env, state->frame[0],
19294 			BPF_MAIN_FUNC /* callsite */,
19295 			0 /* frameno */,
19296 			subprog);
19297 	state->first_insn_idx = env->subprog_info[subprog].start;
19298 	state->last_insn_idx = -1;
19299 
19300 	regs = state->frame[state->curframe]->regs;
19301 	if (subprog || env->prog->type == BPF_PROG_TYPE_EXT) {
19302 		const char *sub_name = bpf_subprog_name(env, subprog);
19303 		struct bpf_subprog_arg_info *arg;
19304 		struct bpf_reg_state *reg;
19305 
19306 		if (env->log.level & BPF_LOG_LEVEL)
19307 			verbose(env, "Validating %s() func#%d...\n", sub_name, subprog);
19308 		ret = btf_prepare_func_args(env, subprog);
19309 		if (ret)
19310 			goto out;
19311 
19312 		if (subprog_is_exc_cb(env, subprog)) {
19313 			state->frame[0]->in_exception_callback_fn = true;
19314 
19315 			/*
19316 			 * Global functions are scalar or void, make sure
19317 			 * we return a scalar.
19318 			 */
19319 			if (subprog_returns_void(env, subprog)) {
19320 				verbose(env, "exception cb cannot return void\n");
19321 				ret = -EINVAL;
19322 				goto out;
19323 			}
19324 
19325 			/* Also ensure the callback only has a single scalar argument. */
19326 			if (sub->arg_cnt != 1 || sub->args[0].arg_type != ARG_ANYTHING) {
19327 				verbose(env, "exception cb only supports single integer argument\n");
19328 				ret = -EINVAL;
19329 				goto out;
19330 			}
19331 		}
19332 		for (i = BPF_REG_1; i <= min_t(u32, sub->arg_cnt, MAX_BPF_FUNC_REG_ARGS); i++) {
19333 			arg = &sub->args[i - BPF_REG_1];
19334 			reg = &regs[i];
19335 
19336 			if (arg->arg_type == ARG_PTR_TO_CTX) {
19337 				reg->type = PTR_TO_CTX;
19338 				mark_reg_known_zero(env, regs, i);
19339 			} else if (arg->arg_type == ARG_ANYTHING) {
19340 				reg->type = SCALAR_VALUE;
19341 				mark_reg_unknown(env, regs, i);
19342 			} else if (arg->arg_type == ARG_PTR_TO_DYNPTR) {
19343 				/* assume unspecial LOCAL dynptr type */
19344 				__mark_dynptr_reg(reg, BPF_DYNPTR_TYPE_LOCAL, true, ++env->id_gen, 0);
19345 			} else if (base_type(arg->arg_type) == ARG_PTR_TO_MEM) {
19346 				reg->type = PTR_TO_MEM;
19347 				reg->type |= arg->arg_type &
19348 					     (PTR_MAYBE_NULL | PTR_UNTRUSTED | MEM_RDONLY);
19349 				mark_reg_known_zero(env, regs, i);
19350 				reg->mem_size = arg->mem_size;
19351 				if (arg->arg_type & PTR_MAYBE_NULL)
19352 					reg->id = ++env->id_gen;
19353 			} else if (base_type(arg->arg_type) == ARG_PTR_TO_BTF_ID) {
19354 				reg->type = PTR_TO_BTF_ID;
19355 				if (arg->arg_type & PTR_MAYBE_NULL)
19356 					reg->type |= PTR_MAYBE_NULL;
19357 				if (arg->arg_type & PTR_UNTRUSTED)
19358 					reg->type |= PTR_UNTRUSTED;
19359 				if (arg->arg_type & PTR_TRUSTED)
19360 					reg->type |= PTR_TRUSTED;
19361 				mark_reg_known_zero(env, regs, i);
19362 				reg->btf = bpf_get_btf_vmlinux(); /* can't fail at this point */
19363 				reg->btf_id = arg->btf_id;
19364 				reg->id = ++env->id_gen;
19365 			} else if (base_type(arg->arg_type) == ARG_PTR_TO_ARENA) {
19366 				/* caller can pass either PTR_TO_ARENA or SCALAR */
19367 				mark_reg_unknown(env, regs, i);
19368 			} else {
19369 				verifier_bug(env, "unhandled arg#%d type %d",
19370 					     i - BPF_REG_1 + 1, arg->arg_type);
19371 				ret = -EFAULT;
19372 				goto out;
19373 			}
19374 		}
19375 		if (env->prog->type == BPF_PROG_TYPE_EXT && sub->arg_cnt > MAX_BPF_FUNC_REG_ARGS) {
19376 			verbose(env, "freplace programs with >%d args not supported yet\n",
19377 				MAX_BPF_FUNC_REG_ARGS);
19378 			ret = -EINVAL;
19379 			goto out;
19380 		}
19381 	} else {
19382 		/* if main BPF program has associated BTF info, validate that
19383 		 * it's matching expected signature, and otherwise mark BTF
19384 		 * info for main program as unreliable
19385 		 */
19386 		if (env->prog->aux->func_info_aux) {
19387 			ret = btf_prepare_func_args(env, 0);
19388 			if (ret || sub->arg_cnt != 1 || sub->args[0].arg_type != ARG_PTR_TO_CTX) {
19389 				env->prog->aux->func_info_aux[0].unreliable = true;
19390 				sub->arg_cnt = 1;
19391 				sub->stack_arg_cnt = 0;
19392 			}
19393 		}
19394 
19395 		/* 1st arg to a function */
19396 		regs[BPF_REG_1].type = PTR_TO_CTX;
19397 		mark_reg_known_zero(env, regs, BPF_REG_1);
19398 	}
19399 
19400 	/* Acquire references for struct_ops program arguments tagged with "__ref" */
19401 	if (!subprog && env->prog->type == BPF_PROG_TYPE_STRUCT_OPS) {
19402 		for (i = 0; i < aux->ctx_arg_info_size; i++) {
19403 			ret = aux->ctx_arg_info[i].refcounted ? acquire_reference(env, 0, 0) : 0;
19404 			if (ret < 0)
19405 				goto out;
19406 
19407 			aux->ctx_arg_info[i].ref_id = ret;
19408 		}
19409 	}
19410 
19411 	ret = do_check(env);
19412 out:
19413 	account_current_path(env);
19414 	if (!ret) {
19415 		if (pop_log)
19416 			bpf_vlog_reset(&env->log, 0);
19417 		bpf_diag_event_log_restore(env, 0);
19418 	}
19419 	free_states(env);
19420 
19421 	/*
19422 	 * The override is needed to account for async subprograms, which
19423 	 * are verified with their own set of stack frames and thus are
19424 	 * not accounted as callees by account_current_path().
19425 	 * Accumulate their total counts as total counts of the main or
19426 	 * global subprog hosting the async call.
19427 	 */
19428 	env->subprog_info[subprog].insns_total = env->insn_processed - insn_processed;
19429 	return ret;
19430 }
19431 
19432 /* Lazily verify all global functions based on their BTF, if they are called
19433  * from main BPF program or any of subprograms transitively.
19434  * BPF global subprogs called from dead code are not validated.
19435  * All callable global functions must pass verification.
19436  * Otherwise the whole program is rejected.
19437  * Consider:
19438  * int bar(int);
19439  * int foo(int f)
19440  * {
19441  *    return bar(f);
19442  * }
19443  * int bar(int b)
19444  * {
19445  *    ...
19446  * }
19447  * foo() will be verified first for R1=any_scalar_value. During verification it
19448  * will be assumed that bar() already verified successfully and call to bar()
19449  * from foo() will be checked for type match only. Later bar() will be verified
19450  * independently to check that it's safe for R1=any_scalar_value.
19451  */
do_check_subprogs(struct bpf_verifier_env * env)19452 static int do_check_subprogs(struct bpf_verifier_env *env)
19453 {
19454 	struct bpf_prog_aux *aux = env->prog->aux;
19455 	struct bpf_func_info_aux *sub_aux;
19456 	int i, ret, new_cnt;
19457 
19458 	if (!aux->func_info)
19459 		return 0;
19460 
19461 	/* exception callback is presumed to be always called */
19462 	if (env->exception_callback_subprog)
19463 		subprog_aux(env, env->exception_callback_subprog)->called = true;
19464 
19465 again:
19466 	new_cnt = 0;
19467 	for (i = 1; i < env->subprog_cnt; i++) {
19468 		if (!bpf_subprog_is_global(env, i))
19469 			continue;
19470 
19471 		sub_aux = subprog_aux(env, i);
19472 		if (!sub_aux->called || sub_aux->verified)
19473 			continue;
19474 
19475 		env->insn_idx = env->subprog_info[i].start;
19476 		WARN_ON_ONCE(env->insn_idx == 0);
19477 		ret = do_check_common(env, i);
19478 		if (ret) {
19479 			return ret;
19480 		} else if (env->log.level & BPF_LOG_LEVEL) {
19481 			verbose(env, "Func#%d ('%s') is safe for any args that match its prototype\n",
19482 				i, bpf_subprog_name(env, i));
19483 		}
19484 
19485 		/* We verified new global subprog, it might have called some
19486 		 * more global subprogs that we haven't verified yet, so we
19487 		 * need to do another pass over subprogs to verify those.
19488 		 */
19489 		sub_aux->verified = true;
19490 		new_cnt++;
19491 	}
19492 
19493 	/* We can't loop forever as we verify at least one global subprog on
19494 	 * each pass.
19495 	 */
19496 	if (new_cnt)
19497 		goto again;
19498 
19499 	return 0;
19500 }
19501 
do_check_main(struct bpf_verifier_env * env)19502 static int do_check_main(struct bpf_verifier_env *env)
19503 {
19504 	int ret;
19505 
19506 	env->insn_idx = 0;
19507 	ret = do_check_common(env, 0);
19508 	if (!ret)
19509 		env->prog->aux->stack_depth = env->subprog_info[0].stack_depth;
19510 	return ret;
19511 }
19512 
print_verification_stats(struct bpf_verifier_env * env)19513 static void print_verification_stats(struct bpf_verifier_env *env)
19514 {
19515 	/* Skip over hidden subprogs which are not verified. */
19516 	int i, subprog_cnt = env->subprog_cnt - env->hidden_subprog_cnt;
19517 
19518 	if (env->log.level & BPF_LOG_STATS) {
19519 		verbose(env, "verification time %lld usec\n",
19520 			div_u64(env->verification_time, 1000));
19521 		verbose(env, "stack depth max %d\n", env->max_stack_depth);
19522 		for (i = 0; i < subprog_cnt; i++) {
19523 			const char *name = env->subprog_info[i].name;
19524 			const char *kind;
19525 
19526 			if (!name || !name[0])
19527 				name = "<unknown>";
19528 			kind = i == 0 ? "main" :
19529 			       bpf_subprog_is_global(env, i) ? "global" : "static";
19530 			verbose(env, "subprog %d (%s) %s insns_self %d insns_total %d stack %d\n",
19531 				i, name, kind, env->subprog_info[i].insns_self,
19532 				env->subprog_info[i].insns_total,
19533 				env->subprog_info[i].stack_depth);
19534 		}
19535 	}
19536 	verbose(env, "processed %d insns (limit %d) max_states_per_insn %d "
19537 		"total_states %d peak_states %d mark_read %d\n",
19538 		env->insn_processed, BPF_COMPLEXITY_LIMIT_INSNS,
19539 		env->max_states_per_insn, env->total_states,
19540 		env->peak_states, env->longest_mark_read_walk);
19541 }
19542 
bpf_prog_ctx_arg_info_init(struct bpf_prog * prog,const struct bpf_ctx_arg_aux * info,u32 cnt)19543 int bpf_prog_ctx_arg_info_init(struct bpf_prog *prog,
19544 			       const struct bpf_ctx_arg_aux *info, u32 cnt)
19545 {
19546 	prog->aux->ctx_arg_info = kmemdup_array(info, cnt, sizeof(*info), GFP_KERNEL_ACCOUNT);
19547 	prog->aux->ctx_arg_info_size = cnt;
19548 
19549 	return prog->aux->ctx_arg_info ? 0 : -ENOMEM;
19550 }
19551 
check_struct_ops_btf_id(struct bpf_verifier_env * env)19552 static int check_struct_ops_btf_id(struct bpf_verifier_env *env)
19553 {
19554 	const struct btf_type *t, *func_proto;
19555 	const struct bpf_struct_ops_desc *st_ops_desc;
19556 	const struct bpf_struct_ops_arg_info *arg_info;
19557 	const struct bpf_struct_ops *st_ops;
19558 	const struct btf_member *member;
19559 	struct bpf_prog *prog = env->prog;
19560 	bool has_refcounted_arg = false;
19561 	u32 btf_id, member_idx, member_off;
19562 	struct btf *btf;
19563 	const char *mname;
19564 	int i, err;
19565 
19566 	if (!prog->gpl_compatible) {
19567 		verbose(env, "struct ops programs must have a GPL compatible license\n");
19568 		return -EINVAL;
19569 	}
19570 
19571 	if (!prog->aux->attach_btf_id)
19572 		return -ENOTSUPP;
19573 
19574 	btf = prog->aux->attach_btf;
19575 	if (btf_is_module(btf)) {
19576 		/* Make sure st_ops is valid through the lifetime of env */
19577 		env->attach_btf_mod = btf_try_get_module(btf);
19578 		if (!env->attach_btf_mod) {
19579 			verbose(env, "struct_ops module %s is not found\n",
19580 				btf_get_name(btf));
19581 			return -ENOTSUPP;
19582 		}
19583 	}
19584 
19585 	btf_id = prog->aux->attach_btf_id;
19586 	st_ops_desc = bpf_struct_ops_find(btf, btf_id);
19587 	if (!st_ops_desc) {
19588 		verbose(env, "attach_btf_id %u is not a supported struct\n",
19589 			btf_id);
19590 		return -ENOTSUPP;
19591 	}
19592 	st_ops = st_ops_desc->st_ops;
19593 
19594 	t = st_ops_desc->type;
19595 	member_idx = prog->expected_attach_type;
19596 	if (member_idx >= btf_type_vlen(t)) {
19597 		verbose(env, "attach to invalid member idx %u of struct %s\n",
19598 			member_idx, st_ops->name);
19599 		return -EINVAL;
19600 	}
19601 
19602 	member = &btf_type_member(t)[member_idx];
19603 	mname = btf_name_by_offset(btf, member->name_off);
19604 	func_proto = btf_type_resolve_func_ptr(btf, member->type,
19605 					       NULL);
19606 	if (!func_proto) {
19607 		verbose(env, "attach to invalid member %s(@idx %u) of struct %s\n",
19608 			mname, member_idx, st_ops->name);
19609 		return -EINVAL;
19610 	}
19611 
19612 	member_off = __btf_member_bit_offset(t, member) / 8;
19613 	err = bpf_struct_ops_supported(st_ops, member_off);
19614 	if (err) {
19615 		verbose(env, "attach to unsupported member %s of struct %s\n",
19616 			mname, st_ops->name);
19617 		return err;
19618 	}
19619 
19620 	if (st_ops->check_member) {
19621 		err = st_ops->check_member(t, member, prog);
19622 
19623 		if (err) {
19624 			verbose(env, "attach to unsupported member %s of struct %s\n",
19625 				mname, st_ops->name);
19626 			return err;
19627 		}
19628 	}
19629 
19630 	if (prog->aux->priv_stack_requested && !bpf_jit_supports_private_stack()) {
19631 		verbose(env, "Private stack not supported by jit\n");
19632 		return -EACCES;
19633 	}
19634 
19635 	arg_info = &st_ops_desc->arg_info[member_idx];
19636 	for (i = 0; i < arg_info->cnt; i++) {
19637 		const struct bpf_ctx_arg_aux *info = &arg_info->info[i];
19638 
19639 		if (info->refcounted)
19640 			has_refcounted_arg = true;
19641 		if (base_type(info->reg_type) == PTR_TO_ARENA) {
19642 			if (!bpf_jit_supports_arena_args()) {
19643 				verbose(env, "JIT does not support arena arguments\n");
19644 				return -ENOTSUPP;
19645 			}
19646 			if (!prog->aux->arena) {
19647 				verbose(env,
19648 					"arena argument of %s requires a program with an associated arena\n",
19649 					mname);
19650 				return -EINVAL;
19651 			}
19652 		}
19653 	}
19654 
19655 	/* Tail call is not allowed for programs with refcounted arguments since we
19656 	 * cannot guarantee that valid refcounted kptrs will be passed to the callee.
19657 	 */
19658 	for (i = 0; i < env->subprog_cnt; i++) {
19659 		if (has_refcounted_arg && env->subprog_info[i].has_tail_call) {
19660 			verbose(env, "program with __ref argument cannot tail call\n");
19661 			return -EINVAL;
19662 		}
19663 	}
19664 
19665 	prog->aux->st_ops = st_ops;
19666 	prog->aux->attach_st_ops_member_off = member_off;
19667 
19668 	prog->aux->attach_func_proto = func_proto;
19669 	prog->aux->attach_func_name = mname;
19670 	env->ops = st_ops->verifier_ops;
19671 
19672 	return bpf_prog_ctx_arg_info_init(prog, arg_info->info, arg_info->cnt);
19673 }
19674 #define SECURITY_PREFIX "security_"
19675 
19676 #ifdef CONFIG_FUNCTION_ERROR_INJECTION
19677 
19678 /* list of non-sleepable functions that are otherwise on
19679  * ALLOW_ERROR_INJECTION list
19680  */
19681 BTF_SET_START(btf_non_sleepable_error_inject)
19682 /* Three functions below can be called from sleepable and non-sleepable context.
19683  * Assume non-sleepable from bpf safety point of view.
19684  */
BTF_ID(func,__filemap_add_folio)19685 BTF_ID(func, __filemap_add_folio)
19686 #ifdef CONFIG_FAIL_PAGE_ALLOC
19687 BTF_ID(func, should_fail_alloc_page)
19688 #endif
19689 #ifdef CONFIG_FAILSLAB
19690 BTF_ID(func, should_failslab)
19691 #endif
19692 BTF_SET_END(btf_non_sleepable_error_inject)
19693 
19694 static int check_non_sleepable_error_inject(u32 btf_id)
19695 {
19696 	return btf_id_set_contains(&btf_non_sleepable_error_inject, btf_id);
19697 }
19698 
check_attach_sleepable(u32 btf_id,unsigned long addr,const char * func_name)19699 static int check_attach_sleepable(u32 btf_id, unsigned long addr, const char *func_name)
19700 {
19701 	/* fentry/fexit/fmod_ret progs can be sleepable if they are
19702 	 * attached to ALLOW_ERROR_INJECTION and are not in denylist.
19703 	 */
19704 	if (!check_non_sleepable_error_inject(btf_id) &&
19705 	    within_error_injection_list(addr))
19706 		return 0;
19707 
19708 	return -EINVAL;
19709 }
19710 
check_attach_modify_return(unsigned long addr,const char * func_name)19711 static int check_attach_modify_return(unsigned long addr, const char *func_name)
19712 {
19713 	if (within_error_injection_list(addr) ||
19714 	    !strncmp(SECURITY_PREFIX, func_name, sizeof(SECURITY_PREFIX) - 1))
19715 		return 0;
19716 
19717 	return -EINVAL;
19718 }
19719 
19720 #else
19721 
19722 /* Unfortunately, the arch-specific prefixes are hard-coded in arch syscall code
19723  * so we need to hard-code them, too. Ftrace has arch_syscall_match_sym_name()
19724  * but that just compares two concrete function names.
19725  */
has_arch_syscall_prefix(const char * func_name)19726 static bool has_arch_syscall_prefix(const char *func_name)
19727 {
19728 #if defined(__x86_64__)
19729 	return !strncmp(func_name, "__x64_", 6);
19730 #elif defined(__i386__)
19731 	return !strncmp(func_name, "__ia32_", 7);
19732 #elif defined(__s390x__)
19733 	return !strncmp(func_name, "__s390x_", 8);
19734 #elif defined(__aarch64__)
19735 	return !strncmp(func_name, "__arm64_", 8);
19736 #elif defined(__riscv)
19737 	return !strncmp(func_name, "__riscv_", 8);
19738 #elif defined(__powerpc__) || defined(__powerpc64__)
19739 	return !strncmp(func_name, "sys_", 4);
19740 #elif defined(__loongarch__)
19741 	return !strncmp(func_name, "sys_", 4);
19742 #else
19743 	return false;
19744 #endif
19745 }
19746 
19747 /* Without error injection, allow sleepable and fmod_ret progs on syscalls. */
19748 
check_attach_sleepable(u32 btf_id,unsigned long addr,const char * func_name)19749 static int check_attach_sleepable(u32 btf_id, unsigned long addr, const char *func_name)
19750 {
19751 	if (has_arch_syscall_prefix(func_name))
19752 		return 0;
19753 
19754 	return -EINVAL;
19755 }
19756 
check_attach_modify_return(unsigned long addr,const char * func_name)19757 static int check_attach_modify_return(unsigned long addr, const char *func_name)
19758 {
19759 	if (has_arch_syscall_prefix(func_name) ||
19760 	    !strncmp(SECURITY_PREFIX, func_name, sizeof(SECURITY_PREFIX) - 1))
19761 		return 0;
19762 
19763 	return -EINVAL;
19764 }
19765 
19766 #endif /* CONFIG_FUNCTION_ERROR_INJECTION */
19767 
is_tracing_multi_id(const struct bpf_prog * prog,u32 btf_id)19768 static bool is_tracing_multi_id(const struct bpf_prog *prog, u32 btf_id)
19769 {
19770 	return is_tracing_multi(prog->expected_attach_type) && bpf_multi_func_btf_id[0] == btf_id;
19771 }
19772 
btf_id_allow_sleepable(u32 btf_id,unsigned long addr,const struct bpf_prog * prog,const struct btf * btf)19773 static int btf_id_allow_sleepable(u32 btf_id, unsigned long addr, const struct bpf_prog *prog,
19774 				  const struct btf *btf)
19775 {
19776 	const struct btf_type *t;
19777 	const char *tname;
19778 
19779 	if (!btf_is_kernel(btf))
19780 		return -EINVAL;
19781 
19782 	switch (prog->type) {
19783 	case BPF_PROG_TYPE_TRACING:
19784 		t = btf_type_by_id(btf, btf_id);
19785 		if (!t)
19786 			return -EINVAL;
19787 		tname = btf_name_by_offset(btf, t->name_off);
19788 		if (!tname)
19789 			return -EINVAL;
19790 
19791 		/*
19792 		 * *.multi sleepable programs will pass initial sleepable check,
19793 		 * the actual attached btf ids are checked later during the link
19794 		 * attachment.
19795 		 */
19796 		if (is_tracing_multi_id(prog, btf_id))
19797 			return 0;
19798 		if (!check_attach_sleepable(btf_id, addr, tname))
19799 			return 0;
19800 		/*
19801 		 * fentry/fexit/fmod_ret progs can also be sleepable if they are
19802 		 * in the fmodret id set with the KF_SLEEPABLE flag.
19803 		 */
19804 		else {
19805 			u32 *flags = btf_kfunc_is_modify_return(btf, btf_id, prog);
19806 
19807 			if (flags && (*flags & KF_SLEEPABLE))
19808 				return 0;
19809 		}
19810 		break;
19811 	case BPF_PROG_TYPE_LSM:
19812 		/*
19813 		 * LSM progs check that they are attached to bpf_lsm_*() funcs.
19814 		 * Only some of them are sleepable.
19815 		 */
19816 		if (bpf_lsm_is_sleepable_hook(btf_id))
19817 			return 0;
19818 		break;
19819 	default:
19820 		break;
19821 	}
19822 	return -EINVAL;
19823 }
19824 
19825 /*
19826  * Resolve the prototype describing a trace target's real ABI. A
19827  * KF_IMPLICIT_ARGS kfunc has its injected args stripped from the public
19828  * prototype, so use the _impl prototype; other targets use their own.
19829  */
19830 static const struct btf_type *
btf_attach_func_proto(struct bpf_verifier_log * log,struct btf * btf,u32 func_id)19831 btf_attach_func_proto(struct bpf_verifier_log *log, struct btf *btf, u32 func_id)
19832 {
19833 	const struct btf_type *func;
19834 	struct module *mod = NULL;
19835 	const char *name;
19836 	int implicit;
19837 
19838 	func = btf_type_by_id(btf, func_id);
19839 	if (!func || !btf_type_is_func(func))
19840 		return NULL;
19841 	name = btf_name_by_offset(btf, func->name_off);
19842 
19843 	/*
19844 	 * btf_kfunc_check_flag() reads kfunc_set_tab, which for a module is
19845 	 * stable only once it is live; hold a module ref across the read to
19846 	 * exclude a concurrent module load.
19847 	 */
19848 	if (btf_is_module(btf)) {
19849 		mod = btf_try_get_module(btf);
19850 		if (!mod)
19851 			return NULL;
19852 	}
19853 	implicit = btf_kfunc_check_flag(btf, func_id, KF_IMPLICIT_ARGS);
19854 	module_put(mod);
19855 
19856 	if (implicit == -EINVAL) {
19857 		bpf_log(log, "kfunc %s has inconsistent KF_IMPLICIT_ARGS\n", name);
19858 		return NULL;
19859 	}
19860 	if (implicit > 0)
19861 		return find_kfunc_impl_proto(log, btf, name);
19862 
19863 	return btf_type_by_id(btf, func->type);
19864 }
19865 
attach_uses_trampoline_retval(enum bpf_attach_type type)19866 static bool attach_uses_trampoline_retval(enum bpf_attach_type type)
19867 {
19868 	switch (type) {
19869 	case BPF_MODIFY_RETURN:
19870 	case BPF_TRACE_FEXIT:
19871 	case BPF_TRACE_FEXIT_MULTI:
19872 	case BPF_TRACE_FSESSION:
19873 	case BPF_TRACE_FSESSION_MULTI:
19874 		return true;
19875 	default:
19876 		return false;
19877 	}
19878 }
19879 
bpf_check_attach_target(struct bpf_verifier_log * log,const struct bpf_prog * prog,const struct bpf_prog * tgt_prog,u32 btf_id,struct bpf_attach_target_info * tgt_info)19880 int bpf_check_attach_target(struct bpf_verifier_log *log,
19881 			    const struct bpf_prog *prog,
19882 			    const struct bpf_prog *tgt_prog,
19883 			    u32 btf_id,
19884 			    struct bpf_attach_target_info *tgt_info)
19885 {
19886 	bool prog_extension = prog->type == BPF_PROG_TYPE_EXT;
19887 	bool prog_tracing = prog->type == BPF_PROG_TYPE_TRACING;
19888 	char trace_symbol[KSYM_SYMBOL_LEN];
19889 	const char prefix[] = "btf_trace_";
19890 	struct bpf_raw_event_map *btp;
19891 	int ret = 0, subprog = -1, i;
19892 	const struct btf_type *t;
19893 	bool conservative = true;
19894 	const char *tname, *fname;
19895 	struct btf *btf;
19896 	long addr = 0;
19897 	struct module *mod = NULL;
19898 
19899 	if (!btf_id) {
19900 		bpf_log(log, "Tracing programs must provide btf_id\n");
19901 		return -EINVAL;
19902 	}
19903 	btf = tgt_prog ? tgt_prog->aux->btf : prog->aux->attach_btf;
19904 	if (!btf) {
19905 		bpf_log(log,
19906 			"Tracing program can only be attached to another program annotated with BTF\n");
19907 		return -EINVAL;
19908 	}
19909 	t = btf_type_by_id(btf, btf_id);
19910 	if (!t) {
19911 		bpf_log(log, "attach_btf_id %u is invalid\n", btf_id);
19912 		return -EINVAL;
19913 	}
19914 	tname = btf_name_by_offset(btf, t->name_off);
19915 	if (!tname) {
19916 		bpf_log(log, "attach_btf_id %u doesn't have a name\n", btf_id);
19917 		return -EINVAL;
19918 	}
19919 	if (tgt_prog) {
19920 		struct bpf_prog_aux *aux = tgt_prog->aux;
19921 		bool tgt_changes_pkt_data;
19922 		bool tgt_might_sleep;
19923 
19924 		if (bpf_prog_is_dev_bound(prog->aux) &&
19925 		    !bpf_prog_dev_bound_match(prog, tgt_prog)) {
19926 			bpf_log(log, "Target program bound device mismatch");
19927 			return -EINVAL;
19928 		}
19929 
19930 		for (i = 0; i < aux->func_info_cnt; i++)
19931 			if (aux->func_info[i].type_id == btf_id) {
19932 				subprog = i;
19933 				break;
19934 			}
19935 		if (subprog == -1) {
19936 			bpf_log(log, "Subprog %s doesn't exist\n", tname);
19937 			return -EINVAL;
19938 		}
19939 		/*
19940 		 * A struct_ops indirect trampoline converts arena arguments
19941 		 * before invoking its program. A tracing or extension program
19942 		 * attached to the main program would see the converted offset as a
19943 		 * regular BTF pointer.
19944 		 */
19945 		if (subprog == 0 && bpf_prog_has_arena_ctx_arg(tgt_prog)) {
19946 			bpf_log(log, "Cannot attach to a target with arena context arguments\n");
19947 			return -EOPNOTSUPP;
19948 		}
19949 		if (aux->func && aux->func[subprog]->aux->exception_cb) {
19950 			bpf_log(log,
19951 				"%s programs cannot attach to exception callback\n",
19952 				prog_extension ? "Extension" : "Tracing");
19953 			return -EINVAL;
19954 		}
19955 		conservative = aux->func_info_aux[subprog].unreliable;
19956 		if (prog_extension) {
19957 			if (conservative) {
19958 				bpf_log(log,
19959 					"Cannot replace static functions\n");
19960 				return -EINVAL;
19961 			}
19962 			if (!prog->jit_requested) {
19963 				bpf_log(log,
19964 					"Extension programs should be JITed\n");
19965 				return -EINVAL;
19966 			}
19967 			tgt_changes_pkt_data = aux->func
19968 					       ? aux->func[subprog]->aux->changes_pkt_data
19969 					       : aux->changes_pkt_data;
19970 			if (prog->aux->changes_pkt_data && !tgt_changes_pkt_data) {
19971 				bpf_log(log,
19972 					"Extension program changes packet data, while original does not\n");
19973 				return -EINVAL;
19974 			}
19975 
19976 			tgt_might_sleep = aux->func
19977 					  ? aux->func[subprog]->aux->might_sleep
19978 					  : aux->might_sleep;
19979 			if (prog->aux->might_sleep && !tgt_might_sleep) {
19980 				bpf_log(log,
19981 					"Extension program may sleep, while original does not\n");
19982 				return -EINVAL;
19983 			}
19984 		}
19985 		if (!tgt_prog->jited) {
19986 			bpf_log(log, "Can attach to only JITed progs\n");
19987 			return -EINVAL;
19988 		}
19989 		if (prog_tracing) {
19990 			if (aux->attach_tracing_prog) {
19991 				/*
19992 				 * Target program is an fentry/fexit which is already attached
19993 				 * to another tracing program. More levels of nesting
19994 				 * attachment are not allowed.
19995 				 */
19996 				bpf_log(log, "Cannot nest tracing program attach more than once\n");
19997 				return -EINVAL;
19998 			}
19999 		} else if (tgt_prog->type == prog->type) {
20000 			/*
20001 			 * To avoid potential call chain cycles, prevent attaching of a
20002 			 * program extension to another extension. It's ok to attach
20003 			 * fentry/fexit to extension program.
20004 			 */
20005 			bpf_log(log, "Cannot recursively attach\n");
20006 			return -EINVAL;
20007 		}
20008 		if (tgt_prog->type == BPF_PROG_TYPE_TRACING &&
20009 		    prog_extension &&
20010 		    (tgt_prog->expected_attach_type == BPF_TRACE_FENTRY ||
20011 		     tgt_prog->expected_attach_type == BPF_TRACE_FEXIT ||
20012 		     tgt_prog->expected_attach_type == BPF_TRACE_FENTRY_MULTI ||
20013 		     tgt_prog->expected_attach_type == BPF_TRACE_FEXIT_MULTI ||
20014 		     tgt_prog->expected_attach_type == BPF_TRACE_FSESSION ||
20015 		     tgt_prog->expected_attach_type == BPF_TRACE_FSESSION_MULTI)) {
20016 			/* Program extensions can extend all program types
20017 			 * except fentry/fexit. The reason is the following.
20018 			 * The fentry/fexit programs are used for performance
20019 			 * analysis, stats and can be attached to any program
20020 			 * type. When extension program is replacing XDP function
20021 			 * it is necessary to allow performance analysis of all
20022 			 * functions. Both original XDP program and its program
20023 			 * extension. Hence attaching fentry/fexit to
20024 			 * BPF_PROG_TYPE_EXT is allowed. If extending of
20025 			 * fentry/fexit was allowed it would be possible to create
20026 			 * long call chain fentry->extension->fentry->extension
20027 			 * beyond reasonable stack size. Hence extending fentry
20028 			 * is not allowed.
20029 			 */
20030 			bpf_log(log, "Cannot extend fentry/fexit/fsession\n");
20031 			return -EINVAL;
20032 		}
20033 	} else {
20034 		if (prog_extension) {
20035 			bpf_log(log, "Cannot replace kernel functions\n");
20036 			return -EINVAL;
20037 		}
20038 	}
20039 
20040 	switch (prog->expected_attach_type) {
20041 	case BPF_TRACE_RAW_TP:
20042 		if (tgt_prog) {
20043 			bpf_log(log,
20044 				"Only FENTRY/FEXIT/FSESSION progs are attachable to another BPF prog\n");
20045 			return -EINVAL;
20046 		}
20047 		if (!btf_type_is_typedef(t)) {
20048 			bpf_log(log, "attach_btf_id %u is not a typedef\n",
20049 				btf_id);
20050 			return -EINVAL;
20051 		}
20052 		if (strncmp(prefix, tname, sizeof(prefix) - 1)) {
20053 			bpf_log(log, "attach_btf_id %u points to wrong type name %s\n",
20054 				btf_id, tname);
20055 			return -EINVAL;
20056 		}
20057 		tname += sizeof(prefix) - 1;
20058 
20059 		/* The func_proto of "btf_trace_##tname" is generated from typedef without argument
20060 		 * names. Thus using bpf_raw_event_map to get argument names.
20061 		 */
20062 		btp = bpf_get_raw_tracepoint(tname);
20063 		if (!btp)
20064 			return -EINVAL;
20065 		if (prog->sleepable && !tracepoint_is_faultable(btp->tp)) {
20066 			bpf_log(log, "Sleepable program cannot attach to non-faultable tracepoint %s\n",
20067 				tname);
20068 			bpf_put_raw_tracepoint(btp);
20069 			return -EINVAL;
20070 		}
20071 		fname = kallsyms_lookup((unsigned long)btp->bpf_func, NULL, NULL, NULL,
20072 					trace_symbol);
20073 		bpf_put_raw_tracepoint(btp);
20074 
20075 		if (fname)
20076 			ret = btf_find_by_name_kind(btf, fname, BTF_KIND_FUNC);
20077 
20078 		if (!fname || ret < 0) {
20079 			bpf_log(log, "Cannot find btf of tracepoint template, fall back to %s%s.\n",
20080 				prefix, tname);
20081 			t = btf_type_by_id(btf, t->type);
20082 			if (!btf_type_is_ptr(t))
20083 				/* should never happen in valid vmlinux build */
20084 				return -EINVAL;
20085 		} else {
20086 			t = btf_type_by_id(btf, ret);
20087 			if (!btf_type_is_func(t))
20088 				/* should never happen in valid vmlinux build */
20089 				return -EINVAL;
20090 		}
20091 
20092 		t = btf_type_by_id(btf, t->type);
20093 		if (!btf_type_is_func_proto(t))
20094 			/* should never happen in valid vmlinux build */
20095 			return -EINVAL;
20096 
20097 		break;
20098 	case BPF_TRACE_ITER:
20099 		if (!btf_type_is_func(t)) {
20100 			bpf_log(log, "attach_btf_id %u is not a function\n",
20101 				btf_id);
20102 			return -EINVAL;
20103 		}
20104 		t = btf_type_by_id(btf, t->type);
20105 		if (!btf_type_is_func_proto(t))
20106 			return -EINVAL;
20107 		ret = btf_distill_func_proto(log, btf, t, tname, &tgt_info->fmodel);
20108 		if (ret)
20109 			return ret;
20110 		break;
20111 	default:
20112 		if (!prog_extension)
20113 			return -EINVAL;
20114 		fallthrough;
20115 	case BPF_MODIFY_RETURN:
20116 	case BPF_LSM_MAC:
20117 	case BPF_LSM_CGROUP:
20118 	case BPF_TRACE_FENTRY:
20119 	case BPF_TRACE_FEXIT:
20120 	case BPF_TRACE_FSESSION:
20121 	case BPF_TRACE_FSESSION_MULTI:
20122 	case BPF_TRACE_FENTRY_MULTI:
20123 	case BPF_TRACE_FEXIT_MULTI:
20124 		if ((prog->expected_attach_type == BPF_TRACE_FSESSION ||
20125 		    prog->expected_attach_type == BPF_TRACE_FSESSION_MULTI) &&
20126 		    !bpf_jit_supports_fsession()) {
20127 			bpf_log(log, "JIT does not support fsession\n");
20128 			return -EOPNOTSUPP;
20129 		}
20130 		if (!btf_type_is_func(t)) {
20131 			bpf_log(log, "attach_btf_id %u is not a function\n",
20132 				btf_id);
20133 			return -EINVAL;
20134 		}
20135 		if (prog_extension &&
20136 		    btf_check_type_match(log, prog, btf, t))
20137 			return -EINVAL;
20138 		t = btf_attach_func_proto(log, btf, btf_id);
20139 		if (!t || !btf_type_is_func_proto(t))
20140 			return -EINVAL;
20141 
20142 		if ((prog->aux->saved_dst_prog_type || prog->aux->saved_dst_attach_type) &&
20143 		    (!tgt_prog || prog->aux->saved_dst_prog_type != tgt_prog->type ||
20144 		     prog->aux->saved_dst_attach_type != tgt_prog->expected_attach_type))
20145 			return -EINVAL;
20146 
20147 		if (tgt_prog && conservative)
20148 			t = NULL;
20149 
20150 		ret = btf_distill_func_proto(log, btf, t, tname, &tgt_info->fmodel);
20151 		if (ret < 0)
20152 			return ret;
20153 
20154 		if (tgt_info->fmodel.ret_size > 8 &&
20155 		    attach_uses_trampoline_retval(prog->expected_attach_type)) {
20156 			bpf_log(log,
20157 				"Attach to function %s with a >8 byte return value is not supported for this attach type\n",
20158 				tname);
20159 			return -EOPNOTSUPP;
20160 		}
20161 
20162 		/*
20163 		 * *.multi programs don't need an address during program
20164 		 * verification, we just take the module ref if needed.
20165 		 */
20166 		if (is_tracing_multi_id(prog, btf_id)) {
20167 			if (btf_is_module(btf)) {
20168 				mod = btf_try_get_module(btf);
20169 				if (!mod)
20170 					return -ENOENT;
20171 			}
20172 			addr = 0;
20173 		} else if (tgt_prog) {
20174 			if (subprog == 0)
20175 				addr = (long) tgt_prog->bpf_func;
20176 			else
20177 				addr = (long) tgt_prog->aux->func[subprog]->bpf_func;
20178 		} else {
20179 			if (btf_is_module(btf)) {
20180 				mod = btf_try_get_module(btf);
20181 				if (mod)
20182 					addr = find_kallsyms_symbol_value(mod, tname);
20183 				else
20184 					addr = 0;
20185 			} else {
20186 				addr = kallsyms_lookup_name(tname);
20187 			}
20188 			if (!addr) {
20189 				module_put(mod);
20190 				bpf_log(log,
20191 					"The address of function %s cannot be found\n",
20192 					tname);
20193 				return -ENOENT;
20194 			}
20195 		}
20196 
20197 		if (prog->sleepable) {
20198 			ret = btf_id_allow_sleepable(btf_id, addr, prog, btf);
20199 			if (ret) {
20200 				module_put(mod);
20201 				bpf_log(log, "%s is not sleepable\n", tname);
20202 				return ret;
20203 			}
20204 		} else if (prog->expected_attach_type == BPF_MODIFY_RETURN) {
20205 			if (tgt_prog) {
20206 				module_put(mod);
20207 				bpf_log(log, "can't modify return codes of BPF programs\n");
20208 				return -EINVAL;
20209 			}
20210 			ret = -EINVAL;
20211 			if (btf_kfunc_is_modify_return(btf, btf_id, prog) ||
20212 			    !check_attach_modify_return(addr, tname))
20213 				ret = 0;
20214 			if (ret) {
20215 				module_put(mod);
20216 				bpf_log(log, "%s() is not modifiable\n", tname);
20217 				return ret;
20218 			}
20219 		}
20220 
20221 		break;
20222 	}
20223 	tgt_info->tgt_addr = addr;
20224 	tgt_info->tgt_name = tname;
20225 	tgt_info->tgt_type = t;
20226 	tgt_info->tgt_mod = mod;
20227 	return 0;
20228 }
20229 
BTF_SET_START(btf_id_deny)20230 BTF_SET_START(btf_id_deny)
20231 BTF_ID_UNUSED
20232 #ifdef CONFIG_SMP
20233 BTF_ID(func, ___migrate_enable)
20234 BTF_ID(func, migrate_disable)
20235 BTF_ID(func, migrate_enable)
20236 #endif
20237 #if !defined CONFIG_PREEMPT_RCU && !defined CONFIG_TINY_RCU
20238 BTF_ID(func, rcu_read_unlock_strict)
20239 #endif
20240 #if defined(CONFIG_DEBUG_PREEMPT) || defined(CONFIG_TRACE_PREEMPT_TOGGLE)
20241 BTF_ID(func, preempt_count_add)
20242 BTF_ID(func, preempt_count_sub)
20243 #endif
20244 #ifdef CONFIG_PREEMPT_RCU
20245 BTF_ID(func, __rcu_read_lock)
20246 BTF_ID(func, __rcu_read_unlock)
20247 #endif
20248 BTF_SET_END(btf_id_deny)
20249 
20250 /* fexit and fmod_ret can't be used to attach to __noreturn functions.
20251  * Currently, we must manually list all __noreturn functions here. Once a more
20252  * robust solution is implemented, this workaround can be removed.
20253  */
20254 BTF_SET_START(noreturn_deny)
20255 #ifdef CONFIG_IA32_EMULATION
20256 BTF_ID(func, __ia32_sys_exit)
20257 BTF_ID(func, __ia32_sys_exit_group)
20258 #endif
20259 #ifdef CONFIG_KUNIT
20260 BTF_ID(func, __kunit_abort)
20261 BTF_ID(func, kunit_try_catch_throw)
20262 #endif
20263 #ifdef CONFIG_MODULES
20264 BTF_ID(func, __module_put_and_kthread_exit)
20265 #endif
20266 #ifdef CONFIG_X86_64
20267 BTF_ID(func, __x64_sys_exit)
20268 BTF_ID(func, __x64_sys_exit_group)
20269 #endif
20270 BTF_ID(func, do_exit)
20271 BTF_ID(func, do_group_exit)
20272 BTF_ID(func, kthread_complete_and_exit)
20273 BTF_ID(func, make_task_dead)
20274 BTF_SET_END(noreturn_deny)
20275 
20276 static bool can_be_sleepable(struct bpf_prog *prog)
20277 {
20278 	if (prog->type == BPF_PROG_TYPE_TRACING) {
20279 		switch (prog->expected_attach_type) {
20280 		case BPF_TRACE_FENTRY:
20281 		case BPF_TRACE_FEXIT:
20282 		case BPF_MODIFY_RETURN:
20283 		case BPF_TRACE_ITER:
20284 		case BPF_TRACE_FSESSION:
20285 		case BPF_TRACE_RAW_TP:
20286 		case BPF_TRACE_FENTRY_MULTI:
20287 		case BPF_TRACE_FEXIT_MULTI:
20288 		case BPF_TRACE_FSESSION_MULTI:
20289 			return true;
20290 		default:
20291 			return false;
20292 		}
20293 	}
20294 	if (prog->type == BPF_PROG_TYPE_LSM)
20295 		return prog->expected_attach_type != BPF_LSM_CGROUP;
20296 
20297 	return prog->type == BPF_PROG_TYPE_KPROBE /* only for uprobes */ ||
20298 	       prog->type == BPF_PROG_TYPE_STRUCT_OPS ||
20299 	       prog->type == BPF_PROG_TYPE_RAW_TRACEPOINT ||
20300 	       prog->type == BPF_PROG_TYPE_TRACEPOINT;
20301 }
20302 
check_attach_btf_id(struct bpf_verifier_env * env)20303 static int check_attach_btf_id(struct bpf_verifier_env *env)
20304 {
20305 	struct bpf_prog *prog = env->prog;
20306 	struct bpf_prog *tgt_prog = prog->aux->dst_prog;
20307 	struct bpf_attach_target_info tgt_info = {};
20308 	u32 btf_id = prog->aux->attach_btf_id;
20309 	struct bpf_trampoline *tr;
20310 	int ret;
20311 	u64 key;
20312 
20313 	if (prog->type == BPF_PROG_TYPE_SYSCALL) {
20314 		if (prog->sleepable)
20315 			/* attach_btf_id checked to be zero already */
20316 			return 0;
20317 		verbose(env, "Syscall programs can only be sleepable\n");
20318 		return -EINVAL;
20319 	}
20320 
20321 	if (prog->sleepable && !can_be_sleepable(prog)) {
20322 		verbose(env, "Program of this type cannot be sleepable\n");
20323 		return -EINVAL;
20324 	}
20325 
20326 	if (prog->type == BPF_PROG_TYPE_STRUCT_OPS)
20327 		return check_struct_ops_btf_id(env);
20328 
20329 	if (prog->type != BPF_PROG_TYPE_TRACING &&
20330 	    prog->type != BPF_PROG_TYPE_LSM &&
20331 	    prog->type != BPF_PROG_TYPE_EXT)
20332 		return 0;
20333 
20334 	ret = bpf_check_attach_target(&env->log, prog, tgt_prog, btf_id, &tgt_info);
20335 	if (ret)
20336 		return ret;
20337 
20338 	if (tgt_prog && prog->type == BPF_PROG_TYPE_EXT) {
20339 		/* to make freplace equivalent to their targets, they need to
20340 		 * inherit env->ops and expected_attach_type for the rest of the
20341 		 * verification
20342 		 */
20343 		env->ops = bpf_verifier_ops[tgt_prog->type];
20344 		prog->expected_attach_type = tgt_prog->expected_attach_type;
20345 	}
20346 
20347 	/* store info about the attachment target that will be used later */
20348 	prog->aux->attach_func_proto = tgt_info.tgt_type;
20349 	prog->aux->attach_func_name = tgt_info.tgt_name;
20350 	prog->aux->mod = tgt_info.tgt_mod;
20351 
20352 	if (tgt_prog) {
20353 		prog->aux->saved_dst_prog_type = tgt_prog->type;
20354 		prog->aux->saved_dst_attach_type = tgt_prog->expected_attach_type;
20355 	}
20356 
20357 	if (prog->expected_attach_type == BPF_TRACE_RAW_TP) {
20358 		prog->aux->attach_btf_trace = true;
20359 		return 0;
20360 	} else if (prog->expected_attach_type == BPF_TRACE_ITER) {
20361 		return bpf_iter_prog_supported(prog);
20362 	}
20363 
20364 	if (prog->type == BPF_PROG_TYPE_LSM) {
20365 		ret = bpf_lsm_verify_prog(&env->log, prog);
20366 		if (ret < 0)
20367 			return ret;
20368 	} else if (prog->type == BPF_PROG_TYPE_TRACING &&
20369 		   btf_id_set_contains(&btf_id_deny, btf_id)) {
20370 		verbose(env, "Attaching tracing programs to function '%s' is rejected.\n",
20371 			tgt_info.tgt_name);
20372 		return -EINVAL;
20373 	} else if ((prog->expected_attach_type == BPF_TRACE_FEXIT ||
20374 		   prog->expected_attach_type == BPF_TRACE_FSESSION ||
20375 		   prog->expected_attach_type == BPF_TRACE_FSESSION_MULTI ||
20376 		   prog->expected_attach_type == BPF_MODIFY_RETURN) &&
20377 		   btf_id_set_contains(&noreturn_deny, btf_id)) {
20378 		verbose(env, "Attaching fexit/fsession/fmod_ret to __noreturn function '%s' is rejected.\n",
20379 			tgt_info.tgt_name);
20380 		return -EINVAL;
20381 	}
20382 
20383 	/*
20384 	 * We don't get trampoline for tracing_multi programs at this point,
20385 	 * it's done when tracing_multi link is created.
20386 	 */
20387 	if (prog->type == BPF_PROG_TYPE_TRACING &&
20388 	    is_tracing_multi(prog->expected_attach_type))
20389 		return 0;
20390 
20391 	key = bpf_trampoline_compute_key(tgt_prog, prog->aux->attach_btf, btf_id);
20392 	tr = bpf_trampoline_get(key, &tgt_info);
20393 	if (!tr)
20394 		return -ENOMEM;
20395 
20396 	if (tgt_prog && tgt_prog->aux->tail_call_reachable)
20397 		bpf_trampoline_set_flags(tr, BPF_TRAMP_F_TAIL_CALL_CTX);
20398 
20399 	prog->aux->dst_trampoline = tr;
20400 	return 0;
20401 }
20402 
bpf_check_attach_btf_id_multi(struct btf * btf,struct bpf_prog * prog,u32 btf_id,struct bpf_attach_target_info * tgt_info)20403 int bpf_check_attach_btf_id_multi(struct btf *btf, struct bpf_prog *prog, u32 btf_id,
20404 				  struct bpf_attach_target_info *tgt_info)
20405 {
20406 	const struct btf_type *t;
20407 	unsigned long addr;
20408 	const char *tname;
20409 	int err;
20410 
20411 	if (!btf_id || !btf)
20412 		return -EINVAL;
20413 
20414 	/* Check noreturn attachment. */
20415 	if ((prog->expected_attach_type == BPF_TRACE_FEXIT_MULTI ||
20416 	     prog->expected_attach_type == BPF_TRACE_FSESSION_MULTI) &&
20417 	     btf_id_set_contains(&noreturn_deny, btf_id))
20418 		return -EINVAL;
20419 	/* Check denied attachment. */
20420 	if (btf_id_set_contains(&btf_id_deny, btf_id))
20421 		return -EINVAL;
20422 
20423 	/* Check and get function target data. */
20424 	t = btf_type_by_id(btf, btf_id);
20425 	if (!t)
20426 		return -EINVAL;
20427 	tname = btf_name_by_offset(btf, t->name_off);
20428 	if (!tname)
20429 		return -EINVAL;
20430 	t = btf_attach_func_proto(NULL, btf, btf_id);
20431 	if (!t || !btf_type_is_func_proto(t))
20432 		return -EINVAL;
20433 	err = btf_distill_func_proto(NULL, btf, t, tname, &tgt_info->fmodel);
20434 	if (err < 0)
20435 		return err;
20436 	if (tgt_info->fmodel.ret_size > 8 &&
20437 	    attach_uses_trampoline_retval(prog->expected_attach_type))
20438 		return -EOPNOTSUPP;
20439 	if (btf_is_module(btf)) {
20440 		/* The bpf program already holds reference to module. */
20441 		if (WARN_ON_ONCE(!prog->aux->mod))
20442 			return -EINVAL;
20443 		addr = find_kallsyms_symbol_value(prog->aux->mod, tname);
20444 	} else {
20445 		addr = kallsyms_lookup_name(tname);
20446 	}
20447 	if (!addr || !ftrace_location(addr))
20448 		return -ENOENT;
20449 
20450 	/* Check sleepable program attachment. */
20451 	if (prog->sleepable) {
20452 		err = btf_id_allow_sleepable(btf_id, addr, prog, btf);
20453 		if (err)
20454 			return err;
20455 	}
20456 	tgt_info->tgt_addr = addr;
20457 	return 0;
20458 }
20459 
bpf_get_btf_vmlinux(void)20460 struct btf *bpf_get_btf_vmlinux(void)
20461 {
20462 	/* Pairs with the smp_store_release() on the parse path below. */
20463 	struct btf *btf = smp_load_acquire(&btf_vmlinux);
20464 
20465 	if (!btf && IS_ENABLED(CONFIG_DEBUG_INFO_BTF)) {
20466 		mutex_lock(&btf_vmlinux_lock);
20467 		btf = btf_vmlinux;
20468 		if (!btf) {
20469 			btf = btf_parse_vmlinux();
20470 			/*
20471 			 * Order the parsed BTF contents and the globals the
20472 			 * parse populated (e.g. bpf_ctx_convert.t) before
20473 			 * the pointer publication. Pairs with the acquire
20474 			 * on the lockless fast path above.
20475 			 */
20476 			smp_store_release(&btf_vmlinux, btf);
20477 		}
20478 		mutex_unlock(&btf_vmlinux_lock);
20479 	}
20480 	return btf;
20481 }
20482 
20483 /*
20484  * The add_fd_from_fd_array() is executed only if fd_array_cnt is non-zero. In
20485  * this case expect that every file descriptor in the array is either a map or
20486  * a BTF. Everything else is considered to be trash.
20487  */
add_fd_from_fd_array(struct bpf_verifier_env * env,u32 idx,int fd)20488 static int add_fd_from_fd_array(struct bpf_verifier_env *env, u32 idx, int fd)
20489 {
20490 	struct bpf_map *map;
20491 	struct btf *btf;
20492 	CLASS(fd, f)(fd);
20493 	int err;
20494 
20495 	map = __bpf_map_get(f);
20496 	if (!IS_ERR(map)) {
20497 		err = __add_used_map(env, map);
20498 		if (err < 0)
20499 			return err;
20500 		fd_slot_set_map(&env->fd_array[idx], map);
20501 		return 0;
20502 	}
20503 
20504 	btf = __btf_get_by_fd(f);
20505 	if (!IS_ERR(btf)) {
20506 		btf_get(btf);
20507 		err = __add_used_btf(env, btf);
20508 		if (err < 0)
20509 			return err;
20510 		fd_slot_set_btf(&env->fd_array[idx], btf);
20511 		return 0;
20512 	}
20513 
20514 	verbose(env, "fd %d is not pointing to valid bpf_map or btf\n", fd);
20515 	return PTR_ERR(map);
20516 }
20517 
20518 /*
20519  * A continuous fd_array is resolved into an in-memory cache with one slot
20520  * per entry. The bound here is deliberately generous and not derived from
20521  * the per-program object limits: Duplicate entries /are/ permitted, and
20522  * the number of distinct maps and BTFs a program can bind is enforced when
20523  * each entry is resolved by __add_used_map() and __add_used_btf().
20524  */
20525 #define MAX_FD_ARRAY_CNT 4096
20526 
process_fd_array_continuous(struct bpf_verifier_env * env,bpfptr_t fd_array,u32 cnt)20527 static int process_fd_array_continuous(struct bpf_verifier_env *env,
20528 				       bpfptr_t fd_array, u32 cnt)
20529 {
20530 	int fd, ret;
20531 	u32 i;
20532 
20533 	if (cnt > MAX_FD_ARRAY_CNT) {
20534 		verbose(env, "fd_array has too many entries (%u, max %u)\n",
20535 			cnt, MAX_FD_ARRAY_CNT);
20536 		return -E2BIG;
20537 	}
20538 
20539 	env->fd_array = kvcalloc(cnt, sizeof(*env->fd_array),
20540 				 GFP_KERNEL_ACCOUNT);
20541 	if (!env->fd_array)
20542 		return -ENOMEM;
20543 	env->fd_array_cnt = cnt;
20544 	for (i = 0; i < cnt; i++) {
20545 		if (copy_from_bpfptr_offset(&fd, fd_array,
20546 					    (size_t)i * sizeof(fd), sizeof(fd)))
20547 			return -EFAULT;
20548 		ret = add_fd_from_fd_array(env, i, fd);
20549 		if (ret)
20550 			return ret;
20551 	}
20552 	return 0;
20553 }
20554 
process_fd_array(struct bpf_verifier_env * env,union bpf_attr * attr,bpfptr_t uattr)20555 static int process_fd_array(struct bpf_verifier_env *env,
20556 			    union bpf_attr *attr, bpfptr_t uattr)
20557 {
20558 	bpfptr_t fd_array = make_bpfptr(attr->fd_array, uattr.is_kernel);
20559 
20560 	if (bpfptr_is_null(fd_array)) {
20561 		if (attr->fd_array_cnt) {
20562 			verbose(env, "fd_array_cnt %u without fd_array is invalid\n",
20563 				attr->fd_array_cnt);
20564 			return -EINVAL;
20565 		}
20566 		return 0;
20567 	}
20568 	/*
20569 	 * New API: the caller passes fd_array_cnt and a continuous array that
20570 	 * is resolved and bound up front. Legacy API (no fd_array_cnt): keep
20571 	 * the caller's array and resolve entries on the spot at each reference.
20572 	 */
20573 	if (attr->fd_array_cnt)
20574 		return process_fd_array_continuous(env, fd_array,
20575 						   attr->fd_array_cnt);
20576 	env->fd_array_raw = fd_array;
20577 	return 0;
20578 }
20579 
20580 /* replace a generic kfunc with a specialized version if necessary */
specialize_kfunc(struct bpf_verifier_env * env,struct bpf_kfunc_desc * desc,int insn_idx)20581 static int specialize_kfunc(struct bpf_verifier_env *env, struct bpf_kfunc_desc *desc, int insn_idx)
20582 {
20583 	struct bpf_prog *prog = env->prog;
20584 	bool seen_direct_write;
20585 	void *xdp_kfunc;
20586 	bool is_rdonly;
20587 	u32 func_id = desc->func_id;
20588 	u16 offset = desc->offset;
20589 	unsigned long addr = desc->addr;
20590 
20591 	if (offset) /* return if module BTF is used */
20592 		return 0;
20593 
20594 	if (bpf_dev_bound_kfunc_id(func_id)) {
20595 		xdp_kfunc = bpf_dev_bound_resolve_kfunc(prog, func_id);
20596 		if (xdp_kfunc)
20597 			addr = (unsigned long)xdp_kfunc;
20598 		/* fallback to default kfunc when not supported by netdev */
20599 	} else if (func_id == special_kfunc_list[KF_bpf_dynptr_from_skb]) {
20600 		seen_direct_write = env->seen_direct_write;
20601 		is_rdonly = !may_access_direct_pkt_data(env, NULL, BPF_WRITE);
20602 
20603 		if (is_rdonly)
20604 			addr = (unsigned long)bpf_dynptr_from_skb_rdonly;
20605 
20606 		/* restore env->seen_direct_write to its original value, since
20607 		 * may_access_direct_pkt_data mutates it
20608 		 */
20609 		env->seen_direct_write = seen_direct_write;
20610 	} else if (func_id == special_kfunc_list[KF_bpf_set_dentry_xattr]) {
20611 		if (bpf_lsm_has_d_inode_locked(prog))
20612 			addr = (unsigned long)bpf_set_dentry_xattr_locked;
20613 	} else if (func_id == special_kfunc_list[KF_bpf_remove_dentry_xattr]) {
20614 		if (bpf_lsm_has_d_inode_locked(prog))
20615 			addr = (unsigned long)bpf_remove_dentry_xattr_locked;
20616 	} else if (func_id == special_kfunc_list[KF_bpf_dynptr_from_file]) {
20617 		if (!env->insn_aux_data[insn_idx].non_sleepable)
20618 			addr = (unsigned long)bpf_dynptr_from_file_sleepable;
20619 	} else if (func_id == special_kfunc_list[KF_bpf_arena_alloc_pages]) {
20620 		if (env->insn_aux_data[insn_idx].non_sleepable)
20621 			addr = (unsigned long)bpf_arena_alloc_pages_non_sleepable;
20622 	} else if (func_id == special_kfunc_list[KF_bpf_arena_free_pages]) {
20623 		if (env->insn_aux_data[insn_idx].non_sleepable)
20624 			addr = (unsigned long)bpf_arena_free_pages_non_sleepable;
20625 	}
20626 	desc->addr = addr;
20627 	return 0;
20628 }
20629 
__fixup_collection_insert_kfunc(struct bpf_insn_aux_data * insn_aux,u16 struct_meta_reg,u16 node_offset_reg,struct bpf_insn * insn,struct bpf_insn * insn_buf,int * cnt)20630 static void __fixup_collection_insert_kfunc(struct bpf_insn_aux_data *insn_aux,
20631 					    u16 struct_meta_reg,
20632 					    u16 node_offset_reg,
20633 					    struct bpf_insn *insn,
20634 					    struct bpf_insn *insn_buf,
20635 					    int *cnt)
20636 {
20637 	struct btf_struct_meta *kptr_struct_meta = insn_aux->kptr_struct_meta;
20638 	struct bpf_insn addr[2] = { BPF_LD_IMM64(struct_meta_reg, (long)kptr_struct_meta) };
20639 
20640 	insn_buf[0] = addr[0];
20641 	insn_buf[1] = addr[1];
20642 	insn_buf[2] = BPF_MOV64_IMM(node_offset_reg, insn_aux->insert_off);
20643 	insn_buf[3] = *insn;
20644 	*cnt = 4;
20645 }
20646 
bpf_fixup_kfunc_call(struct bpf_verifier_env * env,struct bpf_insn * insn,struct bpf_insn * insn_buf,int insn_idx,int * cnt)20647 int bpf_fixup_kfunc_call(struct bpf_verifier_env *env, struct bpf_insn *insn,
20648 		     struct bpf_insn *insn_buf, int insn_idx, int *cnt)
20649 {
20650 	struct bpf_kfunc_desc *desc;
20651 	int err;
20652 
20653 	if (!insn->imm) {
20654 		verbose(env, "invalid kernel function call not eliminated in verifier pass\n");
20655 		return -EINVAL;
20656 	}
20657 
20658 	*cnt = 0;
20659 
20660 	/* insn->imm has the btf func_id. Replace it with an offset relative to
20661 	 * __bpf_call_base, unless the JIT needs to call functions that are
20662 	 * further than 32 bits away (bpf_jit_supports_far_kfunc_call()).
20663 	 */
20664 	desc = find_kfunc_desc(env->prog, insn->imm, insn->off);
20665 	if (!desc) {
20666 		verifier_bug(env, "kernel function descriptor not found for func_id %u",
20667 			     insn->imm);
20668 		return -EFAULT;
20669 	}
20670 
20671 	err = specialize_kfunc(env, desc, insn_idx);
20672 	if (err)
20673 		return err;
20674 
20675 	if (!bpf_jit_supports_far_kfunc_call())
20676 		insn->imm = BPF_CALL_IMM(desc->addr);
20677 
20678 	if (is_bpf_obj_new_kfunc(desc->func_id) || is_bpf_percpu_obj_new_kfunc(desc->func_id)) {
20679 		struct btf_struct_meta *kptr_struct_meta = env->insn_aux_data[insn_idx].kptr_struct_meta;
20680 		struct bpf_insn addr[2] = { BPF_LD_IMM64(BPF_REG_2, (long)kptr_struct_meta) };
20681 		u64 obj_new_size = env->insn_aux_data[insn_idx].obj_new_size;
20682 
20683 		if (is_bpf_percpu_obj_new_kfunc(desc->func_id) && kptr_struct_meta) {
20684 			verifier_bug(env, "NULL kptr_struct_meta expected at insn_idx %d",
20685 				     insn_idx);
20686 			return -EFAULT;
20687 		}
20688 
20689 		insn_buf[0] = BPF_MOV64_IMM(BPF_REG_1, obj_new_size);
20690 		insn_buf[1] = addr[0];
20691 		insn_buf[2] = addr[1];
20692 		insn_buf[3] = *insn;
20693 		*cnt = 4;
20694 	} else if (is_bpf_obj_drop_kfunc(desc->func_id) ||
20695 		   is_bpf_percpu_obj_drop_kfunc(desc->func_id) ||
20696 		   is_bpf_refcount_acquire_kfunc(desc->func_id)) {
20697 		struct btf_struct_meta *kptr_struct_meta = env->insn_aux_data[insn_idx].kptr_struct_meta;
20698 		struct bpf_insn addr[2] = { BPF_LD_IMM64(BPF_REG_2, (long)kptr_struct_meta) };
20699 
20700 		if (is_bpf_percpu_obj_drop_kfunc(desc->func_id) && kptr_struct_meta) {
20701 			verifier_bug(env, "NULL kptr_struct_meta expected at insn_idx %d",
20702 				     insn_idx);
20703 			return -EFAULT;
20704 		}
20705 
20706 		if (is_bpf_refcount_acquire_kfunc(desc->func_id) && !kptr_struct_meta) {
20707 			verifier_bug(env, "kptr_struct_meta expected at insn_idx %d",
20708 				     insn_idx);
20709 			return -EFAULT;
20710 		}
20711 
20712 		insn_buf[0] = addr[0];
20713 		insn_buf[1] = addr[1];
20714 		insn_buf[2] = *insn;
20715 		*cnt = 3;
20716 	} else if (is_bpf_list_push_kfunc(desc->func_id) ||
20717 		   is_bpf_rbtree_add_kfunc(desc->func_id)) {
20718 		struct btf_struct_meta *kptr_struct_meta = env->insn_aux_data[insn_idx].kptr_struct_meta;
20719 		int struct_meta_reg = BPF_REG_3;
20720 		int node_offset_reg = BPF_REG_4;
20721 
20722 		/* list_add/rbtree_add have an extra arg (prev/less),
20723 		 * so args-to-fixup are in diff regs.
20724 		 */
20725 		if (desc->func_id == special_kfunc_list[KF_bpf_list_add] ||
20726 		    is_bpf_rbtree_add_kfunc(desc->func_id)) {
20727 			struct_meta_reg = BPF_REG_4;
20728 			node_offset_reg = BPF_REG_5;
20729 		}
20730 
20731 		if (!kptr_struct_meta) {
20732 			verifier_bug(env, "kptr_struct_meta expected at insn_idx %d",
20733 				     insn_idx);
20734 			return -EFAULT;
20735 		}
20736 
20737 		__fixup_collection_insert_kfunc(&env->insn_aux_data[insn_idx], struct_meta_reg,
20738 						node_offset_reg, insn, insn_buf, cnt);
20739 	} else if (desc->func_id == special_kfunc_list[KF_bpf_cast_to_kern_ctx] ||
20740 		   desc->func_id == special_kfunc_list[KF_bpf_rdonly_cast]) {
20741 		insn_buf[0] = BPF_MOV64_REG(BPF_REG_0, BPF_REG_1);
20742 		*cnt = 1;
20743 	} else if (desc->func_id == special_kfunc_list[KF_bpf_session_is_return] &&
20744 		   (env->prog->expected_attach_type == BPF_TRACE_FSESSION ||
20745 		    env->prog->expected_attach_type == BPF_TRACE_FSESSION_MULTI)) {
20746 
20747 		/*
20748 		 * inline the bpf_session_is_return() for fsession:
20749 		 *   bool bpf_session_is_return(void *ctx)
20750 		 *   {
20751 		 *       return (((u64 *)ctx)[-1] >> BPF_TRAMP_IS_RETURN_SHIFT) & 1;
20752 		 *   }
20753 		 */
20754 		insn_buf[0] = BPF_LDX_MEM(BPF_DW, BPF_REG_0, BPF_REG_1, -8);
20755 		insn_buf[1] = BPF_ALU64_IMM(BPF_RSH, BPF_REG_0, BPF_TRAMP_IS_RETURN_SHIFT);
20756 		insn_buf[2] = BPF_ALU64_IMM(BPF_AND, BPF_REG_0, 1);
20757 		*cnt = 3;
20758 	} else if (desc->func_id == special_kfunc_list[KF_bpf_session_cookie] &&
20759 		   (env->prog->expected_attach_type == BPF_TRACE_FSESSION ||
20760 		    env->prog->expected_attach_type == BPF_TRACE_FSESSION_MULTI)) {
20761 		/*
20762 		 * inline bpf_session_cookie() for fsession:
20763 		 *   __u64 *bpf_session_cookie(void *ctx)
20764 		 *   {
20765 		 *       u64 off = (((u64 *)ctx)[-1] >> BPF_TRAMP_COOKIE_INDEX_SHIFT) & 0xFF;
20766 		 *       return &((u64 *)ctx)[-off];
20767 		 *   }
20768 		 */
20769 		insn_buf[0] = BPF_LDX_MEM(BPF_DW, BPF_REG_0, BPF_REG_1, -8);
20770 		insn_buf[1] = BPF_ALU64_IMM(BPF_RSH, BPF_REG_0, BPF_TRAMP_COOKIE_INDEX_SHIFT);
20771 		insn_buf[2] = BPF_ALU64_IMM(BPF_AND, BPF_REG_0, 0xFF);
20772 		insn_buf[3] = BPF_ALU64_IMM(BPF_LSH, BPF_REG_0, 3);
20773 		insn_buf[4] = BPF_ALU64_REG(BPF_SUB, BPF_REG_0, BPF_REG_1);
20774 		insn_buf[5] = BPF_ALU64_IMM(BPF_NEG, BPF_REG_0, 0);
20775 		*cnt = 6;
20776 	} else if (desc->func_id == special_kfunc_list[KF_bpf_iter_num_new]) {
20777 		/* inline bpf_iter_num_new(&it, start, end); R1=&it, R2=start, R3=end */
20778 		int i = 0;
20779 
20780 		/* if (start > end) goto einval; */
20781 		insn_buf[i++] = BPF_JMP32_REG(BPF_JSGT, BPF_REG_2, BPF_REG_3, 8);
20782 		/* r0 = (u32)end - (u32)start; if (r0 > BPF_MAX_LOOPS) goto e2big; */
20783 		insn_buf[i++] = BPF_MOV32_REG(BPF_REG_0, BPF_REG_3);
20784 		insn_buf[i++] = BPF_ALU32_REG(BPF_SUB, BPF_REG_0, BPF_REG_2);
20785 		insn_buf[i++] = BPF_JMP_IMM(BPF_JGT, BPF_REG_0, BPF_MAX_LOOPS, 8);
20786 		/* s->cur = start - 1; s->end = end; return 0; */
20787 		insn_buf[i++] = BPF_ALU32_IMM(BPF_ADD, BPF_REG_2, -1);
20788 		insn_buf[i++] = BPF_STX_MEM(BPF_W, BPF_REG_1, BPF_REG_2, 0);
20789 		insn_buf[i++] = BPF_STX_MEM(BPF_W, BPF_REG_1, BPF_REG_3, 4);
20790 		insn_buf[i++] = BPF_MOV64_IMM(BPF_REG_0, 0);
20791 		insn_buf[i++] = BPF_JMP_A(5);
20792 		/* einval: s->cur = s->end = 0; return -EINVAL; */
20793 		insn_buf[i++] = BPF_ST_MEM(BPF_DW, BPF_REG_1, 0, 0);
20794 		insn_buf[i++] = BPF_MOV64_IMM(BPF_REG_0, -EINVAL);
20795 		insn_buf[i++] = BPF_JMP_A(2);
20796 		/* e2big: s->cur = s->end = 0; return -E2BIG; */
20797 		insn_buf[i++] = BPF_ST_MEM(BPF_DW, BPF_REG_1, 0, 0);
20798 		insn_buf[i++] = BPF_MOV64_IMM(BPF_REG_0, -E2BIG);
20799 		*cnt = i;
20800 	} else if (desc->func_id == special_kfunc_list[KF_bpf_iter_num_next]) {
20801 		/* inline bpf_iter_num_next(&it); R1=&it, returns &s->cur or NULL */
20802 		int i = 0;
20803 
20804 		/* r0 = s->cur + 1; if ((s32)r0 >= s->end) goto done; */
20805 		insn_buf[i++] = BPF_LDX_MEM(BPF_W, BPF_REG_0, BPF_REG_1, 0);
20806 		insn_buf[i++] = BPF_ALU32_IMM(BPF_ADD, BPF_REG_0, 1);
20807 		insn_buf[i++] = BPF_LDX_MEM(BPF_W, BPF_REG_2, BPF_REG_1, 4);
20808 		insn_buf[i++] = BPF_JMP32_REG(BPF_JSGE, BPF_REG_0, BPF_REG_2, 3);
20809 		/* s->cur = r0; return &s->cur; */
20810 		insn_buf[i++] = BPF_STX_MEM(BPF_W, BPF_REG_1, BPF_REG_0, 0);
20811 		insn_buf[i++] = BPF_MOV64_REG(BPF_REG_0, BPF_REG_1);
20812 		insn_buf[i++] = BPF_JMP_A(2);
20813 		/* done: s->cur = s->end = 0; return NULL; */
20814 		insn_buf[i++] = BPF_ST_MEM(BPF_DW, BPF_REG_1, 0, 0);
20815 		insn_buf[i++] = BPF_MOV64_IMM(BPF_REG_0, 0);
20816 		*cnt = i;
20817 	} else if (desc->func_id == special_kfunc_list[KF_bpf_iter_num_destroy]) {
20818 		/* bpf_iter_num_destroy() is a no-op; emit a nop to drop the call */
20819 		insn_buf[0] = BPF_JMP_A(0);
20820 		*cnt = 1;
20821 	}
20822 
20823 	if (env->insn_aux_data[insn_idx].arg_prog) {
20824 		u32 regno = env->insn_aux_data[insn_idx].arg_prog;
20825 		struct bpf_insn ld_addrs[2] = { BPF_LD_IMM64(regno, (long)env->prog->aux) };
20826 		int idx = *cnt;
20827 
20828 		insn_buf[idx++] = ld_addrs[0];
20829 		insn_buf[idx++] = ld_addrs[1];
20830 		insn_buf[idx++] = *insn;
20831 		*cnt = idx;
20832 	}
20833 	return 0;
20834 }
20835 
bpf_classify_keyring(s32 keyring_id)20836 static enum bpf_sig_keyring bpf_classify_keyring(s32 keyring_id)
20837 {
20838 	switch (keyring_id) {
20839 	case 0:
20840 		return BPF_SIG_KEYRING_BUILTIN;
20841 	case (s32)(unsigned long)VERIFY_USE_SECONDARY_KEYRING:
20842 		return BPF_SIG_KEYRING_SECONDARY;
20843 	case (s32)(unsigned long)VERIFY_USE_PLATFORM_KEYRING:
20844 		return BPF_SIG_KEYRING_PLATFORM;
20845 	default:
20846 		return BPF_SIG_KEYRING_USER;
20847 	}
20848 }
20849 
20850 /*
20851  * Verify the PKCS#7 signature of a loaded program. Called from bpf_check()
20852  * once the program's metadata maps have been resolved into used_maps, so
20853  * the exact maps folded into the signature are the ones the program binds.
20854  *
20855  * The signature covers the instructions followed by the frozen contents of
20856  * each map, in @maps order: insns || map_0 || map_1 || [...]. On success the
20857  * verdict and keyring info are recorded on prog->aux.
20858  */
bpf_prog_verify_signature(struct bpf_verifier_env * env,union bpf_attr * attr,bool is_kernel)20859 static int bpf_prog_verify_signature(struct bpf_verifier_env *env,
20860 				     union bpf_attr *attr, bool is_kernel)
20861 {
20862 	bpfptr_t usig = make_bpfptr(attr->signature, is_kernel);
20863 	struct bpf_dynptr_kern sig_ptr, data_ptr;
20864 	struct bpf_prog *prog = env->prog;
20865 	struct bpf_map **maps = env->used_maps;
20866 	struct bpf_key *key = NULL;
20867 	void *sig, *data = NULL;
20868 	u32 map_cnt = env->used_map_cnt;
20869 	u32 i, off, insns_sz;
20870 	u64 data_sz;
20871 	int err = 0;
20872 
20873 	/*
20874 	 * Don't attempt to use kmalloc_large or vmalloc for signatures.
20875 	 * Practical signature for BPF program should be below this limit.
20876 	 */
20877 	if (!attr->signature_size ||
20878 	    attr->signature_size > KMALLOC_MAX_CACHE_SIZE)
20879 		return -EINVAL;
20880 	if (system_keyring_id_check(attr->keyring_id) == 0)
20881 		key = bpf_lookup_system_key(attr->keyring_id);
20882 	else
20883 		key = bpf_lookup_user_key(attr->keyring_id, 0);
20884 	if (!key) {
20885 		verbose(env, "cannot resolve signing keyring with keyring_id %d\n",
20886 			attr->keyring_id);
20887 		return -EINVAL;
20888 	}
20889 
20890 	sig = kvmemdup_bpfptr(usig, attr->signature_size);
20891 	if (IS_ERR(sig)) {
20892 		bpf_key_put(key);
20893 		return PTR_ERR(sig);
20894 	}
20895 
20896 	insns_sz = prog->len * sizeof(struct bpf_insn);
20897 	data_sz = insns_sz;
20898 	for (i = 0; i < map_cnt; i++) {
20899 		struct bpf_map *map = maps[i];
20900 
20901 		if (map->map_type != BPF_MAP_TYPE_ARRAY ||
20902 		    !map->ops->map_direct_value_addr) {
20903 			verbose(env, "signed program metadata map '%s' must be an array\n",
20904 				map->name);
20905 			err = -EINVAL;
20906 			goto out;
20907 		}
20908 		if (!READ_ONCE(map->frozen)) {
20909 			verbose(env, "signed program metadata map '%s' must be frozen\n",
20910 				map->name);
20911 			err = -EPERM;
20912 			goto out;
20913 		}
20914 		if (bpf_map_write_active(map)) {
20915 			verbose(env, "signed program metadata map '%s' has active writers\n",
20916 				map->name);
20917 			err = -EBUSY;
20918 			goto out;
20919 		}
20920 		if (!map->excl_prog_sha) {
20921 			verbose(env, "signed program metadata map '%s' must be exclusive\n",
20922 				map->name);
20923 			err = -EPERM;
20924 			goto out;
20925 		}
20926 		data_sz += map->value_size;
20927 	}
20928 	if (bpf_dynptr_check_size(data_sz)) {
20929 		verbose(env, "signed payload too large: %llu bytes\n", data_sz);
20930 		err = -E2BIG;
20931 		goto out;
20932 	}
20933 	data = kvmalloc(data_sz, GFP_KERNEL_ACCOUNT | __GFP_ZERO);
20934 	if (!data) {
20935 		err = -ENOMEM;
20936 		goto out;
20937 	}
20938 	memcpy(data, prog->insnsi, insns_sz);
20939 	off = insns_sz;
20940 	for (i = 0; i < map_cnt; i++) {
20941 		struct bpf_map *map = maps[i];
20942 		u64 addr;
20943 
20944 		err = map->ops->map_direct_value_addr(map, &addr, 0);
20945 		if (err) {
20946 			verbose(env, "failed to read signed metadata map '%s': %d\n",
20947 				map->name, err);
20948 			goto out;
20949 		}
20950 		memcpy(data + off, (void *)(unsigned long)addr,
20951 		       map->value_size);
20952 		off += map->value_size;
20953 	}
20954 
20955 	bpf_dynptr_init(&data_ptr, data, BPF_DYNPTR_TYPE_LOCAL, 0, data_sz);
20956 	bpf_dynptr_init(&sig_ptr, sig, BPF_DYNPTR_TYPE_LOCAL, 0,
20957 			attr->signature_size);
20958 
20959 	err = bpf_verify_pkcs7_signature((struct bpf_dynptr *)&data_ptr,
20960 					 (struct bpf_dynptr *)&sig_ptr, key);
20961 	if (err) {
20962 		verbose(env, "signature verification failed: %d\n", err);
20963 	} else {
20964 		verbose(env, "signature verification passed\n");
20965 		prog->aux->sig.keyring_serial = bpf_key_serial(key);
20966 		prog->aux->sig.keyring_type = bpf_classify_keyring(attr->keyring_id);
20967 		prog->aux->sig.verdict = BPF_SIG_VERIFIED;
20968 	}
20969 out:
20970 	kvfree(data);
20971 	bpf_key_put(key);
20972 	kvfree(sig);
20973 	return err;
20974 }
20975 
bpf_check(struct bpf_prog ** prog,union bpf_attr * attr,bpfptr_t uattr,struct bpf_log_attr * attr_log)20976 int bpf_check(struct bpf_prog **prog, union bpf_attr *attr, bpfptr_t uattr,
20977 	      struct bpf_log_attr *attr_log)
20978 {
20979 	u64 start_time = ktime_get_ns();
20980 	struct bpf_verifier_env *env;
20981 	int i, len, ret = -EINVAL, err;
20982 	bool is_priv;
20983 
20984 	BTF_TYPE_EMIT(enum bpf_features);
20985 
20986 	/* no program is valid */
20987 	if (ARRAY_SIZE(bpf_verifier_ops) == 0)
20988 		return -EINVAL;
20989 
20990 	/* 'struct bpf_verifier_env' can be global, but since it's not small,
20991 	 * allocate/free it every time bpf_check() is called
20992 	 */
20993 	env = kvzalloc_obj(struct bpf_verifier_env, GFP_KERNEL_ACCOUNT);
20994 	if (!env)
20995 		return -ENOMEM;
20996 
20997 	env->bt.env = env;
20998 	env->prog = *prog;
20999 	env->ops = bpf_verifier_ops[env->prog->type];
21000 
21001 	env->allow_ptr_leaks = bpf_allow_ptr_leaks(env->prog->aux->token);
21002 	env->allow_uninit_stack = bpf_allow_uninit_stack(env->prog->aux->token);
21003 	env->bypass_spec_v1 = bpf_bypass_spec_v1(env->prog->aux->token);
21004 	env->bypass_spec_v4 = bpf_bypass_spec_v4(env->prog->aux->token);
21005 	env->bpf_capable = is_priv = bpf_token_capable(env->prog->aux->token, CAP_BPF);
21006 	env->signature = attr->signature;
21007 
21008 	/* user could have requested verbose verifier output
21009 	 * and supplied buffer to store the verification trace
21010 	 */
21011 	ret = bpf_vlog_init(&env->log, attr_log->level, attr_log->ubuf, attr_log->size);
21012 	if (ret)
21013 		goto err_free_env;
21014 	ret = bpf_diag_init(env);
21015 	if (ret)
21016 		goto err_prep;
21017 	if (env->signature) {
21018 		ret = bpf_prog_calc_tag(env->prog);
21019 		if (ret < 0)
21020 			goto err_prep;
21021 	}
21022 
21023 	ret = process_fd_array(env, attr, uattr);
21024 	if (ret)
21025 		goto err_prep;
21026 
21027 	if (env->signature) {
21028 		ret = bpf_prog_verify_signature(env, attr, uattr.is_kernel);
21029 		if (ret)
21030 			goto err_prep;
21031 	}
21032 
21033 	ret = security_bpf_prog_load(env->prog, attr, env->prog->aux->token,
21034 				     uattr.is_kernel);
21035 	if (ret)
21036 		goto err_prep;
21037 
21038 	bpf_get_btf_vmlinux();
21039 
21040 	/* Serialize verification of unprivileged programs. */
21041 	if (!is_priv)
21042 		mutex_lock(&bpf_verifier_lock);
21043 
21044 	len = env->insn_aux_data_len = env->prog->len;
21045 	env->insn_aux_data =
21046 		__vmalloc(array_size(sizeof(struct bpf_insn_aux_data), len),
21047 			  GFP_KERNEL_ACCOUNT | __GFP_ZERO);
21048 	ret = -ENOMEM;
21049 	if (!env->insn_aux_data)
21050 		goto skip_full_check;
21051 	for (i = 0; i < len; i++)
21052 		env->insn_aux_data[i].orig_idx = i;
21053 	env->succ = bpf_iarray_realloc(NULL, 2);
21054 	if (!env->succ)
21055 		goto skip_full_check;
21056 
21057 	mark_verifier_state_clean(env);
21058 
21059 	if (IS_ERR(btf_vmlinux)) {
21060 		/* Either gcc or pahole or kernel are broken. */
21061 		verbose(env, "in-kernel BTF is malformed\n");
21062 		ret = PTR_ERR(btf_vmlinux);
21063 		goto skip_full_check;
21064 	}
21065 
21066 	env->strict_alignment = !!(attr->prog_flags & BPF_F_STRICT_ALIGNMENT);
21067 	if (!IS_ENABLED(CONFIG_HAVE_EFFICIENT_UNALIGNED_ACCESS))
21068 		env->strict_alignment = true;
21069 	if (attr->prog_flags & BPF_F_ANY_ALIGNMENT)
21070 		env->strict_alignment = false;
21071 
21072 	if (is_priv)
21073 		env->test_state_freq = attr->prog_flags & BPF_F_TEST_STATE_FREQ;
21074 	env->test_reg_invariants = attr->prog_flags & BPF_F_TEST_REG_INVARIANTS;
21075 
21076 	env->explored_states = kvzalloc_objs(struct list_head,
21077 					     state_htab_size(env),
21078 					     GFP_KERNEL_ACCOUNT);
21079 	ret = -ENOMEM;
21080 	if (!env->explored_states)
21081 		goto skip_full_check;
21082 
21083 	for (i = 0; i < state_htab_size(env); i++)
21084 		INIT_LIST_HEAD(&env->explored_states[i]);
21085 	INIT_LIST_HEAD(&env->free_list);
21086 
21087 	/* Prepare BTF and func_info needed to discover all subprograms. */
21088 	ret = bpf_prepare_btf_info(env, attr, uattr);
21089 	if (ret < 0)
21090 		goto skip_full_check;
21091 
21092 	/* Discover all subprograms before validating their layout and BTF. */
21093 	ret = add_subprogs(env);
21094 	if (ret < 0)
21095 		goto skip_full_check;
21096 
21097 	ret = check_subprogs(env);
21098 	if (ret < 0)
21099 		goto skip_full_check;
21100 
21101 	/* Validate BTF against the complete subprogram layout and apply CO-RE. */
21102 	ret = bpf_check_btf_info(env, attr, uattr);
21103 	if (ret < 0)
21104 		goto skip_full_check;
21105 
21106 	/* Validate instructions and resolve the program's referenced resources. */
21107 	ret = check_and_resolve_insns(env);
21108 	if (ret < 0)
21109 		goto skip_full_check;
21110 
21111 	/* Build kfunc prototypes after resolving program resources. */
21112 	ret = add_kfuncs(env);
21113 	if (ret < 0)
21114 		goto skip_full_check;
21115 
21116 	if (bpf_prog_is_offloaded(env->prog->aux)) {
21117 		ret = bpf_prog_offload_verifier_prep(env->prog);
21118 		if (ret)
21119 			goto skip_full_check;
21120 	}
21121 
21122 	ret = bpf_check_cfg(env);
21123 	if (ret < 0)
21124 		goto skip_full_check;
21125 
21126 	ret = bpf_compute_postorder(env);
21127 	if (ret < 0)
21128 		goto skip_full_check;
21129 
21130 	ret = bpf_stack_liveness_init(env);
21131 	if (ret)
21132 		goto skip_full_check;
21133 
21134 	ret = check_attach_btf_id(env);
21135 	if (ret)
21136 		goto skip_full_check;
21137 
21138 	ret = bpf_compute_const_regs(env);
21139 	if (ret < 0)
21140 		goto skip_full_check;
21141 
21142 	ret = bpf_prune_dead_branches(env);
21143 	if (ret < 0)
21144 		goto skip_full_check;
21145 
21146 	ret = sort_subprogs_topo(env);
21147 	if (ret < 0)
21148 		goto skip_full_check;
21149 
21150 	ret = bpf_compute_scc(env);
21151 	if (ret < 0)
21152 		goto skip_full_check;
21153 
21154 	ret = bpf_compute_live_registers(env);
21155 	if (ret < 0)
21156 		goto skip_full_check;
21157 
21158 	ret = mark_fastcall_patterns(env);
21159 	if (ret < 0)
21160 		goto skip_full_check;
21161 
21162 	ret = do_check_main(env);
21163 	ret = ret ?: do_check_subprogs(env);
21164 
21165 	if (ret == 0 && bpf_prog_is_offloaded(env->prog->aux))
21166 		ret = bpf_prog_offload_finalize(env);
21167 
21168 skip_full_check:
21169 	kvfree(env->explored_states);
21170 
21171 	/* might decrease stack depth, keep it before passes that
21172 	 * allocate additional slots.
21173 	 */
21174 	if (ret == 0)
21175 		ret = bpf_remove_fastcall_spills_fills(env);
21176 
21177 	if (ret == 0)
21178 		ret = check_max_stack_depth(env);
21179 
21180 	/* instruction rewrites happen after this point */
21181 	if (ret == 0)
21182 		ret = bpf_optimize_bpf_loop(env);
21183 
21184 	if (is_priv) {
21185 		if (ret == 0)
21186 			bpf_opt_hard_wire_dead_code_branches(env);
21187 		if (ret == 0)
21188 			ret = bpf_opt_remove_dead_code(env);
21189 		if (ret == 0)
21190 			ret = bpf_opt_remove_nops(env);
21191 	} else {
21192 		if (ret == 0)
21193 			sanitize_dead_code(env);
21194 	}
21195 
21196 	if (ret == 0)
21197 		/* program is valid, convert *(u32*)(ctx + off) accesses */
21198 		ret = bpf_convert_ctx_accesses(env);
21199 
21200 	if (ret == 0)
21201 		ret = bpf_do_misc_fixups(env);
21202 
21203 	/* do 32-bit optimization after insn patching has done so those patched
21204 	 * insns could be handled correctly.
21205 	 */
21206 	if (ret == 0 && !bpf_prog_is_offloaded(env->prog->aux)) {
21207 		ret = bpf_opt_subreg_zext_lo32_rnd_hi32(env, attr);
21208 		env->prog->aux->verifier_zext = bpf_jit_needs_zext() ? !ret
21209 								     : false;
21210 	}
21211 
21212 	if (ret == 0)
21213 		ret = bpf_fixup_call_args(env);
21214 
21215 	env->verification_time = ktime_get_ns() - start_time;
21216 	print_verification_stats(env);
21217 	env->prog->aux->verified_insns = env->insn_processed;
21218 
21219 	/* preserve original error even if log finalization is successful */
21220 	err = bpf_log_attr_finalize(attr_log, &env->log);
21221 	if (err)
21222 		ret = err;
21223 
21224 	if (ret)
21225 		goto err_release_maps;
21226 
21227 	if (env->used_map_cnt) {
21228 		/* if program passed verifier, update used_maps in bpf_prog_info */
21229 		env->prog->aux->used_maps = kmalloc_objs(env->used_maps[0],
21230 							 env->used_map_cnt,
21231 							 GFP_KERNEL_ACCOUNT);
21232 
21233 		if (!env->prog->aux->used_maps) {
21234 			ret = -ENOMEM;
21235 			goto err_release_maps;
21236 		}
21237 
21238 		memcpy(env->prog->aux->used_maps, env->used_maps,
21239 		       sizeof(env->used_maps[0]) * env->used_map_cnt);
21240 		env->prog->aux->used_map_cnt = env->used_map_cnt;
21241 	}
21242 	if (env->used_btf_cnt) {
21243 		/* if program passed verifier, update used_btfs in bpf_prog_aux */
21244 		env->prog->aux->used_btfs = kmalloc_objs(env->used_btfs[0],
21245 							 env->used_btf_cnt,
21246 							 GFP_KERNEL_ACCOUNT);
21247 		if (!env->prog->aux->used_btfs) {
21248 			ret = -ENOMEM;
21249 			goto err_release_maps;
21250 		}
21251 
21252 		memcpy(env->prog->aux->used_btfs, env->used_btfs,
21253 		       sizeof(env->used_btfs[0]) * env->used_btf_cnt);
21254 		env->prog->aux->used_btf_cnt = env->used_btf_cnt;
21255 	}
21256 	if (env->used_map_cnt || env->used_btf_cnt) {
21257 		/* program is valid. Convert pseudo bpf_ld_imm64 into generic
21258 		 * bpf_ld_imm64 instructions
21259 		 */
21260 		convert_pseudo_ld_imm64(env);
21261 	}
21262 
21263 	adjust_btf_func(env);
21264 
21265 	/* extension progs temporarily inherit the attach_type of their targets
21266 	   for verification purposes, so set it back to zero before returning
21267 	 */
21268 	if (env->prog->type == BPF_PROG_TYPE_EXT)
21269 		env->prog->expected_attach_type = 0;
21270 
21271 	env->prog = __bpf_prog_select_runtime(env, env->prog, &ret);
21272 
21273 err_release_maps:
21274 	if (ret)
21275 		release_insn_arrays(env);
21276 	if (!env->prog->aux->used_maps)
21277 		/* if we didn't copy map pointers into bpf_prog_info, release
21278 		 * them now. Otherwise free_used_maps() will release them.
21279 		 */
21280 		release_maps(env);
21281 	if (!env->prog->aux->used_btfs)
21282 		release_btfs(env);
21283 
21284 	*prog = env->prog;
21285 
21286 	module_put(env->attach_btf_mod);
21287 	if (!is_priv)
21288 		mutex_unlock(&bpf_verifier_lock);
21289 	goto err_free_env;
21290 err_prep:
21291 	err = bpf_log_attr_finalize(attr_log, &env->log);
21292 	if (err)
21293 		ret = err;
21294 	release_insn_arrays(env);
21295 	release_maps(env);
21296 	release_btfs(env);
21297 err_free_env:
21298 	if (env->insn_aux_data)
21299 		bpf_clear_insn_aux_data(env, 0, env->insn_aux_data_len);
21300 	vfree(env->insn_aux_data);
21301 	kvfree(env->fd_array);
21302 	bpf_stack_liveness_free(env);
21303 	kvfree(env->cfg.insn_postorder);
21304 	kvfree(env->scc_info);
21305 	kvfree(env->succ);
21306 	kvfree(env->gotox_tmp_buf);
21307 	bpf_diag_free(env);
21308 	kvfree(env);
21309 	return ret;
21310 }
21311